{
  "schema_version": "catalog-1.1",
  "id": "dfir-staged-0.6.0",
  "generated_at": "2026-09-27T14:32:25.917914Z",
  "status": "catalog",
  "dataset_version": "0.6.0",
  "protocol_version": "0.6.0",
  "methodology_version": "0.6.0",
  "campaign_hash": null,
  "development_cases": 48,
  "evaluation_cases": 0,
  "scenario_families": 16,
  "tests": [
    {
      "id": "DFIR-DEV-301",
      "title": "Remote management investigation",
      "split": "development",
      "group": "remote_management",
      "difficulty": "standard",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "edr",
        "proxy",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 3 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: Initial alert",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Factual reconstruction",
          "question": "Subject: host. Value: full image path of the process that started the alerted activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-011",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-012",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: the exact observed command, task action or API operation that explains the activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "basic",
      "checkpoints": 3
    },
    {
      "id": "DFIR-DEV-302",
      "title": "Remote management investigation",
      "split": "development",
      "group": "remote_management",
      "difficulty": "standard",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "proxy",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 3 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: Initial alert",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Factual reconstruction",
          "question": "Subject: host. Value: full image path of the process that started the alerted activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-011",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-012",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: the exact observed command, task action or API operation that explains the activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "basic",
      "checkpoints": 3
    },
    {
      "id": "DFIR-DEV-303",
      "title": "Remote management investigation",
      "split": "development",
      "group": "remote_management",
      "difficulty": "standard",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "proxy",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 3 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: Initial alert",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Factual reconstruction",
          "question": "Subject: host. Value: full image path of the process that started the alerted activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-011",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-012",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: the exact observed command, task action or API operation that explains the activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "basic",
      "checkpoints": 3
    },
    {
      "id": "DFIR-DEV-304",
      "title": "Scripted collection investigation",
      "split": "development",
      "group": "scripted_collection",
      "difficulty": "standard",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "powershell",
        "proxy",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 3 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: Initial alert",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-011",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: the exact observed command, task action or API operation that explains the activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "basic",
      "checkpoints": 3
    },
    {
      "id": "DFIR-DEV-305",
      "title": "Scripted collection investigation",
      "split": "development",
      "group": "scripted_collection",
      "difficulty": "standard",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "powershell",
        "proxy",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 3 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: Initial alert",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-011",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: the exact observed command, task action or API operation that explains the activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "basic",
      "checkpoints": 3
    },
    {
      "id": "DFIR-DEV-306",
      "title": "Scripted collection investigation",
      "split": "development",
      "group": "scripted_collection",
      "difficulty": "standard",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "powershell",
        "proxy",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 3 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: Initial alert",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-011",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: the exact observed command, task action or API operation that explains the activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "basic",
      "checkpoints": 3
    },
    {
      "id": "DFIR-DEV-307",
      "title": "Service lateral investigation",
      "split": "development",
      "group": "service_lateral",
      "difficulty": "standard",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "proxy",
        "security",
        "sysmon",
        "system"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 3 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: Initial alert",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: IPv4 address the alerted activity originated from.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-011",
          "stage": "Factual reconstruction",
          "question": "Subject: host. Value: hostname that initiated the alerted remote activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-012",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "basic",
      "checkpoints": 3
    },
    {
      "id": "DFIR-DEV-308",
      "title": "Service lateral investigation",
      "split": "development",
      "group": "service_lateral",
      "difficulty": "standard",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "proxy",
        "security",
        "sysmon",
        "system"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 3 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: Initial alert",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: IPv4 address the alerted activity originated from.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-011",
          "stage": "Factual reconstruction",
          "question": "Subject: host. Value: hostname that initiated the alerted remote activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-012",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "basic",
      "checkpoints": 3
    },
    {
      "id": "DFIR-DEV-309",
      "title": "Service lateral investigation",
      "split": "development",
      "group": "service_lateral",
      "difficulty": "standard",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "proxy",
        "security",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 3 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: Initial alert",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: IPv4 address the alerted activity originated from.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-011",
          "stage": "Factual reconstruction",
          "question": "Subject: host. Value: hostname that initiated the alerted remote activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-012",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "basic",
      "checkpoints": 3
    },
    {
      "id": "DFIR-DEV-310",
      "title": "Wmi execution investigation",
      "split": "development",
      "group": "wmi_execution",
      "difficulty": "standard",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "proxy",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 3 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: Initial alert",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Factual reconstruction",
          "question": "Subject: host. Value: full image path of the process that started the alerted activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-011",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-012",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: the exact observed command, task action or API operation that explains the activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "basic",
      "checkpoints": 3
    },
    {
      "id": "DFIR-DEV-311",
      "title": "Wmi execution investigation",
      "split": "development",
      "group": "wmi_execution",
      "difficulty": "standard",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "proxy",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 3 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: Initial alert",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Factual reconstruction",
          "question": "Subject: host. Value: full image path of the process that started the alerted activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-011",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-012",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: the exact observed command, task action or API operation that explains the activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "basic",
      "checkpoints": 3
    },
    {
      "id": "DFIR-DEV-312",
      "title": "Wmi execution investigation",
      "split": "development",
      "group": "wmi_execution",
      "difficulty": "standard",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "proxy",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 3 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: Initial alert",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Factual reconstruction",
          "question": "Subject: host. Value: full image path of the process that started the alerted activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-011",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-012",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: the exact observed command, task action or API operation that explains the activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "basic",
      "checkpoints": 3
    },
    {
      "id": "DFIR-DEV-313",
      "title": "Scheduled persistence investigation",
      "split": "development",
      "group": "scheduled_persistence",
      "difficulty": "standard",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "proxy",
        "security",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 3 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: Initial alert",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-011",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: the exact observed command, task action or API operation that explains the activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "basic",
      "checkpoints": 3
    },
    {
      "id": "DFIR-DEV-314",
      "title": "Scheduled persistence investigation",
      "split": "development",
      "group": "scheduled_persistence",
      "difficulty": "standard",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "proxy",
        "security",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 3 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: Initial alert",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-011",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: the exact observed command, task action or API operation that explains the activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "basic",
      "checkpoints": 3
    },
    {
      "id": "DFIR-DEV-315",
      "title": "Scheduled persistence investigation",
      "split": "development",
      "group": "scheduled_persistence",
      "difficulty": "standard",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "proxy",
        "security",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 3 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: Initial alert",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-011",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: the exact observed command, task action or API operation that explains the activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "basic",
      "checkpoints": 3
    },
    {
      "id": "DFIR-DEV-316",
      "title": "Webshell investigation",
      "split": "development",
      "group": "webshell",
      "difficulty": "standard",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "iis",
        "proxy",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 3 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: Initial alert",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-011",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: the exact observed command, task action or API operation that explains the activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "basic",
      "checkpoints": 3
    },
    {
      "id": "DFIR-DEV-317",
      "title": "Webshell investigation",
      "split": "development",
      "group": "webshell",
      "difficulty": "standard",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "iis",
        "proxy",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 3 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: Initial alert",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-011",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: the exact observed command, task action or API operation that explains the activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "basic",
      "checkpoints": 3
    },
    {
      "id": "DFIR-DEV-318",
      "title": "Webshell investigation",
      "split": "development",
      "group": "webshell",
      "difficulty": "standard",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "iis",
        "proxy",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 3 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: Initial alert",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "basic",
      "checkpoints": 3
    },
    {
      "id": "DFIR-DEV-319",
      "title": "Browser credentials investigation",
      "split": "development",
      "group": "browser_credentials",
      "difficulty": "standard",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "edr",
        "proxy",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 3 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: Initial alert",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Factual reconstruction",
          "question": "Subject: host. Value: full image path of the process that started the alerted activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-011",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-012",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: the exact observed command, task action or API operation that explains the activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "basic",
      "checkpoints": 3
    },
    {
      "id": "DFIR-DEV-320",
      "title": "Browser credentials investigation",
      "split": "development",
      "group": "browser_credentials",
      "difficulty": "standard",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "proxy",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 3 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: Initial alert",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Factual reconstruction",
          "question": "Subject: host. Value: full image path of the process that started the alerted activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-011",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-012",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: the exact observed command, task action or API operation that explains the activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "basic",
      "checkpoints": 3
    },
    {
      "id": "DFIR-DEV-321",
      "title": "Browser credentials investigation",
      "split": "development",
      "group": "browser_credentials",
      "difficulty": "standard",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "proxy",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 3 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: Initial alert",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Factual reconstruction",
          "question": "Subject: host. Value: full image path of the process that started the alerted activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-011",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-012",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: the exact observed command, task action or API operation that explains the activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "basic",
      "checkpoints": 3
    },
    {
      "id": "DFIR-DEV-322",
      "title": "Data transfer investigation",
      "split": "development",
      "group": "data_transfer",
      "difficulty": "standard",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "edr",
        "proxy",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 3 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: Initial alert",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Factual reconstruction",
          "question": "Subject: host. Value: full image path of the process that started the alerted activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-011",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-012",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: the exact observed command, task action or API operation that explains the activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "basic",
      "checkpoints": 3
    },
    {
      "id": "DFIR-DEV-323",
      "title": "Data transfer investigation",
      "split": "development",
      "group": "data_transfer",
      "difficulty": "standard",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "proxy",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 3 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: Initial alert",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Factual reconstruction",
          "question": "Subject: host. Value: full image path of the process that started the alerted activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-011",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-012",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: the exact observed command, task action or API operation that explains the activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "basic",
      "checkpoints": 3
    },
    {
      "id": "DFIR-DEV-324",
      "title": "Data transfer investigation",
      "split": "development",
      "group": "data_transfer",
      "difficulty": "standard",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "proxy",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 3 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: Initial alert",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Factual reconstruction",
          "question": "Subject: host. Value: full image path of the process that started the alerted activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-011",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-012",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: the exact observed command, task action or API operation that explains the activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "basic",
      "checkpoints": 3
    },
    {
      "id": "DFIR-DEV-325",
      "title": "Mailbox bec investigation",
      "split": "development",
      "group": "mailbox_bec",
      "difficulty": "standard",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "entra",
        "m365",
        "proxy",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 3 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: Initial alert",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-011",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: the exact observed command, task action or API operation that explains the activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "basic",
      "checkpoints": 3
    },
    {
      "id": "DFIR-DEV-326",
      "title": "Mailbox bec investigation",
      "split": "development",
      "group": "mailbox_bec",
      "difficulty": "standard",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "entra",
        "m365",
        "proxy",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 3 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: Initial alert",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-011",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: the exact observed command, task action or API operation that explains the activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "basic",
      "checkpoints": 3
    },
    {
      "id": "DFIR-DEV-327",
      "title": "Mailbox bec investigation",
      "split": "development",
      "group": "mailbox_bec",
      "difficulty": "standard",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "entra",
        "proxy",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 3 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: Initial alert",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "basic",
      "checkpoints": 3
    },
    {
      "id": "DFIR-DEV-328",
      "title": "Cloud app abuse investigation",
      "split": "development",
      "group": "cloud_app_abuse",
      "difficulty": "standard",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "entra",
        "graph",
        "proxy",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 3 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: Initial alert",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-011",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: the exact observed command, task action or API operation that explains the activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "basic",
      "checkpoints": 3
    },
    {
      "id": "DFIR-DEV-329",
      "title": "Cloud app abuse investigation",
      "split": "development",
      "group": "cloud_app_abuse",
      "difficulty": "standard",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "entra",
        "graph",
        "proxy",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 3 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: Initial alert",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-011",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: the exact observed command, task action or API operation that explains the activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "basic",
      "checkpoints": 3
    },
    {
      "id": "DFIR-DEV-330",
      "title": "Cloud app abuse investigation",
      "split": "development",
      "group": "cloud_app_abuse",
      "difficulty": "standard",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "entra",
        "proxy",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 3 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: Initial alert",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "basic",
      "checkpoints": 3
    },
    {
      "id": "DFIR-DEV-331",
      "title": "Unwanted extension investigation",
      "split": "development",
      "group": "unwanted_extension",
      "difficulty": "standard",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "browser",
        "collector",
        "edr",
        "proxy",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 3 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: Initial alert",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-011",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: the exact observed command, task action or API operation that explains the activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "basic",
      "checkpoints": 3
    },
    {
      "id": "DFIR-DEV-332",
      "title": "Unwanted extension investigation",
      "split": "development",
      "group": "unwanted_extension",
      "difficulty": "standard",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "browser",
        "collector",
        "proxy",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 3 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: Initial alert",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "basic",
      "checkpoints": 3
    },
    {
      "id": "DFIR-DEV-333",
      "title": "Unwanted extension investigation",
      "split": "development",
      "group": "unwanted_extension",
      "difficulty": "standard",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "browser",
        "collector",
        "proxy",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 3 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: Initial alert",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "basic",
      "checkpoints": 3
    },
    {
      "id": "DFIR-DEV-334",
      "title": "Installer sideload investigation",
      "split": "development",
      "group": "installer_sideload",
      "difficulty": "standard",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "proxy",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 3 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: Initial alert",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Factual reconstruction",
          "question": "Subject: host. Value: full image path of the process that started the alerted activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-011",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-012",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: the exact observed command, task action or API operation that explains the activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "basic",
      "checkpoints": 3
    },
    {
      "id": "DFIR-DEV-335",
      "title": "Installer sideload investigation",
      "split": "development",
      "group": "installer_sideload",
      "difficulty": "standard",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "proxy",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 3 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: Initial alert",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Factual reconstruction",
          "question": "Subject: host. Value: full image path of the process that started the alerted activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-011",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-012",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: the exact observed command, task action or API operation that explains the activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "basic",
      "checkpoints": 3
    },
    {
      "id": "DFIR-DEV-336",
      "title": "Installer sideload investigation",
      "split": "development",
      "group": "installer_sideload",
      "difficulty": "standard",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "proxy",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 3 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: Initial alert",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: Initial alert",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: Correlated evidence",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: Effects and coverage",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Factual reconstruction",
          "question": "Subject: host. Value: full image path of the process that started the alerted activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-011",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-012",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: the exact observed command, task action or API operation that explains the activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "basic",
      "checkpoints": 3
    },
    {
      "id": "DFIR-DEV-337",
      "title": "Loader to ransomware investigation",
      "split": "development",
      "group": "loader_to_ransomware",
      "difficulty": "advanced",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "edr",
        "proxy",
        "security",
        "sysmon",
        "system"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 6 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: initial access",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: initial access",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: initial access",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: discovery",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: discovery",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: discovery",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: lateral dc",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: lateral dc",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: lateral dc",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Checkpoint 4: lateral fileserver",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-011",
          "stage": "Checkpoint 4: lateral fileserver",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-012",
          "stage": "Checkpoint 4: lateral fileserver",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-013",
          "stage": "Checkpoint 5: exfiltration",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-014",
          "stage": "Checkpoint 5: exfiltration",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-015",
          "stage": "Checkpoint 5: exfiltration",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-016",
          "stage": "Checkpoint 6: impact",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-017",
          "stage": "Checkpoint 6: impact",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-018",
          "stage": "Checkpoint 6: impact",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-019",
          "stage": "Factual reconstruction",
          "question": "Subject: host. Value: full image path of the process that started the alerted activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-020",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-021",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: the exact observed command, task action or API operation that explains the activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-022",
          "stage": "Factual reconstruction",
          "question": "Subject: incident. Value: observed entry artifact (URL, image path or source IP), with observed_at.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-023",
          "stage": "Factual reconstruction",
          "question": "Subject: incident. Value: first host where attacker code ran.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-024",
          "stage": "Factual reconstruction",
          "question": "Subject: host. Value: image path of the first attacker process, with observed_at.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-025",
          "stage": "Factual reconstruction",
          "question": "Subject: source host. Value: host reached by lateral movement, with observed_at of the first execution there.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-026",
          "stage": "Factual reconstruction",
          "question": "Subject: source host. Value: host reached by lateral movement, with observed_at of the first execution there.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-027",
          "stage": "Factual reconstruction",
          "question": "Subject: incident. Value: domain that received stolen data.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-028",
          "stage": "Factual reconstruction",
          "question": "Subject: incident. Value: host where destructive impact began, with observed_at.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "intrusion",
      "checkpoints": 6
    },
    {
      "id": "DFIR-DEV-338",
      "title": "Loader to ransomware investigation",
      "split": "development",
      "group": "loader_to_ransomware",
      "difficulty": "advanced",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "proxy",
        "security",
        "sysmon",
        "system"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 6 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: initial access",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: initial access",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: initial access",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: discovery",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: discovery",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: discovery",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: lateral dc",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: lateral dc",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: lateral dc",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Checkpoint 4: lateral fileserver",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-011",
          "stage": "Checkpoint 4: lateral fileserver",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-012",
          "stage": "Checkpoint 4: lateral fileserver",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-013",
          "stage": "Checkpoint 5: exfiltration",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-014",
          "stage": "Checkpoint 5: exfiltration",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-015",
          "stage": "Checkpoint 5: exfiltration",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-016",
          "stage": "Checkpoint 6: impact",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-017",
          "stage": "Checkpoint 6: impact",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-018",
          "stage": "Checkpoint 6: impact",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-019",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: IPv4 address the alerted activity originated from.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-020",
          "stage": "Factual reconstruction",
          "question": "Subject: host. Value: hostname that initiated the alerted remote activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-021",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "intrusion",
      "checkpoints": 6
    },
    {
      "id": "DFIR-DEV-339",
      "title": "Loader to ransomware investigation",
      "split": "development",
      "group": "loader_to_ransomware",
      "difficulty": "advanced",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "proxy",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 6 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: initial access",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: initial access",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: initial access",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: discovery",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: discovery",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: discovery",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: lateral dc",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: lateral dc",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: lateral dc",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Checkpoint 4: lateral fileserver",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-011",
          "stage": "Checkpoint 4: lateral fileserver",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-012",
          "stage": "Checkpoint 4: lateral fileserver",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-013",
          "stage": "Checkpoint 5: exfiltration",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-014",
          "stage": "Checkpoint 5: exfiltration",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-015",
          "stage": "Checkpoint 5: exfiltration",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-016",
          "stage": "Checkpoint 6: impact",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-017",
          "stage": "Checkpoint 6: impact",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-018",
          "stage": "Checkpoint 6: impact",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-019",
          "stage": "Factual reconstruction",
          "question": "Subject: host. Value: full image path of the process that started the alerted activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-020",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-021",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: the exact observed command, task action or API operation that explains the activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "intrusion",
      "checkpoints": 6
    },
    {
      "id": "DFIR-DEV-340",
      "title": "Exposed web app investigation",
      "split": "development",
      "group": "exposed_web_app",
      "difficulty": "advanced",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "iis",
        "proxy",
        "security",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 5 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: initial access",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: initial access",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: initial access",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: credential access",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: credential access",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: credential access",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: lateral fileserver",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: lateral fileserver",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: lateral fileserver",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Checkpoint 4: lateral dc",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-011",
          "stage": "Checkpoint 4: lateral dc",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-012",
          "stage": "Checkpoint 4: lateral dc",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-013",
          "stage": "Checkpoint 5: exfiltration",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-014",
          "stage": "Checkpoint 5: exfiltration",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-015",
          "stage": "Checkpoint 5: exfiltration",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-016",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-017",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: the exact observed command, task action or API operation that explains the activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-018",
          "stage": "Factual reconstruction",
          "question": "Subject: incident. Value: observed entry artifact (URL, image path or source IP), with observed_at.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-019",
          "stage": "Factual reconstruction",
          "question": "Subject: incident. Value: first host where attacker code ran.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-020",
          "stage": "Factual reconstruction",
          "question": "Subject: host. Value: image path of the first attacker process, with observed_at.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-021",
          "stage": "Factual reconstruction",
          "question": "Subject: source host. Value: host reached by lateral movement, with observed_at of the first execution there.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-022",
          "stage": "Factual reconstruction",
          "question": "Subject: source host. Value: host reached by lateral movement, with observed_at of the first execution there.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-023",
          "stage": "Factual reconstruction",
          "question": "Subject: incident. Value: domain that received stolen data.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "intrusion",
      "checkpoints": 5
    },
    {
      "id": "DFIR-DEV-341",
      "title": "Exposed web app investigation",
      "split": "development",
      "group": "exposed_web_app",
      "difficulty": "advanced",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "iis",
        "proxy",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 5 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: initial access",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: initial access",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: initial access",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: credential access",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: credential access",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: credential access",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: lateral fileserver",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: lateral fileserver",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: lateral fileserver",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Checkpoint 4: lateral dc",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-011",
          "stage": "Checkpoint 4: lateral dc",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-012",
          "stage": "Checkpoint 4: lateral dc",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-013",
          "stage": "Checkpoint 5: exfiltration",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-014",
          "stage": "Checkpoint 5: exfiltration",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-015",
          "stage": "Checkpoint 5: exfiltration",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-016",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-017",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: the exact observed command, task action or API operation that explains the activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "intrusion",
      "checkpoints": 5
    },
    {
      "id": "DFIR-DEV-342",
      "title": "Exposed web app investigation",
      "split": "development",
      "group": "exposed_web_app",
      "difficulty": "advanced",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "proxy",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 5 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: initial access",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: initial access",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: initial access",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: credential access",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: credential access",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: credential access",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: lateral fileserver",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: lateral fileserver",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: lateral fileserver",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Checkpoint 4: lateral dc",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-011",
          "stage": "Checkpoint 4: lateral dc",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-012",
          "stage": "Checkpoint 4: lateral dc",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-013",
          "stage": "Checkpoint 5: exfiltration",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-014",
          "stage": "Checkpoint 5: exfiltration",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-015",
          "stage": "Checkpoint 5: exfiltration",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-016",
          "stage": "Factual reconstruction",
          "question": "Subject: host. Value: full image path of the process that started the alerted activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-017",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-018",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: the exact observed command, task action or API operation that explains the activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "intrusion",
      "checkpoints": 5
    },
    {
      "id": "DFIR-DEV-343",
      "title": "Fake browser update investigation",
      "split": "development",
      "group": "fake_browser_update",
      "difficulty": "advanced",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "edr",
        "proxy",
        "security",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 5 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: initial access",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: initial access",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: initial access",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: credential access",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: credential access",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: credential access",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: lateral backup",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: lateral backup",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: lateral backup",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Checkpoint 4: lateral fileserver",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-011",
          "stage": "Checkpoint 4: lateral fileserver",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-012",
          "stage": "Checkpoint 4: lateral fileserver",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-013",
          "stage": "Checkpoint 5: impact",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-014",
          "stage": "Checkpoint 5: impact",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-015",
          "stage": "Checkpoint 5: impact",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-016",
          "stage": "Factual reconstruction",
          "question": "Subject: host. Value: full image path of the process that started the alerted activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-017",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-018",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: the exact observed command, task action or API operation that explains the activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-019",
          "stage": "Factual reconstruction",
          "question": "Subject: incident. Value: observed entry artifact (URL, image path or source IP), with observed_at.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-020",
          "stage": "Factual reconstruction",
          "question": "Subject: incident. Value: first host where attacker code ran.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-021",
          "stage": "Factual reconstruction",
          "question": "Subject: host. Value: image path of the first attacker process, with observed_at.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-022",
          "stage": "Factual reconstruction",
          "question": "Subject: source host. Value: host reached by lateral movement, with observed_at of the first execution there.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-023",
          "stage": "Factual reconstruction",
          "question": "Subject: source host. Value: host reached by lateral movement, with observed_at of the first execution there.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-024",
          "stage": "Factual reconstruction",
          "question": "Subject: incident. Value: host where destructive impact began, with observed_at.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "intrusion",
      "checkpoints": 5
    },
    {
      "id": "DFIR-DEV-344",
      "title": "Fake browser update investigation",
      "split": "development",
      "group": "fake_browser_update",
      "difficulty": "advanced",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "proxy",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 5 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: initial access",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: initial access",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: initial access",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: credential access",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: credential access",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: credential access",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: lateral backup",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: lateral backup",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: lateral backup",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Checkpoint 4: lateral fileserver",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-011",
          "stage": "Checkpoint 4: lateral fileserver",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-012",
          "stage": "Checkpoint 4: lateral fileserver",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-013",
          "stage": "Checkpoint 5: impact",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-014",
          "stage": "Checkpoint 5: impact",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-015",
          "stage": "Checkpoint 5: impact",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-016",
          "stage": "Factual reconstruction",
          "question": "Subject: host. Value: full image path of the process that started the alerted activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-017",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-018",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: the exact observed command, task action or API operation that explains the activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "intrusion",
      "checkpoints": 5
    },
    {
      "id": "DFIR-DEV-345",
      "title": "Fake browser update investigation",
      "split": "development",
      "group": "fake_browser_update",
      "difficulty": "advanced",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "proxy",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 5 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: initial access",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: initial access",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: initial access",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: credential access",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: credential access",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: credential access",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: lateral backup",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: lateral backup",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: lateral backup",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Checkpoint 4: lateral fileserver",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-011",
          "stage": "Checkpoint 4: lateral fileserver",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-012",
          "stage": "Checkpoint 4: lateral fileserver",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-013",
          "stage": "Checkpoint 5: impact",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-014",
          "stage": "Checkpoint 5: impact",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-015",
          "stage": "Checkpoint 5: impact",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-016",
          "stage": "Factual reconstruction",
          "question": "Subject: host. Value: full image path of the process that started the alerted activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-017",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-018",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: the exact observed command, task action or API operation that explains the activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "intrusion",
      "checkpoints": 5
    },
    {
      "id": "DFIR-DEV-346",
      "title": "External rdp bruteforce investigation",
      "split": "development",
      "group": "external_rdp_bruteforce",
      "difficulty": "advanced",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "edr",
        "proxy",
        "security",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 6 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: initial access",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: initial access",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: initial access",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: persistence",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: persistence",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: persistence",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: credential access",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: credential access",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: credential access",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Checkpoint 4: lateral share",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-011",
          "stage": "Checkpoint 4: lateral share",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-012",
          "stage": "Checkpoint 4: lateral share",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-013",
          "stage": "Checkpoint 5: lateral backup",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-014",
          "stage": "Checkpoint 5: lateral backup",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-015",
          "stage": "Checkpoint 5: lateral backup",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-016",
          "stage": "Checkpoint 6: impact",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-017",
          "stage": "Checkpoint 6: impact",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-018",
          "stage": "Checkpoint 6: impact",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-019",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: IPv4 address the alerted activity originated from.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-020",
          "stage": "Factual reconstruction",
          "question": "Subject: host. Value: hostname that initiated the alerted remote activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-021",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-022",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: the exact observed command, task action or API operation that explains the activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-023",
          "stage": "Factual reconstruction",
          "question": "Subject: incident. Value: observed entry artifact (URL, image path or source IP), with observed_at.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-024",
          "stage": "Factual reconstruction",
          "question": "Subject: incident. Value: first host where attacker code ran.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-025",
          "stage": "Factual reconstruction",
          "question": "Subject: host. Value: image path of the first attacker process, with observed_at.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-026",
          "stage": "Factual reconstruction",
          "question": "Subject: source host. Value: host reached by lateral movement, with observed_at of the first execution there.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-027",
          "stage": "Factual reconstruction",
          "question": "Subject: source host. Value: host reached by lateral movement, with observed_at of the first execution there.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-028",
          "stage": "Factual reconstruction",
          "question": "Subject: incident. Value: host where destructive impact began, with observed_at.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "intrusion",
      "checkpoints": 6
    },
    {
      "id": "DFIR-DEV-347",
      "title": "External rdp bruteforce investigation",
      "split": "development",
      "group": "external_rdp_bruteforce",
      "difficulty": "advanced",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "proxy",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 6 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: initial access",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: initial access",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: initial access",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: persistence",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: persistence",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: persistence",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: credential access",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: credential access",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: credential access",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Checkpoint 4: lateral share",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-011",
          "stage": "Checkpoint 4: lateral share",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-012",
          "stage": "Checkpoint 4: lateral share",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-013",
          "stage": "Checkpoint 5: lateral backup",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-014",
          "stage": "Checkpoint 5: lateral backup",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-015",
          "stage": "Checkpoint 5: lateral backup",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-016",
          "stage": "Checkpoint 6: impact",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-017",
          "stage": "Checkpoint 6: impact",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-018",
          "stage": "Checkpoint 6: impact",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-019",
          "stage": "Factual reconstruction",
          "question": "Subject: host. Value: full image path of the process that started the alerted activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-020",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-021",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: the exact observed command, task action or API operation that explains the activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "intrusion",
      "checkpoints": 6
    },
    {
      "id": "DFIR-DEV-348",
      "title": "External rdp bruteforce investigation",
      "split": "development",
      "group": "external_rdp_bruteforce",
      "difficulty": "advanced",
      "summary": "Investigate the alert, distinguish competing explanations, trace scope and cite the indicators that matter.",
      "evidence_sources": [
        "collector",
        "proxy",
        "sysmon"
      ],
      "assessments": [],
      "capabilities": [
        "Competing malicious and benign explanations",
        "Verdicts at 6 evidence checkpoints",
        "Evidence-backed hypothesis updates",
        "Final verdict and observed scope",
        "Bounded evidence queries",
        "Compromise versus unwanted software"
      ],
      "checks": [
        {
          "id": "C-001",
          "stage": "Checkpoint 1: initial access",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-002",
          "stage": "Checkpoint 1: initial access",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-003",
          "stage": "Checkpoint 1: initial access",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-004",
          "stage": "Checkpoint 2: persistence",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-005",
          "stage": "Checkpoint 2: persistence",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-006",
          "stage": "Checkpoint 2: persistence",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-007",
          "stage": "Checkpoint 3: credential access",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-008",
          "stage": "Checkpoint 3: credential access",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-009",
          "stage": "Checkpoint 3: credential access",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-010",
          "stage": "Checkpoint 4: lateral share",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-011",
          "stage": "Checkpoint 4: lateral share",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-012",
          "stage": "Checkpoint 4: lateral share",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-013",
          "stage": "Checkpoint 5: lateral backup",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-014",
          "stage": "Checkpoint 5: lateral backup",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-015",
          "stage": "Checkpoint 5: lateral backup",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-016",
          "stage": "Checkpoint 6: impact",
          "question": "What conclusion is justified at this checkpoint?",
          "response": "Malicious / benign / insufficient evidence + citations",
          "grading": "Verdict accuracy and evidence-backed correctness are measured separately. Later records cannot support earlier decisions."
        },
        {
          "id": "C-017",
          "stage": "Checkpoint 6: impact",
          "question": "Does the evidence support or contradict the malicious explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-018",
          "stage": "Checkpoint 6: impact",
          "question": "Does the evidence support or contradict the benign explanation?",
          "response": "Supported / contradicted / unresolved + citations",
          "grading": "Correct hypothesis status with an accepted evidence set available at this checkpoint."
        },
        {
          "id": "C-019",
          "stage": "Factual reconstruction",
          "question": "Subject: host. Value: full image path of the process that started the alerted activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-020",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: collection channel name; observed_at is the last verified collection boundary for that channel.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "C-021",
          "stage": "Factual reconstruction",
          "question": "Subject: host or tenant. Value: the exact observed command, task action or API operation that explains the activity.",
          "response": "Structured finding with citations and timestamp where requested",
          "grading": "Typed value matching, accepted evidence sets and timestamp tolerance."
        },
        {
          "id": "A-001",
          "stage": "Evidence acquisition",
          "question": "Does the agent find the evidence needed to distinguish the explanations?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Credit each acquired evidence milestone using any accepted proof route. Report sufficiency separately from the final decision."
        },
        {
          "id": "A-002",
          "stage": "Evidence acquisition",
          "question": "How many queries are needed to find sufficient evidence?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count query attempts to the first complete accepted proof; no credit for never reaching sufficiency."
        },
        {
          "id": "A-003",
          "stage": "Evidence acquisition",
          "question": "How much unrelated or repeated evidence does the agent retrieve?",
          "response": "Recorded tool queries and returned evidence",
          "grading": "Count unrelated and duplicate records separately. These are efficiency measures, not a combined correctness score."
        },
        {
          "id": "IOC",
          "stage": "Final assessment",
          "question": "Which indicators belong to the scoped compromise?",
          "response": "Typed indicators, attribution and citations",
          "grading": "Normalized precision, recall and grounded recall; count benign lookalikes and indicators from another incident separately."
        },
        {
          "id": "ENV",
          "stage": "Final assessment",
          "question": "Does the agent recognize a different incident elsewhere?",
          "response": "Separate hosts, accounts, evidence and indicators",
          "grading": "Grounded incident recall and false separate findings; no effect on the scoped verdict."
        },
        {
          "id": "HOST",
          "stage": "Final assessment",
          "question": "Which hosts belong to this activity?",
          "response": "Scoped host set",
          "grading": "Host precision, recall and exact set match."
        }
      ],
      "tier": "intrusion",
      "checkpoints": 6
    }
  ],
  "limitations": [
    "No model campaign has been funded for 0.6. Scores from archived releases are not comparable.",
    "Related variants share a construction family. Scripted author routes test implementation, not model performance."
  ],
  "release_hash": "2a5858745432de623f4f7b51be589ca5e515314e5eb7e54277d9f78afd703824"
}
