{
  "case_id": "capsule-2916503",
  "title": "Reproduce a published structural-equation analysis",
  "source_url": "https://doi.org/10.24433/CO.8235849.v1",
  "date": "2026-09-15",
  "model": "Sonnet 5",
  "host": "Claude Code 2.1.272",
  "arms": {
    "baseline": {
      "scientific_pass": true,
      "numerical_correct": 3,
      "numerical_total": 3,
      "figure_created": true,
      "execution_verified": true,
      "skill_activated": false,
      "cost_usd": 0.1161922999999998,
      "seconds": 131.288434291,
      "skill_available": false
    },
    "checklist": {
      "scientific_pass": true,
      "numerical_correct": 3,
      "numerical_total": 3,
      "figure_created": true,
      "execution_verified": true,
      "skill_activated": false,
      "cost_usd": 0.16070990000000007,
      "seconds": 258.91085725,
      "skill_available": false
    },
    "skill": {
      "scientific_pass": true,
      "numerical_correct": 3,
      "numerical_total": 3,
      "figure_created": true,
      "execution_verified": true,
      "skill_activated": false,
      "cost_usd": 0.11626410000000043,
      "seconds": 137.144009542,
      "skill_available": true
    }
  },
  "total_cost_usd": 0.3931663000000003,
  "campaign_total_cost_usd": 2.840126800000001,
  "remaining_budget_usd": 2.159873199999999,
  "interpretation": "All three setups reproduced the same three values and figure. The installed skill was available to the agent, but it was never opened. Execution and independent grading work; this run did not exercise the skill’s guidance and provides no evidence that it improves scientific work.",
  "scope": "One supplied-code teaching case with one attempt per setup. Three values come from one model fit; they are not three independent cases. The checklist includes a task-specific residual-variance reminder. All setups retain the native host’s built-in skills.",
  "skill": {
    "repository": "https://github.com/K-Dense-AI/scientific-agent-skills",
    "revision": "330c8e764435a731eff571e3efdda70b363d0792",
    "path": "skills/statistical-analysis",
    "files_unchanged": 7
  },
  "runtime": {
    "platform": "macOS osx-64 under Rosetta; original capsule used Linux",
    "R": "4.0.5",
    "lavaan": "0.6-9",
    "semPlot": "1.1.2"
  },
  "reported_values": {
    "Exam1 residual variance": 118.195,
    "Exam2 residual variance": 124.754,
    "Exam3 residual variance": 87.973
  },
  "cost_scope": "Measured benchmark model calls only; local compute, runtime setup, storage and assistant development are excluded.",
  "timing_scope": "Whole workflow elapsed time, including filesystem-search delays; one observation per setup, not a speed ranking.",
  "next_step": "Run explicit skill loading as a separate diagnostic. Then develop independent cases that require method choice, keeping natural discovery as the primary installed-skill condition."
}
