{
  "schema_version": 1,
  "date": "2026-09-15",
  "model": "claude-sonnet-5",
  "arm_order": [
    "baseline",
    "checklist",
    "skill"
  ],
  "scope": "Four matched primary experiments in a fixed offline runner. Scores use different metrics and must not be averaged. Public development tasks; one run per condition.",
  "totals": {
    "tasks": 28,
    "assignments": 84,
    "source_clusters": 21,
    "cost_usd": 1.4741190999999998,
    "campaign_cost_usd": 2.4469605000000008
  },
  "experiments": [
    {
      "id": "verify",
      "title": "Check a scientific claim",
      "question": "Does critical-thinking guidance improve claim verification?",
      "metric": "Valid verdict accuracy",
      "unit": "accuracy",
      "tasks": 8,
      "source_clusters": 8,
      "arms": {
        "baseline": {
          "label": "No added skill",
          "score": 1.0,
          "attempts": 8,
          "completed": 8,
          "failed": 0,
          "cost_usd": 0.06199799999999999,
          "mean_seconds": 3.37701154175,
          "schema_valid_count": 8,
          "schema_observed_count": 8,
          "reference_reads": 0
        },
        "checklist": {
          "label": "Model + checklist",
          "score": 1.0,
          "attempts": 8,
          "completed": 8,
          "failed": 0,
          "cost_usd": 0.0469637,
          "mean_seconds": 3.057669666500006,
          "schema_valid_count": 8,
          "schema_observed_count": 8,
          "reference_reads": 0
        },
        "skill": {
          "label": "Model + skill",
          "score": 1.0,
          "attempts": 8,
          "completed": 8,
          "failed": 0,
          "cost_usd": 0.0610368,
          "mean_seconds": 3.221402922,
          "schema_valid_count": 8,
          "schema_observed_count": 8,
          "reference_reads": 0
        }
      },
      "skill": {
        "id": "scientific-critical-thinking",
        "kind": "upstream_package",
        "url": "https://github.com/K-Dense-AI/scientific-agent-skills/tree/330c8e764435a731eff571e3efdda70b363d0792/skills/scientific-critical-thinking",
        "revision": "330c8e764435a731eff571e3efdda70b363d0792",
        "usage": "SKILL.md supplied; no explicit reference-file reads recorded."
      },
      "interpretation": {
        "summary": "All three conditions matched the published verdict on all eight claims. Evidence quality remains unresolved.",
        "evidence_case_ids": [
          "verify-sf1049",
          "verify-sf508"
        ]
      },
      "limitations": [
        "Eight public claims; one run per condition, with no measured verdict headroom.",
        "Published rationale annotations omit some relevant evidence; label agreement is not independent scientific adjudication.",
        "The task supplies abstracts and tests only a narrow part of the critical-thinking package."
      ],
      "cases": [
        {
          "id": "verify-sf971",
          "label": "HPV screening sensitivity",
          "task_summary": "Assess whether HPV-based screening is more sensitive over follow-up than cytology for detecting grade-2 cervical lesions.",
          "source": {
            "dataset": "SciFact",
            "label": "HPV screening sensitivity",
            "cluster_id": "scifact-source-8e49e639da36e853",
            "url": "https://github.com/allenai/scifact",
            "synthetic": false
          },
          "arms": {
            "baseline": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.01938,
              "seconds": 6.429245708,
              "schema_valid": true,
              "outcome": "Recorded SUPPORT; matches the published claim label.",
              "metrics": {
                "verdict_accuracy": 1.0,
                "evidence_f1": 0.5,
                "citation_integrity": 1.0
              },
              "output": {
                "kind": "verdict",
                "value": {
                  "verdict": "SUPPORT",
                  "evidence": [
                    {
                      "doc_id": "9764256",
                      "sentence_id": 13
                    },
                    {
                      "doc_id": "9764256",
                      "sentence_id": 16
                    },
                    {
                      "doc_id": "27873158",
                      "sentence_id": 21
                    }
                  ]
                }
              }
            },
            "checklist": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.0180195,
              "seconds": 4.196742291,
              "schema_valid": true,
              "outcome": "Recorded SUPPORT; matches the published claim label.",
              "metrics": {
                "verdict_accuracy": 1.0,
                "evidence_f1": 0.4,
                "citation_integrity": 1.0
              },
              "output": {
                "kind": "verdict",
                "value": {
                  "verdict": "SUPPORT",
                  "evidence": [
                    {
                      "doc_id": "9764256",
                      "sentence_id": 0
                    },
                    {
                      "doc_id": "9764256",
                      "sentence_id": 13
                    },
                    {
                      "doc_id": "27873158",
                      "sentence_id": 0
                    },
                    {
                      "doc_id": "27873158",
                      "sentence_id": 15
                    }
                  ]
                }
              }
            },
            "skill": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.026382,
              "seconds": 3.504953541999999,
              "schema_valid": true,
              "outcome": "Recorded SUPPORT; matches the published claim label.",
              "metrics": {
                "verdict_accuracy": 1.0,
                "evidence_f1": 0.5,
                "citation_integrity": 1.0
              },
              "output": {
                "kind": "verdict",
                "value": {
                  "verdict": "SUPPORT",
                  "evidence": [
                    {
                      "doc_id": "9764256",
                      "sentence_id": 0
                    },
                    {
                      "doc_id": "9764256",
                      "sentence_id": 13
                    },
                    {
                      "doc_id": "27873158",
                      "sentence_id": 0
                    }
                  ]
                }
              }
            }
          },
          "evidence": {
            "run_id": "four-track-canary-v3",
            "task_id": "verify-sf971"
          }
        },
        {
          "id": "verify-sf1049",
          "label": "Tissue specificity of ribosome disorders",
          "task_summary": "Assess whether ribosome-related disorders produce largely nonspecific effects across cells and tissues.",
          "source": {
            "dataset": "SciFact",
            "label": "Tissue specificity of ribosome disorders",
            "cluster_id": "scifact-source-71d5408dbc571823",
            "url": "https://github.com/allenai/scifact",
            "synthetic": false
          },
          "arms": {
            "baseline": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.005542,
              "seconds": 2.2929334170000004,
              "schema_valid": true,
              "outcome": "Recorded CONTRADICT; matches the published claim label.",
              "metrics": {
                "verdict_accuracy": 1.0,
                "evidence_f1": 0.0,
                "citation_integrity": 1.0
              },
              "output": {
                "kind": "verdict",
                "value": {
                  "verdict": "CONTRADICT",
                  "evidence": [
                    {
                      "doc_id": "12486491",
                      "sentence_id": 1
                    },
                    {
                      "doc_id": "12486491",
                      "sentence_id": 4
                    }
                  ]
                }
              }
            },
            "checklist": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.0037686,
              "seconds": 3.0829484579999997,
              "schema_valid": true,
              "outcome": "Recorded CONTRADICT; matches the published claim label.",
              "metrics": {
                "verdict_accuracy": 1.0,
                "evidence_f1": 0.0,
                "citation_integrity": 1.0
              },
              "output": {
                "kind": "verdict",
                "value": {
                  "verdict": "CONTRADICT",
                  "evidence": [
                    {
                      "doc_id": "12486491",
                      "sentence_id": 1
                    },
                    {
                      "doc_id": "12486491",
                      "sentence_id": 4
                    }
                  ]
                }
              }
            },
            "skill": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.0050344,
              "seconds": 2.8181622500000003,
              "schema_valid": true,
              "outcome": "Recorded CONTRADICT; matches the published claim label.",
              "metrics": {
                "verdict_accuracy": 1.0,
                "evidence_f1": 0.0,
                "citation_integrity": 1.0
              },
              "output": {
                "kind": "verdict",
                "value": {
                  "verdict": "CONTRADICT",
                  "evidence": [
                    {
                      "doc_id": "12486491",
                      "sentence_id": 1
                    },
                    {
                      "doc_id": "12486491",
                      "sentence_id": 4
                    }
                  ]
                }
              }
            }
          },
          "evidence": {
            "run_id": "four-track-canary-v3",
            "task_id": "verify-sf1049"
          }
        },
        {
          "id": "verify-sf508",
          "label": "Stem-cell purification",
          "task_summary": "Assess whether blood-forming stem-cell isolation methods can reach 50% purity.",
          "source": {
            "dataset": "SciFact",
            "label": "Stem-cell purification",
            "cluster_id": "scifact-source-4dec666818d18b9e",
            "url": "https://github.com/allenai/scifact",
            "synthetic": false
          },
          "arms": {
            "baseline": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.004418,
              "seconds": 2.2563993330000045,
              "schema_valid": true,
              "outcome": "Recorded NEI; matches the published claim label.",
              "metrics": {
                "verdict_accuracy": 1.0,
                "citation_integrity": 1.0
              },
              "output": {
                "kind": "verdict",
                "value": {
                  "verdict": "NEI",
                  "evidence": []
                }
              }
            },
            "checklist": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.0024046,
              "seconds": 1.381621999999993,
              "schema_valid": true,
              "outcome": "Recorded NEI; matches the published claim label.",
              "metrics": {
                "verdict_accuracy": 1.0,
                "citation_integrity": 1.0
              },
              "output": {
                "kind": "verdict",
                "value": {
                  "verdict": "NEI",
                  "evidence": []
                }
              }
            },
            "skill": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.0033504,
              "seconds": 2.5803896250000093,
              "schema_valid": true,
              "outcome": "Recorded NEI; matches the published claim label.",
              "metrics": {
                "verdict_accuracy": 1.0,
                "citation_integrity": 1.0
              },
              "output": {
                "kind": "verdict",
                "value": {
                  "verdict": "NEI",
                  "evidence": []
                }
              }
            }
          },
          "evidence": {
            "run_id": "four-track-canary-v3",
            "task_id": "verify-sf508"
          }
        },
        {
          "id": "verify-sf597",
          "label": "Cervical-cancer incidence",
          "task_summary": "Assess a claim of declining cervical-cancer incidence.",
          "source": {
            "dataset": "SciFact",
            "label": "Cervical-cancer incidence",
            "cluster_id": "scifact-source-dcc771eb41e19d23",
            "url": "https://github.com/allenai/scifact",
            "synthetic": false
          },
          "arms": {
            "baseline": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.007118,
              "seconds": 2.5631536250000124,
              "schema_valid": true,
              "outcome": "Recorded SUPPORT; matches the published claim label.",
              "metrics": {
                "verdict_accuracy": 1.0,
                "evidence_f1": 1.0,
                "citation_integrity": 1.0
              },
              "output": {
                "kind": "verdict",
                "value": {
                  "verdict": "SUPPORT",
                  "evidence": [
                    {
                      "doc_id": "36355784",
                      "sentence_id": 6
                    },
                    {
                      "doc_id": "36355784",
                      "sentence_id": 7
                    }
                  ]
                }
              }
            },
            "checklist": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.0053046000000000005,
              "seconds": 2.6211575000000096,
              "schema_valid": true,
              "outcome": "Recorded SUPPORT; matches the published claim label.",
              "metrics": {
                "verdict_accuracy": 1.0,
                "evidence_f1": 1.0,
                "citation_integrity": 1.0
              },
              "output": {
                "kind": "verdict",
                "value": {
                  "verdict": "SUPPORT",
                  "evidence": [
                    {
                      "doc_id": "36355784",
                      "sentence_id": 6
                    },
                    {
                      "doc_id": "36355784",
                      "sentence_id": 7
                    }
                  ]
                }
              }
            },
            "skill": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.0060104,
              "seconds": 1.5688376249999862,
              "schema_valid": true,
              "outcome": "Recorded SUPPORT; matches the published claim label.",
              "metrics": {
                "verdict_accuracy": 1.0,
                "evidence_f1": 1.0,
                "citation_integrity": 1.0
              },
              "output": {
                "kind": "verdict",
                "value": {
                  "verdict": "SUPPORT",
                  "evidence": [
                    {
                      "doc_id": "36355784",
                      "sentence_id": 6
                    },
                    {
                      "doc_id": "36355784",
                      "sentence_id": 7
                    }
                  ]
                }
              }
            }
          },
          "evidence": {
            "run_id": "four-track-canary-v3",
            "task_id": "verify-sf597"
          }
        },
        {
          "id": "verify-sf219",
          "label": "A receptor and airway inflammation",
          "task_summary": "Assess whether CX3CR1 on type-2 helper T cells restrains inflammation in the airways.",
          "source": {
            "dataset": "SciFact",
            "label": "A receptor and airway inflammation",
            "cluster_id": "scifact-source-047fcbae4a7ee0d7",
            "url": "https://github.com/allenai/scifact",
            "synthetic": false
          },
          "arms": {
            "baseline": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.00536,
              "seconds": 3.9645467500000002,
              "schema_valid": true,
              "outcome": "Recorded CONTRADICT; matches the published claim label.",
              "metrics": {
                "verdict_accuracy": 1.0,
                "evidence_f1": 0.6666666666666666,
                "citation_integrity": 1.0
              },
              "output": {
                "kind": "verdict",
                "value": {
                  "verdict": "CONTRADICT",
                  "evidence": [
                    {
                      "doc_id": "21366394",
                      "sentence_id": 2
                    },
                    {
                      "doc_id": "21366394",
                      "sentence_id": 4
                    }
                  ]
                }
              }
            },
            "checklist": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.0036366,
              "seconds": 3.306548042,
              "schema_valid": true,
              "outcome": "Recorded CONTRADICT; matches the published claim label.",
              "metrics": {
                "verdict_accuracy": 1.0,
                "evidence_f1": 0.6666666666666666,
                "citation_integrity": 1.0
              },
              "output": {
                "kind": "verdict",
                "value": {
                  "verdict": "CONTRADICT",
                  "evidence": [
                    {
                      "doc_id": "21366394",
                      "sentence_id": 2
                    },
                    {
                      "doc_id": "21366394",
                      "sentence_id": 4
                    }
                  ]
                }
              }
            },
            "skill": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.0038624,
              "seconds": 1.8031465830000002,
              "schema_valid": true,
              "outcome": "Recorded CONTRADICT; matches the published claim label.",
              "metrics": {
                "verdict_accuracy": 1.0,
                "evidence_f1": 0.6666666666666666,
                "citation_integrity": 1.0
              },
              "output": {
                "kind": "verdict",
                "value": {
                  "verdict": "CONTRADICT",
                  "evidence": [
                    {
                      "doc_id": "21366394",
                      "sentence_id": 2
                    },
                    {
                      "doc_id": "21366394",
                      "sentence_id": 4
                    }
                  ]
                }
              }
            }
          },
          "evidence": {
            "run_id": "new-sources-and-analysis-v3",
            "task_id": "verify-sf219"
          }
        },
        {
          "id": "verify-sf388",
          "label": "Bacterial response to alcohol stress",
          "task_summary": "Assess whether alcohol stress lowers bacterial IBP expression.",
          "source": {
            "dataset": "SciFact",
            "label": "Bacterial response to alcohol stress",
            "cluster_id": "scifact-source-25486d3e4985ae13",
            "url": "https://github.com/allenai/scifact",
            "synthetic": false
          },
          "arms": {
            "baseline": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.004404,
              "seconds": 3.0373122499999994,
              "schema_valid": true,
              "outcome": "Recorded NEI; matches the published claim label.",
              "metrics": {
                "verdict_accuracy": 1.0,
                "citation_integrity": 1.0
              },
              "output": {
                "kind": "verdict",
                "value": {
                  "verdict": "NEI",
                  "evidence": []
                }
              }
            },
            "checklist": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.0025905999999999998,
              "seconds": 3.014375375,
              "schema_valid": true,
              "outcome": "Recorded NEI; matches the published claim label.",
              "metrics": {
                "verdict_accuracy": 1.0,
                "citation_integrity": 1.0
              },
              "output": {
                "kind": "verdict",
                "value": {
                  "verdict": "NEI",
                  "evidence": []
                }
              }
            },
            "skill": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.0034764,
              "seconds": 2.749606834000005,
              "schema_valid": true,
              "outcome": "Recorded NEI; matches the published claim label.",
              "metrics": {
                "verdict_accuracy": 1.0,
                "citation_integrity": 1.0
              },
              "output": {
                "kind": "verdict",
                "value": {
                  "verdict": "NEI",
                  "evidence": []
                }
              }
            }
          },
          "evidence": {
            "run_id": "new-sources-and-analysis-v3",
            "task_id": "verify-sf388"
          }
        },
        {
          "id": "verify-sf130",
          "label": "Open access and citations",
          "task_summary": "Assess whether open access is linked to a higher chance of being cited than conventional publication.",
          "source": {
            "dataset": "SciFact",
            "label": "Open access and citations",
            "cluster_id": "scifact-source-a7d883a17dd4e4dd",
            "url": "https://github.com/allenai/scifact",
            "synthetic": false
          },
          "arms": {
            "baseline": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.00868,
              "seconds": 1.9772612090000052,
              "schema_valid": true,
              "outcome": "Recorded SUPPORT; matches the published claim label.",
              "metrics": {
                "verdict_accuracy": 1.0,
                "evidence_f1": 0.5,
                "citation_integrity": 1.0
              },
              "output": {
                "kind": "verdict",
                "value": {
                  "verdict": "SUPPORT",
                  "evidence": [
                    {
                      "doc_id": "27768226",
                      "sentence_id": 2
                    },
                    {
                      "doc_id": "27768226",
                      "sentence_id": 10
                    },
                    {
                      "doc_id": "27768226",
                      "sentence_id": 11
                    }
                  ]
                }
              }
            },
            "checklist": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.0068666000000000005,
              "seconds": 2.618305915999997,
              "schema_valid": true,
              "outcome": "Recorded SUPPORT; matches the published claim label.",
              "metrics": {
                "verdict_accuracy": 1.0,
                "evidence_f1": 0.5,
                "citation_integrity": 1.0
              },
              "output": {
                "kind": "verdict",
                "value": {
                  "verdict": "SUPPORT",
                  "evidence": [
                    {
                      "doc_id": "27768226",
                      "sentence_id": 2
                    },
                    {
                      "doc_id": "27768226",
                      "sentence_id": 10
                    },
                    {
                      "doc_id": "27768226",
                      "sentence_id": 11
                    }
                  ]
                }
              }
            },
            "skill": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.0075724,
              "seconds": 3.4031522090000124,
              "schema_valid": true,
              "outcome": "Recorded SUPPORT; matches the published claim label.",
              "metrics": {
                "verdict_accuracy": 1.0,
                "evidence_f1": 0.5,
                "citation_integrity": 1.0
              },
              "output": {
                "kind": "verdict",
                "value": {
                  "verdict": "SUPPORT",
                  "evidence": [
                    {
                      "doc_id": "27768226",
                      "sentence_id": 2
                    },
                    {
                      "doc_id": "27768226",
                      "sentence_id": 10
                    },
                    {
                      "doc_id": "27768226",
                      "sentence_id": 11
                    }
                  ]
                }
              }
            }
          },
          "evidence": {
            "run_id": "new-sources-and-analysis-v3",
            "task_id": "verify-sf130"
          }
        },
        {
          "id": "verify-sf1382",
          "label": "Tumour growth and metabolism",
          "task_summary": "Assess whether aPKCz promotes tumour growth through changes in glutamine metabolism.",
          "source": {
            "dataset": "SciFact",
            "label": "Tumour growth and metabolism",
            "cluster_id": "scifact-source-5bea185d0e992a50",
            "url": "https://github.com/allenai/scifact",
            "synthetic": false
          },
          "arms": {
            "baseline": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.007096,
              "seconds": 4.495240041999978,
              "schema_valid": true,
              "outcome": "Recorded CONTRADICT; matches the published claim label.",
              "metrics": {
                "verdict_accuracy": 1.0,
                "evidence_f1": 0.5,
                "citation_integrity": 1.0
              },
              "output": {
                "kind": "verdict",
                "value": {
                  "verdict": "CONTRADICT",
                  "evidence": [
                    {
                      "doc_id": "17755060",
                      "sentence_id": 1
                    },
                    {
                      "doc_id": "17755060",
                      "sentence_id": 3
                    },
                    {
                      "doc_id": "17755060",
                      "sentence_id": 5
                    }
                  ]
                }
              }
            },
            "checklist": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.004372600000000001,
              "seconds": 4.239657750000049,
              "schema_valid": true,
              "outcome": "Recorded CONTRADICT; matches the published claim label.",
              "metrics": {
                "verdict_accuracy": 1.0,
                "evidence_f1": 0.5,
                "citation_integrity": 1.0
              },
              "output": {
                "kind": "verdict",
                "value": {
                  "verdict": "CONTRADICT",
                  "evidence": [
                    {
                      "doc_id": "17755060",
                      "sentence_id": 1
                    },
                    {
                      "doc_id": "17755060",
                      "sentence_id": 3
                    },
                    {
                      "doc_id": "17755060",
                      "sentence_id": 5
                    }
                  ]
                }
              }
            },
            "skill": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.0053484,
              "seconds": 7.342974707999986,
              "schema_valid": true,
              "outcome": "Recorded CONTRADICT; matches the published claim label.",
              "metrics": {
                "verdict_accuracy": 1.0,
                "evidence_f1": 0.5,
                "citation_integrity": 1.0
              },
              "output": {
                "kind": "verdict",
                "value": {
                  "verdict": "CONTRADICT",
                  "evidence": [
                    {
                      "doc_id": "17755060",
                      "sentence_id": 1
                    },
                    {
                      "doc_id": "17755060",
                      "sentence_id": 3
                    },
                    {
                      "doc_id": "17755060",
                      "sentence_id": 5
                    }
                  ]
                }
              }
            }
          },
          "evidence": {
            "run_id": "new-sources-and-analysis-v3",
            "task_id": "verify-sf1382"
          }
        }
      ]
    },
    {
      "id": "find",
      "title": "Choose relevant papers",
      "question": "Does literature-review guidance improve selection from a fixed candidate list?",
      "metric": "Mean recall at five",
      "unit": "recall",
      "tasks": 4,
      "source_clusters": 4,
      "arms": {
        "baseline": {
          "label": "No added skill",
          "score": 0.4375,
          "attempts": 4,
          "completed": 4,
          "failed": 0,
          "cost_usd": 0.120948,
          "mean_seconds": 3.058317156500005,
          "schema_valid_count": 4,
          "schema_observed_count": 4,
          "reference_reads": 0
        },
        "checklist": {
          "label": "Model + checklist",
          "score": 0.4375,
          "attempts": 4,
          "completed": 4,
          "failed": 0,
          "cost_usd": 0.12569200000000003,
          "mean_seconds": 3.979592395749997,
          "schema_valid_count": 4,
          "schema_observed_count": 4,
          "reference_reads": 0
        },
        "skill": {
          "label": "Model + skill",
          "score": 0.4375,
          "attempts": 4,
          "completed": 4,
          "failed": 0,
          "cost_usd": 0.13459129999999997,
          "mean_seconds": 3.643752374499994,
          "schema_valid_count": 4,
          "schema_observed_count": 4,
          "reference_reads": 0
        }
      },
      "skill": {
        "id": "literature-review",
        "kind": "upstream_package",
        "url": "https://github.com/K-Dense-AI/scientific-agent-skills/tree/330c8e764435a731eff571e3efdda70b363d0792/skills/literature-review",
        "revision": "330c8e764435a731eff571e3efdda70b363d0792",
        "usage": "SKILL.md supplied; no explicit reference-file reads recorded."
      },
      "interpretation": {
        "summary": "All three conditions reached 43.8% mean recall. Perfect selection from these lists could reach only 50%.",
        "evidence_case_ids": [
          "find-sf971",
          "find-sf1049"
        ]
      },
      "limitations": [
        "Two of four candidate pools contain no known target source.",
        "The target set contains published cited sources, not exhaustive relevance judgments.",
        "This tests fixed-list reranking; searching databases and writing a literature review were not exercised."
      ],
      "cases": [
        {
          "id": "find-sf971",
          "label": "HPV screening sensitivity",
          "task_summary": "Choose five candidate papers for this question: Assess whether HPV-based screening is more sensitive over follow-up than cytology for detecting grade-2 cervical lesions.",
          "source": {
            "dataset": "SciFact",
            "label": "HPV screening sensitivity",
            "cluster_id": "scifact-source-8e49e639da36e853",
            "url": "https://github.com/allenai/scifact",
            "synthetic": false
          },
          "arms": {
            "baseline": {
              "score": 0.75,
              "status": "completed",
              "cost_usd": 0.038828,
              "seconds": 3.870628542000002,
              "schema_valid": true,
              "outcome": "Selected 3 of 4 known cited sources.",
              "metrics": {
                "recall_at_5": 0.75,
                "precision_at_5": 0.6
              },
              "output": {
                "kind": "paper_selection",
                "value": {
                  "paper_ids": [
                    "46695481",
                    "9764256",
                    "27873158",
                    "27446873",
                    "7650066"
                  ]
                },
                "paper_titles": {
                  "46695481": "Human papillomavirus and Papanicolaou tests to screen for cervical cancer.",
                  "9764256": "Human papillomavirus testing for the detection of high-grade cervical intraepithelial neoplasia and cancer: final results of the POBASCAM randomised controlled trial.",
                  "27873158": "Efficacy of human papillomavirus testing for the detection of invasive cervical cancers and cervical intraepithelial neoplasia: a randomised controlled trial.",
                  "27446873": "Rate of cervical cancer, severe intraepithelial neoplasia, and adenocarcinoma in situ in primary HPV DNA screening with cytology triage: randomised study within organised screening…",
                  "7650066": "Long-term follow-up of cervical disease in women screened by cytology and HPV testing: results from the HART study"
                }
              }
            },
            "checklist": {
              "score": 0.75,
              "status": "completed",
              "cost_usd": 0.038834,
              "seconds": 3.718190958000001,
              "schema_valid": true,
              "outcome": "Selected 3 of 4 known cited sources.",
              "metrics": {
                "recall_at_5": 0.75,
                "precision_at_5": 0.6
              },
              "output": {
                "kind": "paper_selection",
                "value": {
                  "paper_ids": [
                    "46695481",
                    "9764256",
                    "27446873",
                    "27873158",
                    "7650066"
                  ]
                },
                "paper_titles": {
                  "46695481": "Human papillomavirus and Papanicolaou tests to screen for cervical cancer.",
                  "9764256": "Human papillomavirus testing for the detection of high-grade cervical intraepithelial neoplasia and cancer: final results of the POBASCAM randomised controlled trial.",
                  "27446873": "Rate of cervical cancer, severe intraepithelial neoplasia, and adenocarcinoma in situ in primary HPV DNA screening with cytology triage: randomised study within organised screening…",
                  "27873158": "Efficacy of human papillomavirus testing for the detection of invasive cervical cancers and cervical intraepithelial neoplasia: a randomised controlled trial.",
                  "7650066": "Long-term follow-up of cervical disease in women screened by cytology and HPV testing: results from the HART study"
                }
              }
            },
            "skill": {
              "score": 0.75,
              "status": "completed",
              "cost_usd": 0.0526695,
              "seconds": 4.975423541,
              "schema_valid": true,
              "outcome": "Selected 3 of 4 known cited sources.",
              "metrics": {
                "recall_at_5": 0.75,
                "precision_at_5": 0.6
              },
              "output": {
                "kind": "paper_selection",
                "value": {
                  "paper_ids": [
                    "46695481",
                    "9764256",
                    "27873158",
                    "27446873",
                    "6561200"
                  ]
                },
                "paper_titles": {
                  "46695481": "Human papillomavirus and Papanicolaou tests to screen for cervical cancer.",
                  "9764256": "Human papillomavirus testing for the detection of high-grade cervical intraepithelial neoplasia and cancer: final results of the POBASCAM randomised controlled trial.",
                  "27873158": "Efficacy of human papillomavirus testing for the detection of invasive cervical cancers and cervical intraepithelial neoplasia: a randomised controlled trial.",
                  "27446873": "Rate of cervical cancer, severe intraepithelial neoplasia, and adenocarcinoma in situ in primary HPV DNA screening with cytology triage: randomised study within organised screening…",
                  "6561200": "Efficacy of HPV DNA testing with cytology triage and/or repeat HPV DNA testing in primary cervical cancer screening."
                }
              }
            }
          },
          "evidence": {
            "run_id": "four-track-canary-v3",
            "task_id": "find-sf971",
            "candidate_count": 20,
            "candidate_oracle_recall": 1.0
          }
        },
        {
          "id": "find-sf1049",
          "label": "Tissue specificity of ribosome disorders",
          "task_summary": "Choose five candidate papers for this question: Assess whether ribosome-related disorders produce largely nonspecific effects across cells and tissues.",
          "source": {
            "dataset": "SciFact",
            "label": "Tissue specificity of ribosome disorders",
            "cluster_id": "scifact-source-71d5408dbc571823",
            "url": "https://github.com/allenai/scifact",
            "synthetic": false
          },
          "arms": {
            "baseline": {
              "score": 0.0,
              "status": "completed",
              "cost_usd": 0.02593,
              "seconds": 2.530306584,
              "schema_valid": true,
              "outcome": "Selected 0 of 1 known cited sources. No known target source was present in the candidate list.",
              "metrics": {
                "recall_at_5": 0.0,
                "precision_at_5": 0.0
              },
              "output": {
                "kind": "paper_selection",
                "value": {
                  "paper_ids": [
                    "8925851",
                    "24737389",
                    "26112624",
                    "7487927",
                    "31304956"
                  ]
                },
                "paper_titles": {
                  "8925851": "Review article",
                  "24737389": "Growth control and ribosomopathies.",
                  "26112624": "The complexity of human ribosome biogenesis revealed by systematic nucleolar screening of Pre-rRNA processing factors.",
                  "7487927": "The Ribosome Biogenesis Factor Nol11 Is Required for Optimal rDNA Transcription and Craniofacial Development in Xenopus ",
                  "31304956": "Cranial neural crest and the building of the vertebrate head"
                }
              }
            },
            "checklist": {
              "score": 0.0,
              "status": "completed",
              "cost_usd": 0.026696,
              "seconds": 4.079303791999999,
              "schema_valid": true,
              "outcome": "Selected 0 of 1 known cited sources. No known target source was present in the candidate list.",
              "metrics": {
                "recall_at_5": 0.0,
                "precision_at_5": 0.0
              },
              "output": {
                "kind": "paper_selection",
                "value": {
                  "paper_ids": [
                    "8925851",
                    "24737389",
                    "7487927",
                    "26112624",
                    "31304956"
                  ]
                },
                "paper_titles": {
                  "8925851": "Review article",
                  "24737389": "Growth control and ribosomopathies.",
                  "7487927": "The Ribosome Biogenesis Factor Nol11 Is Required for Optimal rDNA Transcription and Craniofacial Development in Xenopus ",
                  "26112624": "The complexity of human ribosome biogenesis revealed by systematic nucleolar screening of Pre-rRNA processing factors.",
                  "31304956": "Cranial neural crest and the building of the vertebrate head"
                }
              }
            },
            "skill": {
              "score": 0.0,
              "status": "completed",
              "cost_usd": 0.025360599999999997,
              "seconds": 2.761421499999999,
              "schema_valid": true,
              "outcome": "Selected 0 of 1 known cited sources. No known target source was present in the candidate list.",
              "metrics": {
                "recall_at_5": 0.0,
                "precision_at_5": 0.0
              },
              "output": {
                "kind": "paper_selection",
                "value": {
                  "paper_ids": [
                    "8925851",
                    "24737389",
                    "7487927",
                    "26112624",
                    "31304956"
                  ]
                },
                "paper_titles": {
                  "8925851": "Review article",
                  "24737389": "Growth control and ribosomopathies.",
                  "7487927": "The Ribosome Biogenesis Factor Nol11 Is Required for Optimal rDNA Transcription and Craniofacial Development in Xenopus ",
                  "26112624": "The complexity of human ribosome biogenesis revealed by systematic nucleolar screening of Pre-rRNA processing factors.",
                  "31304956": "Cranial neural crest and the building of the vertebrate head"
                }
              }
            }
          },
          "evidence": {
            "run_id": "four-track-canary-v3",
            "task_id": "find-sf1049",
            "candidate_count": 20,
            "candidate_oracle_recall": 0.0
          }
        },
        {
          "id": "find-sf508",
          "label": "Stem-cell purification",
          "task_summary": "Choose five candidate papers for this question: Assess whether blood-forming stem-cell isolation methods can reach 50% purity.",
          "source": {
            "dataset": "SciFact",
            "label": "Stem-cell purification",
            "cluster_id": "scifact-source-4dec666818d18b9e",
            "url": "https://github.com/allenai/scifact",
            "synthetic": false
          },
          "arms": {
            "baseline": {
              "score": 0.0,
              "status": "completed",
              "cost_usd": 0.022088,
              "seconds": 2.637679250000005,
              "schema_valid": true,
              "outcome": "Selected 0 of 1 known cited sources. No known target source was present in the candidate list.",
              "metrics": {
                "recall_at_5": 0.0,
                "precision_at_5": 0.0
              },
              "output": {
                "kind": "paper_selection",
                "value": {
                  "paper_ids": [
                    "25516011",
                    "40234452",
                    "11900630",
                    "18374364",
                    "13116880"
                  ]
                },
                "paper_titles": {
                  "25516011": "Purification and characterization of mouse hematopoietic stem cells.",
                  "40234452": "HES-1 preserves purified hematopoietic stem cells ex vivo and accumulates side population cells in vivo.",
                  "11900630": "Hematopoietic stem cells and other hematopoietic cells show broad resistance to chemotherapeutic agents in vivo when overexpressing bcl-2.",
                  "18374364": "In vivo proliferation and cell cycle kinetics of long-term self-renewing hematopoietic stem cells.",
                  "13116880": "Hematopoietic stem cell: self-renewal versus differentiation."
                }
              }
            },
            "checklist": {
              "score": 0.0,
              "status": "completed",
              "cost_usd": 0.025484,
              "seconds": 5.340113207999991,
              "schema_valid": true,
              "outcome": "Selected 0 of 1 known cited sources. No known target source was present in the candidate list.",
              "metrics": {
                "recall_at_5": 0.0,
                "precision_at_5": 0.0
              },
              "output": {
                "kind": "paper_selection",
                "value": {
                  "paper_ids": [
                    "25516011",
                    "40234452",
                    "11900630",
                    "13116880",
                    "34982259"
                  ]
                },
                "paper_titles": {
                  "25516011": "Purification and characterization of mouse hematopoietic stem cells.",
                  "40234452": "HES-1 preserves purified hematopoietic stem cells ex vivo and accumulates side population cells in vivo.",
                  "11900630": "Hematopoietic stem cells and other hematopoietic cells show broad resistance to chemotherapeutic agents in vivo when overexpressing bcl-2.",
                  "13116880": "Hematopoietic stem cell: self-renewal versus differentiation.",
                  "34982259": "Of lineage and legacy: the development of mammalian hematopoietic stem cells"
                }
              }
            },
            "skill": {
              "score": 0.0,
              "status": "completed",
              "cost_usd": 0.0239586,
              "seconds": 5.125500415999994,
              "schema_valid": true,
              "outcome": "Selected 0 of 1 known cited sources. No known target source was present in the candidate list.",
              "metrics": {
                "recall_at_5": 0.0,
                "precision_at_5": 0.0
              },
              "output": {
                "kind": "paper_selection",
                "value": {
                  "paper_ids": [
                    "25516011",
                    "40234452",
                    "11900630",
                    "13116880",
                    "18374364"
                  ]
                },
                "paper_titles": {
                  "25516011": "Purification and characterization of mouse hematopoietic stem cells.",
                  "40234452": "HES-1 preserves purified hematopoietic stem cells ex vivo and accumulates side population cells in vivo.",
                  "11900630": "Hematopoietic stem cells and other hematopoietic cells show broad resistance to chemotherapeutic agents in vivo when overexpressing bcl-2.",
                  "13116880": "Hematopoietic stem cell: self-renewal versus differentiation.",
                  "18374364": "In vivo proliferation and cell cycle kinetics of long-term self-renewing hematopoietic stem cells."
                }
              }
            }
          },
          "evidence": {
            "run_id": "four-track-canary-v3",
            "task_id": "find-sf508",
            "candidate_count": 20,
            "candidate_oracle_recall": 0.0
          }
        },
        {
          "id": "find-sf597",
          "label": "Cervical-cancer incidence",
          "task_summary": "Choose five candidate papers for this question: Assess a claim of declining cervical-cancer incidence.",
          "source": {
            "dataset": "SciFact",
            "label": "Cervical-cancer incidence",
            "cluster_id": "scifact-source-dcc771eb41e19d23",
            "url": "https://github.com/allenai/scifact",
            "synthetic": false
          },
          "arms": {
            "baseline": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.034102,
              "seconds": 3.1946542500000135,
              "schema_valid": true,
              "outcome": "Selected 3 of 3 known cited sources.",
              "metrics": {
                "recall_at_5": 1.0,
                "precision_at_5": 0.6
              },
              "output": {
                "kind": "paper_selection",
                "value": {
                  "paper_ids": [
                    "25742130",
                    "36355784",
                    "12779444",
                    "52188256",
                    "27446873"
                  ]
                },
                "paper_titles": {
                  "25742130": "Mass screening programmes and trends in cervical cancer in Finland and the Netherlands.",
                  "36355784": "The effect of mass screening on incidence and mortality of squamous and adenocarcinoma of cervix uteri.",
                  "12779444": "Effect of screening on cervical cancer mortality in England and Wales: analysis of trends with an age period cohort model.",
                  "52188256": "Global cancer statistics 2018: GLOBOCAN estimates of incidence and mortality worldwide for 36 cancers in 185 countries.",
                  "27446873": "Rate of cervical cancer, severe intraepithelial neoplasia, and adenocarcinoma in situ in primary HPV DNA screening with cytology triage: randomised study within organised screening…"
                }
              }
            },
            "checklist": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.034678,
              "seconds": 2.7807616249999967,
              "schema_valid": true,
              "outcome": "Selected 3 of 3 known cited sources.",
              "metrics": {
                "recall_at_5": 1.0,
                "precision_at_5": 0.6
              },
              "output": {
                "kind": "paper_selection",
                "value": {
                  "paper_ids": [
                    "25742130",
                    "36355784",
                    "12779444",
                    "52188256",
                    "27446873"
                  ]
                },
                "paper_titles": {
                  "25742130": "Mass screening programmes and trends in cervical cancer in Finland and the Netherlands.",
                  "36355784": "The effect of mass screening on incidence and mortality of squamous and adenocarcinoma of cervix uteri.",
                  "12779444": "Effect of screening on cervical cancer mortality in England and Wales: analysis of trends with an age period cohort model.",
                  "52188256": "Global cancer statistics 2018: GLOBOCAN estimates of incidence and mortality worldwide for 36 cancers in 185 countries.",
                  "27446873": "Rate of cervical cancer, severe intraepithelial neoplasia, and adenocarcinoma in situ in primary HPV DNA screening with cytology triage: randomised study within organised screening…"
                }
              }
            },
            "skill": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.032602599999999995,
              "seconds": 1.7126640409999823,
              "schema_valid": true,
              "outcome": "Selected 3 of 3 known cited sources.",
              "metrics": {
                "recall_at_5": 1.0,
                "precision_at_5": 0.6
              },
              "output": {
                "kind": "paper_selection",
                "value": {
                  "paper_ids": [
                    "25742130",
                    "36355784",
                    "12779444",
                    "52188256",
                    "27446873"
                  ]
                },
                "paper_titles": {
                  "25742130": "Mass screening programmes and trends in cervical cancer in Finland and the Netherlands.",
                  "36355784": "The effect of mass screening on incidence and mortality of squamous and adenocarcinoma of cervix uteri.",
                  "12779444": "Effect of screening on cervical cancer mortality in England and Wales: analysis of trends with an age period cohort model.",
                  "52188256": "Global cancer statistics 2018: GLOBOCAN estimates of incidence and mortality worldwide for 36 cancers in 185 countries.",
                  "27446873": "Rate of cervical cancer, severe intraepithelial neoplasia, and adenocarcinoma in situ in primary HPV DNA screening with cytology triage: randomised study within organised screening…"
                }
              }
            }
          },
          "evidence": {
            "run_id": "four-track-canary-v3",
            "task_id": "find-sf597",
            "candidate_count": 20,
            "candidate_oracle_recall": 1.0
          }
        }
      ]
    },
    {
      "id": "extract",
      "title": "Extract entities and relations",
      "question": "Does a local annotation procedure improve structured scientific extraction?",
      "metric": "Mean of entity and relation F1",
      "unit": "f1",
      "tasks": 8,
      "source_clusters": 8,
      "arms": {
        "baseline": {
          "label": "No added skill",
          "score": 0.23965531275876104,
          "attempts": 8,
          "completed": 8,
          "failed": 0,
          "cost_usd": 0.09919619999999998,
          "mean_seconds": 7.410591156375002,
          "schema_valid_count": 5,
          "schema_observed_count": 8,
          "reference_reads": 0
        },
        "checklist": {
          "label": "Model + checklist",
          "score": 0.30550972247342745,
          "attempts": 8,
          "completed": 7,
          "failed": 1,
          "cost_usd": 0.1049132,
          "mean_seconds": 7.478221140750003,
          "schema_valid_count": 6,
          "schema_observed_count": 7,
          "reference_reads": 0
        },
        "skill": {
          "label": "Model + local procedure",
          "score": 0.3343047793039286,
          "attempts": 8,
          "completed": 8,
          "failed": 0,
          "cost_usd": 0.1320054,
          "mean_seconds": 10.311966020750004,
          "schema_valid_count": 7,
          "schema_observed_count": 8,
          "reference_reads": 0
        }
      },
      "skill": {
        "id": "local-scierc-procedure",
        "kind": "local_procedure",
        "url": null,
        "revision": null,
        "usage": "SKILL.md supplied; no explicit reference-file reads recorded."
      },
      "interpretation": {
        "summary": "Mean F1 rose from 0.240 to 0.334 with the local procedure. Most observed improvement came from avoiding malformed outputs.",
        "evidence_case_ids": [
          "extract-scierc-ICCV_2003_151_abs",
          "extract-scierc-C94-1091"
        ]
      },
      "limitations": [
        "The skill condition is a local SciERC procedure, not an upstream scientific-skill package.",
        "Eight AI abstracts; published annotations have not been independently reviewed.",
        "Format failures receive zero. A shared schema-enforcement control is needed before attributing the difference to extraction accuracy.",
        "No PDF parsing, tables, or scientific results extraction were tested."
      ],
      "cases": [
        {
          "id": "extract-scierc-ICCV_2003_151_abs",
          "label": "Preserving eye contact in video calls",
          "task_summary": "Identify stereo-matching and view-generation methods, and map how they support eye contact and reduce visual artifacts.",
          "source": {
            "dataset": "SciERC",
            "label": "SciERC ICCV_2003_151_abs",
            "cluster_id": "scierc:ICCV_2003_151_abs",
            "url": "https://nlp.cs.washington.edu/sciIE/",
            "synthetic": false
          },
          "arms": {
            "baseline": {
              "score": 0.0,
              "status": "completed",
              "cost_usd": 0.016075,
              "seconds": 7.138260542000001,
              "schema_valid": false,
              "outcome": "The submitted structure failed the required schema and received zero.",
              "metrics": {
                "ent_f1": 0.0,
                "rel_f1": 0.0,
                "source_span_precision": 0.0
              },
              "output": {
                "kind": "extraction",
                "value": {
                  "text": {
                    "entities": [
                      {
                        "text": "novel view generation",
                        "type": "Task"
                      },
                      {
                        "text": "one-to-one teleconferencing applications",
                        "type": "Task"
                      },
                      {
                        "text": "video streams",
                        "type": "Material"
                      },
                      {
                        "text": "two cameras",
                        "type": "Material"
                      },
                      {
                        "text": "computer monitor",
                        "type": "Material"
                      }
                    ],
                    "relations": [
                      {
                        "head": "dynamic-programming, stereo algorithm",
                        "tail": "novel-view generation",
                        "type": "USED-FOR"
                      },
                      {
                        "head": "three-plane graph",
                        "tail": "occlusion labeling",
                        "type": "USED-FOR"
                      },
                      {
                        "head": "three-plane graph",
                        "tail": "dense-stereo dynamic-programming",
                        "type": "HYPONYM-OF"
                      },
                      {
                        "head": "geometric derivation",
                        "tail": "novel-view synthesis",
                        "type": "USED-FOR"
                      },
                      {
                        "head": "minimum-cost surface",
                        "tail": "novel-view synthesis",
                        "type": "USED-FOR"
                      }
                    ]
                  }
                },
                "preview_note": "Exact submitted structure; at most five items per array and 120 characters per string. Omitted items still count in the score."
              }
            },
            "checklist": {
              "score": 0.34937124111536355,
              "status": "completed",
              "cost_usd": 0.02387,
              "seconds": 12.742875540999997,
              "schema_valid": true,
              "outcome": "Entity F1 0.441; relation F1 0.258.",
              "metrics": {
                "ent_f1": 0.4406779661016949,
                "rel_f1": 0.25806451612903225,
                "source_span_precision": 0.9574468085106383
              },
              "output": {
                "kind": "extraction",
                "value": {
                  "entities": [
                    {
                      "text": "algorithm",
                      "type": "Method"
                    },
                    {
                      "text": "novel view generation",
                      "type": "Task"
                    },
                    {
                      "text": "one-to-one teleconferencing applications",
                      "type": "Task"
                    },
                    {
                      "text": "video streams",
                      "type": "Material"
                    },
                    {
                      "text": "two cameras",
                      "type": "Material"
                    }
                  ],
                  "relations": [
                    {
                      "head": "algorithm",
                      "tail": "novel view generation",
                      "type": "USED-FOR"
                    },
                    {
                      "head": "novel view generation",
                      "tail": "one-to-one teleconferencing applications",
                      "type": "USED-FOR"
                    },
                    {
                      "head": "virtual camera",
                      "tail": "eye contact",
                      "type": "USED-FOR"
                    },
                    {
                      "head": "dynamic-programming, stereo algorithm",
                      "tail": "novel-view generation",
                      "type": "USED-FOR"
                    },
                    {
                      "head": "three-plane graph",
                      "tail": "dense-stereo dynamic-programming",
                      "type": "USED-FOR"
                    }
                  ]
                },
                "preview_note": "Exact submitted structure; at most five items per array and 120 characters per string. Omitted items still count in the score."
              }
            },
            "skill": {
              "score": 0.28673245614035087,
              "status": "completed",
              "cost_usd": 0.024255,
              "seconds": 12.297270500000003,
              "schema_valid": true,
              "outcome": "Entity F1 0.386; relation F1 0.188.",
              "metrics": {
                "ent_f1": 0.38596491228070173,
                "rel_f1": 0.1875,
                "source_span_precision": 0.9574468085106383
              },
              "output": {
                "kind": "extraction",
                "value": {
                  "entities": [
                    {
                      "text": "novel view generation",
                      "type": "Task"
                    },
                    {
                      "text": "one-to-one teleconferencing applications",
                      "type": "Task"
                    },
                    {
                      "text": "video streams",
                      "type": "Material"
                    },
                    {
                      "text": "two cameras",
                      "type": "Material"
                    },
                    {
                      "text": "computer monitor",
                      "type": "Material"
                    }
                  ],
                  "relations": [
                    {
                      "head": "dynamic-programming, stereo algorithm",
                      "tail": "novel view generation",
                      "type": "USED-FOR"
                    },
                    {
                      "head": "three-plane graph",
                      "tail": "dense-stereo dynamic-programming",
                      "type": "USED-FOR"
                    },
                    {
                      "head": "three-plane graph",
                      "tail": "occlusion labeling",
                      "type": "USED-FOR"
                    },
                    {
                      "head": "geometric derivation",
                      "tail": "novel-view synthesis",
                      "type": "USED-FOR"
                    },
                    {
                      "head": "novel-view synthesis",
                      "tail": "minimum-cost surface",
                      "type": "USED-FOR"
                    }
                  ]
                },
                "preview_note": "Exact submitted structure; at most five items per array and 120 characters per string. Omitted items still count in the score."
              }
            }
          },
          "evidence": {
            "run_id": "four-track-canary-v3",
            "task_id": "extract-scierc-ICCV_2003_151_abs"
          }
        },
        {
          "id": "extract-scierc-C94-1091",
          "label": "Choosing Thai classifier words",
          "task_summary": "Identify corpus-based methods, noun–classifier associations, and linguistic constraints, then map how they support classifier selection.",
          "source": {
            "dataset": "SciERC",
            "label": "SciERC C94-1091",
            "cluster_id": "scierc:C94-1091",
            "url": "https://nlp.cs.washington.edu/sciIE/",
            "synthetic": false
          },
          "arms": {
            "baseline": {
              "score": 0.5,
              "status": "completed",
              "cost_usd": 0.010773600000000001,
              "seconds": 7.064478375,
              "schema_valid": true,
              "outcome": "Entity F1 0.600; relation F1 0.400.",
              "metrics": {
                "ent_f1": 0.6,
                "rel_f1": 0.4,
                "source_span_precision": 1.0
              },
              "output": {
                "kind": "extraction",
                "value": {
                  "entities": [
                    {
                      "text": "algorithm",
                      "type": "Method"
                    },
                    {
                      "text": "classifier word",
                      "type": "OtherScientificTerm"
                    },
                    {
                      "text": "noun",
                      "type": "OtherScientificTerm"
                    },
                    {
                      "text": "Thai language",
                      "type": "Material"
                    },
                    {
                      "text": "classifier",
                      "type": "OtherScientificTerm"
                    }
                  ],
                  "relations": [
                    {
                      "head": "algorithm",
                      "tail": "classifier selection",
                      "type": "USED-FOR"
                    },
                    {
                      "head": "default rule",
                      "tail": "classifier selection",
                      "type": "USED-FOR"
                    },
                    {
                      "head": "rule-based approach",
                      "tail": "classifier selection",
                      "type": "USED-FOR"
                    },
                    {
                      "head": "corpus-based method",
                      "tail": "classifier assignment",
                      "type": "USED-FOR"
                    },
                    {
                      "head": "corpus-based method",
                      "tail": "semantic construction of noun phrase",
                      "type": "USED-FOR"
                    }
                  ]
                },
                "preview_note": "Exact submitted structure; at most five items per array and 120 characters per string. Omitted items still count in the score."
              }
            },
            "checklist": {
              "score": 0.5,
              "status": "completed",
              "cost_usd": 0.014979600000000001,
              "seconds": 8.89421325,
              "schema_valid": true,
              "outcome": "Entity F1 0.571; relation F1 0.429.",
              "metrics": {
                "ent_f1": 0.5714285714285714,
                "rel_f1": 0.42857142857142855,
                "source_span_precision": 1.0
              },
              "output": {
                "kind": "extraction",
                "value": {
                  "entities": [
                    {
                      "text": "algorithm for selecting an appropriate classifier word for a noun",
                      "type": "Task"
                    },
                    {
                      "text": "classifier word",
                      "type": "OtherScientificTerm"
                    },
                    {
                      "text": "noun",
                      "type": "OtherScientificTerm"
                    },
                    {
                      "text": "Thai language",
                      "type": "Material"
                    },
                    {
                      "text": "classifier",
                      "type": "OtherScientificTerm"
                    }
                  ],
                  "relations": [
                    {
                      "head": "corpus-based method",
                      "tail": "Noun Classifier Associations ( NCA )",
                      "type": "USED-FOR"
                    },
                    {
                      "head": "Noun Classifier Associations ( NCA )",
                      "tail": "classifier assignment",
                      "type": "USED-FOR"
                    },
                    {
                      "head": "Noun Classifier Associations ( NCA )",
                      "tail": "semantic construction of noun phrase",
                      "type": "USED-FOR"
                    },
                    {
                      "head": "large corpus",
                      "tail": "Noun Classifier Associations ( NCA )",
                      "type": "USED-FOR"
                    },
                    {
                      "head": "default rule",
                      "tail": "classifier",
                      "type": "USED-FOR"
                    }
                  ]
                },
                "preview_note": "Exact submitted structure; at most five items per array and 120 characters per string. Omitted items still count in the score."
              }
            },
            "skill": {
              "score": 0.3992740471869328,
              "status": "completed",
              "cost_usd": 0.0200192,
              "seconds": 13.2367925,
              "schema_valid": true,
              "outcome": "Entity F1 0.483; relation F1 0.316.",
              "metrics": {
                "ent_f1": 0.4827586206896552,
                "rel_f1": 0.3157894736842105,
                "source_span_precision": 1.0
              },
              "output": {
                "kind": "extraction",
                "value": {
                  "entities": [
                    {
                      "text": "algorithm",
                      "type": "Method"
                    },
                    {
                      "text": "classifier word",
                      "type": "OtherScientificTerm"
                    },
                    {
                      "text": "noun",
                      "type": "OtherScientificTerm"
                    },
                    {
                      "text": "Thai language",
                      "type": "Material"
                    },
                    {
                      "text": "classifier",
                      "type": "OtherScientificTerm"
                    }
                  ],
                  "relations": [
                    {
                      "head": "algorithm",
                      "tail": "classifier word",
                      "type": "USED-FOR"
                    },
                    {
                      "head": "default rule",
                      "tail": "classifier",
                      "type": "USED-FOR"
                    },
                    {
                      "head": "unit classifier",
                      "tail": "classifier",
                      "type": "HYPONYM-OF"
                    },
                    {
                      "head": "corpus-based method",
                      "tail": "classifier assignment",
                      "type": "USED-FOR"
                    },
                    {
                      "head": "corpus-based method",
                      "tail": "semantic construction of noun phrase",
                      "type": "USED-FOR"
                    }
                  ]
                },
                "preview_note": "Exact submitted structure; at most five items per array and 120 characters per string. Omitted items still count in the score."
              }
            }
          },
          "evidence": {
            "run_id": "four-track-canary-v3",
            "task_id": "extract-scierc-C94-1091"
          }
        },
        {
          "id": "extract-scierc-P01-1056",
          "label": "Comparing sentence planners for dialogue",
          "task_summary": "Extract the sentence-planning methods and human-evaluation procedure, including which systems are compared.",
          "source": {
            "dataset": "SciERC",
            "label": "SciERC P01-1056",
            "cluster_id": "scierc:P01-1056",
            "url": "https://nlp.cs.washington.edu/sciIE/",
            "synthetic": false
          },
          "arms": {
            "baseline": {
              "score": 0.37362637362637363,
              "status": "completed",
              "cost_usd": 0.0103556,
              "seconds": 6.686735625000011,
              "schema_valid": true,
              "outcome": "Entity F1 0.462; relation F1 0.286.",
              "metrics": {
                "ent_f1": 0.46153846153846156,
                "rel_f1": 0.2857142857142857,
                "source_span_precision": 1.0
              },
              "output": {
                "kind": "extraction",
                "value": {
                  "entities": [
                    {
                      "text": "natural language generator",
                      "type": "Method"
                    },
                    {
                      "text": "trainable components",
                      "type": "Method"
                    },
                    {
                      "text": "hand-crafted template-based",
                      "type": "Method"
                    },
                    {
                      "text": "rule-based approaches",
                      "type": "Method"
                    },
                    {
                      "text": "trainable sentence planner",
                      "type": "Method"
                    }
                  ],
                  "relations": [
                    {
                      "head": "trainable sentence planner",
                      "tail": "spoken dialogue system",
                      "type": "USED-FOR"
                    },
                    {
                      "head": "hand-crafted template-based",
                      "tail": "rule-based approaches",
                      "type": "CONJUNCTION"
                    },
                    {
                      "head": "trainable sentence planner",
                      "tail": "hand-crafted template-based generation component",
                      "type": "COMPARE"
                    },
                    {
                      "head": "trainable sentence planner",
                      "tail": "rule-based sentence planners",
                      "type": "COMPARE"
                    },
                    {
                      "head": "trainable sentence planner",
                      "tail": "baseline sentence planners",
                      "type": "COMPARE"
                    }
                  ]
                },
                "preview_note": "Exact submitted structure; at most five items per array and 120 characters per string. Omitted items still count in the score."
              }
            },
            "checklist": {
              "score": 0.0,
              "status": "completed",
              "cost_usd": 0.0075816,
              "seconds": 4.193310875000009,
              "schema_valid": false,
              "outcome": "The submitted structure failed the required schema and received zero.",
              "metrics": {
                "ent_f1": 0.0,
                "rel_f1": 0.0,
                "source_span_precision": 0.0
              },
              "output": {
                "kind": "extraction",
                "value": {
                  "args": {
                    "entities": [
                      {
                        "text": "modules of a natural language generator",
                        "type": "OtherScientificTerm"
                      },
                      {
                        "text": "trainable components",
                        "type": "Method"
                      },
                      {
                        "text": "hand-crafted template-based",
                        "type": "Method"
                      },
                      {
                        "text": "rule-based approaches",
                        "type": "Method"
                      },
                      {
                        "text": "trainable sentence planner",
                        "type": "Method"
                      }
                    ],
                    "relations": [
                      {
                        "head": "trainable sentence planner",
                        "tail": "spoken dialogue system",
                        "type": "EVALUATE-FOR"
                      },
                      {
                        "head": "human judgments",
                        "tail": "trainable sentence planner",
                        "type": "EVALUATE-FOR"
                      },
                      {
                        "head": "trainable sentence planner",
                        "tail": "hand-crafted template-based generation component",
                        "type": "COMPARE"
                      },
                      {
                        "head": "trainable sentence planner",
                        "tail": "rule-based sentence planners",
                        "type": "COMPARE"
                      },
                      {
                        "head": "trainable sentence planner",
                        "tail": "baseline sentence planners",
                        "type": "COMPARE"
                      }
                    ]
                  }
                },
                "preview_note": "Exact submitted structure; at most five items per array and 120 characters per string. Omitted items still count in the score."
              }
            },
            "skill": {
              "score": 0.46865203761755486,
              "status": "completed",
              "cost_usd": 0.0087512,
              "seconds": 5.1442365830000085,
              "schema_valid": true,
              "outcome": "Entity F1 0.483; relation F1 0.455.",
              "metrics": {
                "ent_f1": 0.4827586206896552,
                "rel_f1": 0.45454545454545453,
                "source_span_precision": 1.0
              },
              "output": {
                "kind": "extraction",
                "value": {
                  "entities": [
                    {
                      "text": "natural language generator",
                      "type": "Method"
                    },
                    {
                      "text": "trainable components",
                      "type": "Method"
                    },
                    {
                      "text": "template-based",
                      "type": "Method"
                    },
                    {
                      "text": "rule-based approaches",
                      "type": "Method"
                    },
                    {
                      "text": "trainable sentence planner",
                      "type": "Method"
                    }
                  ],
                  "relations": [
                    {
                      "head": "trainable sentence planner",
                      "tail": "spoken dialogue system",
                      "type": "USED-FOR"
                    },
                    {
                      "head": "human judgments",
                      "tail": "trainable sentence planner",
                      "type": "EVALUATE-FOR"
                    },
                    {
                      "head": "trainable sentence planner",
                      "tail": "rule-based systems",
                      "type": "COMPARE"
                    },
                    {
                      "head": "trainable sentence planner",
                      "tail": "baselines",
                      "type": "COMPARE"
                    },
                    {
                      "head": "trainable sentence planner",
                      "tail": "hand-crafted system",
                      "type": "COMPARE"
                    }
                  ]
                },
                "preview_note": "Exact submitted structure; at most five items per array and 120 characters per string. Omitted items still count in the score."
              }
            }
          },
          "evidence": {
            "run_id": "four-track-canary-v3",
            "task_id": "extract-scierc-P01-1056"
          }
        },
        {
          "id": "extract-scierc-ICCV_2001_47_abs",
          "label": "Reconstructing 3D shapes from stereo images",
          "task_summary": "Identify mesh-based reconstruction methods and geometric constraints, and connect them to shape recovery.",
          "source": {
            "dataset": "SciERC",
            "label": "SciERC ICCV_2001_47_abs",
            "cluster_id": "scierc:ICCV_2001_47_abs",
            "url": "https://nlp.cs.washington.edu/sciIE/",
            "synthetic": false
          },
          "arms": {
            "baseline": {
              "score": 0.0,
              "status": "completed",
              "cost_usd": 0.0114196,
              "seconds": 6.87721804200001,
              "schema_valid": false,
              "outcome": "The submitted structure failed the required schema and received zero.",
              "metrics": {
                "ent_f1": 0.0,
                "rel_f1": 0.0,
                "source_span_precision": 0.0
              },
              "output": {
                "kind": "extraction",
                "value": {
                  "text": {
                    "entities": [
                      {
                        "text": "a priori geometric constraints",
                        "type": "OtherScientificTerm"
                      },
                      {
                        "text": "3 -- D stereo reconstruction scheme",
                        "type": "Method"
                      },
                      {
                        "text": "image information",
                        "type": "OtherScientificTerm"
                      },
                      {
                        "text": "3 -- D shape",
                        "type": "OtherScientificTerm"
                      },
                      {
                        "text": "iterative deformation",
                        "type": "Method"
                      }
                    ],
                    "relations": [
                      {
                        "head": "a priori geometric constraints",
                        "tail": "3 -- D stereo reconstruction scheme",
                        "type": "USED-FOR"
                      },
                      {
                        "head": "3 -- D stereo reconstruction scheme",
                        "tail": "3 -- D shape",
                        "type": "USED-FOR"
                      },
                      {
                        "head": "iterative deformation",
                        "tail": "objective function",
                        "type": "USED-FOR"
                      },
                      {
                        "head": "3 -- D surface mesh",
                        "tail": "objective function",
                        "type": "USED-FOR"
                      },
                      {
                        "head": "anisotropic meshing",
                        "tail": "non-quadratic approach to regularization",
                        "type": "CONJUNCTION"
                      }
                    ]
                  }
                },
                "preview_note": "Exact submitted structure; at most five items per array and 120 characters per string. Omitted items still count in the score."
              }
            },
            "checklist": {
              "score": 0.2916666666666667,
              "status": "completed",
              "cost_usd": 0.0114756,
              "seconds": 6.215950875000004,
              "schema_valid": true,
              "outcome": "Entity F1 0.500; relation F1 0.083.",
              "metrics": {
                "ent_f1": 0.5,
                "rel_f1": 0.08333333333333333,
                "source_span_precision": 1.0
              },
              "output": {
                "kind": "extraction",
                "value": {
                  "entities": [
                    {
                      "text": "geometric constraints",
                      "type": "OtherScientificTerm"
                    },
                    {
                      "text": "3 -- D stereo reconstruction scheme",
                      "type": "Method"
                    },
                    {
                      "text": "image information",
                      "type": "OtherScientificTerm"
                    },
                    {
                      "text": "3 -- D shape",
                      "type": "OtherScientificTerm"
                    },
                    {
                      "text": "iterative deformation",
                      "type": "Method"
                    }
                  ],
                  "relations": [
                    {
                      "head": "geometric constraints",
                      "tail": "3 -- D stereo reconstruction scheme",
                      "type": "USED-FOR"
                    },
                    {
                      "head": "iterative deformation",
                      "tail": "objective function",
                      "type": "USED-FOR"
                    },
                    {
                      "head": "anisotropic meshing",
                      "tail": "non-quadratic approach to regularization",
                      "type": "CONJUNCTION"
                    },
                    {
                      "head": "anisotropic meshing",
                      "tail": "reconstruction results",
                      "type": "USED-FOR"
                    },
                    {
                      "head": "non-quadratic approach to regularization",
                      "tail": "reconstruction results",
                      "type": "USED-FOR"
                    }
                  ]
                },
                "preview_note": "Exact submitted structure; at most five items per array and 120 characters per string. Omitted items still count in the score."
              }
            },
            "skill": {
              "score": 0.0,
              "status": "completed",
              "cost_usd": 0.009985200000000001,
              "seconds": 6.182921625000006,
              "schema_valid": false,
              "outcome": "The submitted structure failed the required schema and received zero.",
              "metrics": {
                "ent_f1": 0.0,
                "rel_f1": 0.0,
                "source_span_precision": 0.0
              },
              "output": {
                "kind": "extraction",
                "value": {
                  "parameters": {
                    "entities": [
                      {
                        "text": "a priori geometric constraints",
                        "type": "OtherScientificTerm"
                      },
                      {
                        "text": "3 -- D stereo reconstruction scheme",
                        "type": "Method"
                      },
                      {
                        "text": "image information",
                        "type": "OtherScientificTerm"
                      },
                      {
                        "text": "3 -- D shape",
                        "type": "OtherScientificTerm"
                      },
                      {
                        "text": "iterative deformation",
                        "type": "Method"
                      }
                    ],
                    "relations": [
                      {
                        "head": "a priori geometric constraints",
                        "tail": "3 -- D stereo reconstruction scheme",
                        "type": "USED-FOR"
                      },
                      {
                        "head": "anisotropic meshing",
                        "tail": "non-quadratic approach",
                        "type": "CONJUNCTION"
                      },
                      {
                        "head": "anisotropic meshing",
                        "tail": "triangulations",
                        "type": "USED-FOR"
                      },
                      {
                        "head": "non-quadratic approach",
                        "tail": "regularization",
                        "type": "USED-FOR"
                      },
                      {
                        "head": "Structural or numerical constraints",
                        "tail": "reconstruction process",
                        "type": "USED-FOR"
                      }
                    ]
                  }
                },
                "preview_note": "Exact submitted structure; at most five items per array and 120 characters per string. Omitted items still count in the score."
              }
            }
          },
          "evidence": {
            "run_id": "four-track-canary-v3",
            "task_id": "extract-scierc-ICCV_2001_47_abs"
          }
        },
        {
          "id": "extract-scierc-CVPR_1994_10_abs",
          "label": "Fitting data with many outliers",
          "task_summary": "Extract MINPRAN, its sampling assumptions, comparison method, and evaluation data, and map their roles in robust fitting.",
          "source": {
            "dataset": "SciERC",
            "label": "SciERC CVPR_1994_10_abs",
            "cluster_id": "scierc:CVPR_1994_10_abs",
            "url": "https://nlp.cs.washington.edu/sciIE/",
            "synthetic": false
          },
          "arms": {
            "baseline": {
              "score": 0.4735449735449735,
              "status": "completed",
              "cost_usd": 0.0111196,
              "seconds": 7.055017124999999,
              "schema_valid": true,
              "outcome": "Entity F1 0.519; relation F1 0.429.",
              "metrics": {
                "ent_f1": 0.5185185185185185,
                "rel_f1": 0.42857142857142855,
                "source_span_precision": 0.9615384615384616
              },
              "output": {
                "kind": "extraction",
                "value": {
                  "entities": [
                    {
                      "text": "MINPRAN",
                      "type": "Method"
                    },
                    {
                      "text": "robust operator",
                      "type": "Method"
                    },
                    {
                      "text": "outliers",
                      "type": "OtherScientificTerm"
                    },
                    {
                      "text": "error bound",
                      "type": "OtherScientificTerm"
                    },
                    {
                      "text": "good data",
                      "type": "Material"
                    }
                  ],
                  "relations": [
                    {
                      "head": "MINPRAN",
                      "type": "HYPONYM-OF",
                      "tail": "robust operator"
                    },
                    {
                      "head": "MINPRAN",
                      "type": "COMPARE",
                      "tail": "least median of squares"
                    },
                    {
                      "head": "random sampling",
                      "type": "USED-FOR",
                      "tail": "MINPRAN"
                    },
                    {
                      "head": "MINPRAN",
                      "type": "EVALUATE-FOR",
                      "tail": "synthetic data"
                    },
                    {
                      "head": "MINPRAN",
                      "type": "USED-FOR",
                      "tail": "range and intensity data"
                    }
                  ]
                },
                "preview_note": "Exact submitted structure; at most five items per array and 120 characters per string. Omitted items still count in the score."
              }
            },
            "checklist": {
              "score": 0.5490716180371353,
              "status": "completed",
              "cost_usd": 0.0110556,
              "seconds": 6.9318529170000005,
              "schema_valid": true,
              "outcome": "Entity F1 0.483; relation F1 0.615.",
              "metrics": {
                "ent_f1": 0.4827586206896552,
                "rel_f1": 0.6153846153846154,
                "source_span_precision": 0.9615384615384616
              },
              "output": {
                "kind": "extraction",
                "value": {
                  "entities": [
                    {
                      "text": "MINPRAN",
                      "type": "Method"
                    },
                    {
                      "text": "robust operator",
                      "type": "Method"
                    },
                    {
                      "text": "data sets",
                      "type": "Material"
                    },
                    {
                      "text": "outliers",
                      "type": "OtherScientificTerm"
                    },
                    {
                      "text": "error bound",
                      "type": "OtherScientificTerm"
                    }
                  ],
                  "relations": [
                    {
                      "head": "MINPRAN",
                      "type": "HYPONYM-OF",
                      "tail": "robust operator"
                    },
                    {
                      "head": "random sampling",
                      "type": "USED-FOR",
                      "tail": "MINPRAN"
                    },
                    {
                      "head": "MINPRAN",
                      "type": "COMPARE",
                      "tail": "least median of squares"
                    },
                    {
                      "head": "synthetic data",
                      "type": "EVALUATE-FOR",
                      "tail": "MINPRAN"
                    },
                    {
                      "head": "MINPRAN",
                      "type": "USED-FOR",
                      "tail": "complex range and intensity data"
                    }
                  ]
                },
                "preview_note": "Exact submitted structure; at most five items per array and 120 characters per string. Omitted items still count in the score."
              }
            },
            "skill": {
              "score": 0.4735632183908046,
              "status": "completed",
              "cost_usd": 0.0213852,
              "seconds": 15.626798000000004,
              "schema_valid": true,
              "outcome": "Entity F1 0.414; relation F1 0.533.",
              "metrics": {
                "ent_f1": 0.41379310344827586,
                "rel_f1": 0.5333333333333333,
                "source_span_precision": 1.0
              },
              "output": {
                "kind": "extraction",
                "value": {
                  "entities": [
                    {
                      "text": "MINPRAN",
                      "type": "Method"
                    },
                    {
                      "text": "robust operator",
                      "type": "Method"
                    },
                    {
                      "text": "data sets",
                      "type": "Material"
                    },
                    {
                      "text": "outliers",
                      "type": "OtherScientificTerm"
                    },
                    {
                      "text": "error bound",
                      "type": "OtherScientificTerm"
                    }
                  ],
                  "relations": [
                    {
                      "head": "MINPRAN",
                      "type": "HYPONYM-OF",
                      "tail": "robust operator"
                    },
                    {
                      "head": "random sampling",
                      "type": "USED-FOR",
                      "tail": "MINPRAN"
                    },
                    {
                      "head": "error bound",
                      "type": "FEATURE-OF",
                      "tail": "good data"
                    },
                    {
                      "head": "dynamic range",
                      "type": "PART-OF",
                      "tail": "sensor"
                    },
                    {
                      "head": "MINPRAN",
                      "type": "COMPARE",
                      "tail": "least median of squares"
                    }
                  ]
                },
                "preview_note": "Exact submitted structure; at most five items per array and 120 characters per string. Omitted items still count in the score."
              }
            }
          },
          "evidence": {
            "run_id": "new-sources-and-analysis-v3",
            "task_id": "extract-scierc-CVPR_1994_10_abs"
          }
        },
        {
          "id": "extract-scierc-X96-1059",
          "label": "Recognizing named entities in Japanese",
          "task_summary": "Identify dictionaries, rules, and entity categories, and map how Amorph uses them to recognize names, numbers, and time expressions.",
          "source": {
            "dataset": "SciERC",
            "label": "SciERC X96-1059",
            "cluster_id": "scierc:X96-1059",
            "url": "https://nlp.cs.washington.edu/sciIE/",
            "synthetic": false
          },
          "arms": {
            "baseline": {
              "score": 0.3155555555555556,
              "status": "completed",
              "cost_usd": 0.0092356,
              "seconds": 4.837808124999995,
              "schema_valid": true,
              "outcome": "Entity F1 0.311; relation F1 0.320.",
              "metrics": {
                "ent_f1": 0.3111111111111111,
                "rel_f1": 0.32,
                "source_span_precision": 1.0
              },
              "output": {
                "kind": "extraction",
                "value": {
                  "entities": [
                    {
                      "text": "Recognition of proper nouns in Japanese text",
                      "type": "Task"
                    },
                    {
                      "text": "morphological analysis in Japanese text processing",
                      "type": "Task"
                    },
                    {
                      "text": "Japanese information extraction",
                      "type": "Task"
                    },
                    {
                      "text": "Multi-lingual Evaluation Task",
                      "type": "Task"
                    },
                    {
                      "text": "MET",
                      "type": "Task"
                    }
                  ],
                  "relations": [
                    {
                      "head": "morphological analyzer",
                      "type": "USED-FOR",
                      "tail": "recognition and classification of proper names , numerical and temporal expressions"
                    },
                    {
                      "head": "Amorph",
                      "type": "USED-FOR",
                      "tail": "NE items"
                    },
                    {
                      "head": "dictionary lookup",
                      "type": "CONJUNCTION",
                      "tail": "rule application"
                    },
                    {
                      "head": "dictionaries",
                      "type": "USED-FOR",
                      "tail": "Japanese character strings"
                    },
                    {
                      "head": "rules",
                      "type": "USED-FOR",
                      "tail": "segmented strings"
                    }
                  ]
                },
                "preview_note": "Exact submitted structure; at most five items per array and 120 characters per string. Omitted items still count in the score."
              }
            },
            "checklist": {
              "score": 0.4095238095238095,
              "status": "completed",
              "cost_usd": 0.0158816,
              "seconds": 9.416932583999994,
              "schema_valid": true,
              "outcome": "Entity F1 0.533; relation F1 0.286.",
              "metrics": {
                "ent_f1": 0.5333333333333333,
                "rel_f1": 0.2857142857142857,
                "source_span_precision": 1.0
              },
              "output": {
                "kind": "extraction",
                "value": {
                  "entities": [
                    {
                      "text": "Recognition of proper nouns",
                      "type": "Task"
                    },
                    {
                      "text": "morphological analysis",
                      "type": "Task"
                    },
                    {
                      "text": "Japanese text processing",
                      "type": "Task"
                    },
                    {
                      "text": "Japanese information extraction",
                      "type": "Task"
                    },
                    {
                      "text": "Multi-lingual Evaluation Task ( MET )",
                      "type": "Task"
                    }
                  ],
                  "relations": [
                    {
                      "head": "Recognition of proper nouns",
                      "type": "PART-OF",
                      "tail": "morphological analysis"
                    },
                    {
                      "head": "morphological analysis",
                      "type": "PART-OF",
                      "tail": "Japanese text processing"
                    },
                    {
                      "head": "morphological analyzer",
                      "type": "USED-FOR",
                      "tail": "recognition and classification of proper names"
                    },
                    {
                      "head": "morphological analyzer",
                      "type": "USED-FOR",
                      "tail": "Named Entity ( NE ) items"
                    },
                    {
                      "head": "Amorph",
                      "type": "USED-FOR",
                      "tail": "NE items"
                    }
                  ]
                },
                "preview_note": "Exact submitted structure; at most five items per array and 120 characters per string. Omitted items still count in the score."
              }
            },
            "skill": {
              "score": 0.3670977011494253,
              "status": "completed",
              "cost_usd": 0.0244212,
              "seconds": 15.929454917000001,
              "schema_valid": true,
              "outcome": "Entity F1 0.458; relation F1 0.276.",
              "metrics": {
                "ent_f1": 0.4583333333333333,
                "rel_f1": 0.27586206896551724,
                "source_span_precision": 0.8780487804878049
              },
              "output": {
                "kind": "extraction",
                "value": {
                  "entities": [
                    {
                      "text": "proper nouns",
                      "type": "OtherScientificTerm"
                    },
                    {
                      "text": "Japanese text",
                      "type": "Material"
                    },
                    {
                      "text": "morphological analysis",
                      "type": "Task"
                    },
                    {
                      "text": "Japanese text processing",
                      "type": "Task"
                    },
                    {
                      "text": "Japanese information extraction",
                      "type": "Task"
                    }
                  ],
                  "relations": [
                    {
                      "head": "morphological analysis",
                      "type": "PART-OF",
                      "tail": "Japanese text processing"
                    },
                    {
                      "head": "proper nouns",
                      "type": "HYPONYM-OF",
                      "tail": "morphological analysis"
                    },
                    {
                      "head": "Amorph",
                      "type": "USED-FOR",
                      "tail": "NE items"
                    },
                    {
                      "head": "morphological analyzer",
                      "type": "USED-FOR",
                      "tail": "Named Entity (NE) items"
                    },
                    {
                      "head": "dictionary lookup",
                      "type": "CONJUNCTION",
                      "tail": "rule application"
                    }
                  ]
                },
                "preview_note": "Exact submitted structure; at most five items per array and 120 characters per string. Omitted items still count in the score."
              }
            }
          },
          "evidence": {
            "run_id": "new-sources-and-analysis-v3",
            "task_id": "extract-scierc-X96-1059"
          }
        },
        {
          "id": "extract-scierc-E91-1012",
          "label": "Parsing grammars with functional programs",
          "task_summary": "Extract parser types, grammar classes, and memoization methods, and map their relationships to parsing behavior and computational cost.",
          "source": {
            "dataset": "SciERC",
            "label": "SciERC E91-1012",
            "cluster_id": "scierc:E91-1012",
            "url": "https://nlp.cs.washington.edu/sciIE/",
            "synthetic": false
          },
          "arms": {
            "baseline": {
              "score": 0.2545155993431856,
              "status": "completed",
              "cost_usd": 0.019193599999999998,
              "seconds": 12.710045584,
              "schema_valid": true,
              "outcome": "Entity F1 0.414; relation F1 0.095.",
              "metrics": {
                "ent_f1": 0.41379310344827586,
                "rel_f1": 0.09523809523809523,
                "source_span_precision": 1.0
              },
              "output": {
                "kind": "extraction",
                "value": {
                  "entities": [
                    {
                      "text": "LR-parsers",
                      "type": "Method"
                    },
                    {
                      "text": "correctness proof",
                      "type": "OtherScientificTerm"
                    },
                    {
                      "text": "recursive descent parser",
                      "type": "Method"
                    },
                    {
                      "text": "non-LR grammars",
                      "type": "Material"
                    },
                    {
                      "text": "time-complexity",
                      "type": "Metric"
                    }
                  ],
                  "relations": [
                    {
                      "head": "correctness proof",
                      "type": "USED-FOR",
                      "tail": "LR-parsers"
                    },
                    {
                      "head": "recursive descent parser",
                      "type": "HYPONYM-OF",
                      "tail": "LR-parsers"
                    },
                    {
                      "head": "memo-functions",
                      "type": "USED-FOR",
                      "tail": "LR-parsers"
                    },
                    {
                      "head": "time-complexity",
                      "type": "EVALUATE-FOR",
                      "tail": "non-LR grammars"
                    },
                    {
                      "head": "memo-functions",
                      "type": "USED-FOR",
                      "tail": "parse forest"
                    }
                  ]
                },
                "preview_note": "Exact submitted structure; at most five items per array and 120 characters per string. Omitted items still count in the score."
              }
            },
            "checklist": {
              "score": 0.34444444444444444,
              "status": "completed",
              "cost_usd": 0.0081096,
              "seconds": 4.433818334000023,
              "schema_valid": true,
              "outcome": "Entity F1 0.467; relation F1 0.222.",
              "metrics": {
                "ent_f1": 0.4666666666666667,
                "rel_f1": 0.2222222222222222,
                "source_span_precision": 1.0
              },
              "output": {
                "kind": "extraction",
                "value": {
                  "entities": [
                    {
                      "text": "LR-parsers",
                      "type": "Method"
                    },
                    {
                      "text": "correctness proof",
                      "type": "OtherScientificTerm"
                    },
                    {
                      "text": "recursive descent parser",
                      "type": "Method"
                    },
                    {
                      "text": "non-LR grammars",
                      "type": "Material"
                    },
                    {
                      "text": "time-complexity",
                      "type": "Metric"
                    }
                  ],
                  "relations": [
                    {
                      "head": "LR-parsers",
                      "type": "USED-FOR",
                      "tail": "non-LR grammars"
                    },
                    {
                      "head": "LR-parsers",
                      "type": "HYPONYM-OF",
                      "tail": "recursive descent parser"
                    },
                    {
                      "head": "time-complexity",
                      "type": "EVALUATE-FOR",
                      "tail": "non-LR grammars"
                    },
                    {
                      "head": "memo-functions",
                      "type": "USED-FOR",
                      "tail": "parse forest"
                    },
                    {
                      "head": "recursive ascent parsers",
                      "type": "COMPARE",
                      "tail": "LR ( 0 ) grammars"
                    }
                  ]
                },
                "preview_note": "Exact submitted structure; at most five items per array and 120 characters per string. Omitted items still count in the score."
              }
            },
            "skill": {
              "score": 0.2624521072796935,
              "status": "completed",
              "cost_usd": 0.0147092,
              "seconds": 8.94677362500002,
              "schema_valid": true,
              "outcome": "Entity F1 0.414; relation F1 0.111.",
              "metrics": {
                "ent_f1": 0.41379310344827586,
                "rel_f1": 0.1111111111111111,
                "source_span_precision": 1.0
              },
              "output": {
                "kind": "extraction",
                "value": {
                  "entities": [
                    {
                      "text": "LR-parsers",
                      "type": "Method"
                    },
                    {
                      "text": "correctness proof",
                      "type": "OtherScientificTerm"
                    },
                    {
                      "text": "recursive descent parser",
                      "type": "Method"
                    },
                    {
                      "text": "non-LR grammars",
                      "type": "Material"
                    },
                    {
                      "text": "time-complexity",
                      "type": "Metric"
                    }
                  ],
                  "relations": [
                    {
                      "head": "LR-parsers",
                      "type": "COMPARE",
                      "tail": "recursive descent parser"
                    },
                    {
                      "head": "time-complexity",
                      "type": "FEATURE-OF",
                      "tail": "non-LR grammars"
                    },
                    {
                      "head": "memo-functions",
                      "type": "USED-FOR",
                      "tail": "parse forest"
                    },
                    {
                      "head": "recursive ascent parsers",
                      "type": "COMPARE",
                      "tail": "LR-parsers"
                    },
                    {
                      "head": "regular expressions",
                      "type": "FEATURE-OF",
                      "tail": "Extended CF grammars"
                    }
                  ]
                },
                "preview_note": "Exact submitted structure; at most five items per array and 120 characters per string. Omitted items still count in the score."
              }
            }
          },
          "evidence": {
            "run_id": "new-sources-and-analysis-v3",
            "task_id": "extract-scierc-E91-1012"
          }
        },
        {
          "id": "extract-scierc-P06-1013",
          "label": "Combining systems to resolve word meanings",
          "task_summary": "Identify unsupervised word-sense systems, voting and arbiter strategies, and evaluation datasets, then map how the ensemble components work together.",
          "source": {
            "dataset": "SciERC",
            "label": "SciERC P06-1013",
            "cluster_id": "scierc:P06-1013",
            "url": "https://nlp.cs.washington.edu/sciIE/",
            "synthetic": false
          },
          "arms": {
            "baseline": {
              "score": 0.0,
              "status": "completed",
              "cost_usd": 0.0110236,
              "seconds": 6.915165833000003,
              "schema_valid": false,
              "outcome": "The submitted structure failed the required schema and received zero.",
              "metrics": {
                "ent_f1": 0.0,
                "rel_f1": 0.0,
                "source_span_precision": 0.0
              },
              "output": {
                "kind": "extraction",
                "value": {
                  "parameter_name": "\n",
                  "entities": [
                    {
                      "text": "Combination methods",
                      "type": "Method"
                    },
                    {
                      "text": "system performance",
                      "type": "Metric"
                    },
                    {
                      "text": "system combination",
                      "type": "Method"
                    },
                    {
                      "text": "unsupervised WSD",
                      "type": "Task"
                    },
                    {
                      "text": "voting - and arbiter-based combination strategies",
                      "type": "Method"
                    }
                  ],
                  "relations": [
                    {
                      "head": "system combination",
                      "type": "USED-FOR",
                      "tail": "unsupervised WSD"
                    },
                    {
                      "head": "voting - and arbiter-based combination strategies",
                      "type": "USED-FOR",
                      "tail": "unsupervised WSD systems"
                    },
                    {
                      "head": "predominant senses",
                      "type": "USED-FOR",
                      "tail": "Combination methods"
                    },
                    {
                      "head": "raw text",
                      "type": "USED-FOR",
                      "tail": "predominant senses"
                    },
                    {
                      "head": "SemCor",
                      "type": "CONJUNCTION",
                      "tail": "Senseval-3 data sets"
                    }
                  ]
                },
                "preview_note": "Exact submitted structure; at most five items per array and 120 characters per string. Omitted items still count in the score."
              }
            },
            "checklist": {
              "score": 0.0,
              "status": "failed",
              "cost_usd": 0.0119596,
              "seconds": 6.996814749999999,
              "schema_valid": null,
              "outcome": "The response used multiple final tool calls instead of one answer.",
              "metrics": {},
              "output": null
            },
            "skill": {
              "score": 0.41666666666666663,
              "status": "completed",
              "cost_usd": 0.008479200000000001,
              "seconds": 5.131480415999988,
              "schema_valid": true,
              "outcome": "Entity F1 0.667; relation F1 0.167.",
              "metrics": {
                "ent_f1": 0.6666666666666666,
                "rel_f1": 0.16666666666666666,
                "source_span_precision": 1.0
              },
              "output": {
                "kind": "extraction",
                "value": {
                  "entities": [
                    {
                      "text": "Combination methods",
                      "type": "Method"
                    },
                    {
                      "text": "system performance",
                      "type": "Metric"
                    },
                    {
                      "text": "unsupervised WSD",
                      "type": "Task"
                    },
                    {
                      "text": "voting - and arbiter-based combination strategies",
                      "type": "Method"
                    },
                    {
                      "text": "unsupervised WSD systems",
                      "type": "Method"
                    }
                  ],
                  "relations": [
                    {
                      "head": "Combination methods",
                      "type": "USED-FOR",
                      "tail": "system performance"
                    },
                    {
                      "head": "voting - and arbiter-based combination strategies",
                      "type": "USED-FOR",
                      "tail": "unsupervised WSD systems"
                    },
                    {
                      "head": "predominant senses",
                      "type": "USED-FOR",
                      "tail": "voting - and arbiter-based combination strategies"
                    },
                    {
                      "head": "predominant senses",
                      "type": "HYPONYM-OF",
                      "tail": "raw text"
                    },
                    {
                      "head": "SemCor",
                      "type": "CONJUNCTION",
                      "tail": "Senseval-3 data sets"
                    }
                  ]
                },
                "preview_note": "Exact submitted structure; at most five items per array and 120 characters per string. Omitted items still count in the score."
              }
            }
          },
          "evidence": {
            "run_id": "new-sources-and-analysis-v3",
            "task_id": "extract-scierc-P06-1013"
          }
        }
      ]
    },
    {
      "id": "analyze",
      "title": "Run a statistical analysis",
      "question": "Does statistical guidance improve executable answers to prescribed analyses?",
      "metric": "Original and changed-input pass rate",
      "unit": "pass_rate",
      "tasks": 8,
      "source_clusters": 5,
      "arms": {
        "baseline": {
          "label": "No added skill",
          "score": 0.875,
          "attempts": 8,
          "completed": 7,
          "failed": 1,
          "cost_usd": 0.23212199999999997,
          "mean_seconds": 20.306690911625,
          "schema_valid_count": 7,
          "schema_observed_count": 7,
          "reference_reads": 0
        },
        "checklist": {
          "label": "Model + checklist",
          "score": 1.0,
          "attempts": 8,
          "completed": 8,
          "failed": 0,
          "cost_usd": 0.133108,
          "mean_seconds": 13.126327333375006,
          "schema_valid_count": 8,
          "schema_observed_count": 8,
          "reference_reads": 0
        },
        "skill": {
          "label": "Model + skill",
          "score": 1.0,
          "attempts": 8,
          "completed": 8,
          "failed": 0,
          "cost_usd": 0.22154449999999998,
          "mean_seconds": 17.363840958375,
          "schema_valid_count": 8,
          "schema_observed_count": 8,
          "reference_reads": 0
        }
      },
      "skill": {
        "id": "statistical-analysis",
        "kind": "upstream_package",
        "url": "https://github.com/K-Dense-AI/scientific-agent-skills/tree/330c8e764435a731eff571e3efdda70b363d0792/skills/statistical-analysis",
        "revision": "330c8e764435a731eff571e3efdda70b363d0792",
        "usage": "SKILL.md supplied; no explicit reference-file reads recorded."
      },
      "interpretation": {
        "summary": "Baseline passed seven of eight tasks; checklist and skill passed eight. The extra pass reflects a baseline tool-call failure, not a wrong statistical result.",
        "evidence_case_ids": [
          "analyze-penguin-welch",
          "analyze-penguin-adjusted"
        ]
      },
      "limitations": [
        "Eight tasks share four public datasets and one synthetic fixture; the statistical methods are prescribed.",
        "Passing programs ran on original and hidden changed data, but this does not establish general analysis competence.",
        "The fixed environment supplies NumPy, SciPy and pandas; the full upstream dependency stack is unavailable.",
        "Multiplicity checks do not change the rejection set; a test-only outcome perturbation would strengthen the leakage check."
      ],
      "cases": [
        {
          "id": "analyze-penguin-welch",
          "label": "Penguins: unequal-variance comparison",
          "task_summary": "Estimate male minus female Adelie body mass with Welch uncertainty and an effect size.",
          "source": {
            "dataset": "penguins.csv",
            "label": "Penguins: unequal-variance comparison",
            "cluster_id": "penguins",
            "url": "https://raw.githubusercontent.com/allisonhorst/palmerpenguins/main/inst/extdata/penguins.csv",
            "synthetic": false
          },
          "arms": {
            "baseline": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.011368,
              "seconds": 9.591504792000002,
              "schema_valid": true,
              "outcome": "Both original and changed-input executions passed.",
              "metrics": {
                "original_correct": 1.0,
                "perturbed_correct": 1.0,
                "executable": 1.0
              },
              "output": {
                "kind": "executed_analysis",
                "value": {
                  "method": "welch_independent",
                  "n_left": 73,
                  "n_right": 73,
                  "difference": 674.6575342465753,
                  "se": 51.39744312894034,
                  "df": 135.69357893530383,
                  "p_value": 6.402319748031793e-26,
                  "ci_low": 573.0139063407537,
                  "ci_high": 776.301162152397,
                  "hedges_g": 2.1613425922546154
                },
                "value_source": "Original-input execution stdout",
                "code_excerpt": "\nimport pandas as pd, numpy as np, json\nfrom scipy import stats\n\ndf = pd.read_csv(\"data.csv\")\ndf = df[df['sex'].isin(['male','female'])]\ndf = df.dropna(subset=['body_mass_g'])\ndf['body_mass_g'] = pd.to_numeric(df['body_mass_g'], errors='coerce')\ndf = df.dropna(subset=['body_mass_g'])\n\nm = df.loc[df['sex']=='male','body_mass_g'].values\nf = df.loc[df['sex']=='female','body_mass_g'].values\n\nn1, n2 = len(m), len(f)\nmean1, mean2 = m.mean(), f.mean()\nvar1, var2 = m.var(ddof=1), f.var(ddof=1)\n\ndiff = mean1 - mean2\nse = np.sqrt(var1/n1 + var2/n2)\ndf_welch = (var1/n1 + var2/n2)**2 / ((var1/n1)**2/(n1-1) + (var2/n2)**2/(n2-1))\n\nt = diff / se\np = 2*stats.t.sf(abs(t), df_welch)\n\ntcrit = stats.t.ppf(0.975, df_welch)\nci_low = diff - tcrit*se\nci_high = diff + tcrit*se\n\nsp = np.sqrt(((n1-1)*var1 + (n2-1)*var2)/(n1+n2-2))\nJ = 1 - 3/(4*(n1+n2)-9)\ng = J * diff / sp\n\nresult = {\n    \"method\": \"welch_independent\",\n    \"n_left\": n1,\n    \"n_right\": n2,\n    \"difference\": float(diff),\n    \"se\": float(se),\n    \"df\": float(df_welch),\n    \"p_value\": float(p),\n    \"ci_low\": float(ci_low),\n    \"ci_high\": float(ci_high),\n    \"hedges_g\": float(g)\n}\nprint(json.dumps(result))\n",
                "code_truncated": false,
                "checks": [
                  {
                    "input": "original",
                    "passed": true
                  },
                  {
                    "input": "changed",
                    "passed": true
                  }
                ],
                "execution_warnings": false
              }
            },
            "checklist": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.012654,
              "seconds": 10.067396625,
              "schema_valid": true,
              "outcome": "Both original and changed-input executions passed.",
              "metrics": {
                "original_correct": 1.0,
                "perturbed_correct": 1.0,
                "executable": 1.0
              },
              "output": {
                "kind": "executed_analysis",
                "value": {
                  "method": "welch_independent",
                  "n_left": 73,
                  "n_right": 73,
                  "difference": 674.6575342465753,
                  "se": 51.39744312894034,
                  "df": 135.69357893530383,
                  "p_value": 6.402319748031793e-26,
                  "ci_low": 573.0139063407537,
                  "ci_high": 776.301162152397,
                  "hedges_g": 2.1613425922546154
                },
                "value_source": "Original-input execution stdout",
                "code_excerpt": "\nimport pandas as pd, numpy as np, json\nfrom scipy import stats\n\ndf = pd.read_csv(\"data.csv\")\ndf = df[df['sex'].isin(['male','female'])]\ndf = df.dropna(subset=['body_mass_g'])\ndf['body_mass_g'] = pd.to_numeric(df['body_mass_g'], errors='coerce')\ndf = df.dropna(subset=['body_mass_g'])\n\nm = df.loc[df['sex']=='male','body_mass_g'].astype(float)\nf = df.loc[df['sex']=='female','body_mass_g'].astype(float)\n\nn1, n2 = len(m), len(f)\nmean1, mean2 = m.mean(), f.mean()\nvar1, var2 = m.var(ddof=1), f.var(ddof=1)\n\ndiff = mean1 - mean2\nse = np.sqrt(var1/n1 + var2/n2)\ndf_welch = (var1/n1 + var2/n2)**2 / ((var1/n1)**2/(n1-1) + (var2/n2)**2/(n2-1))\n\nt_stat = diff/se\np_value = 2*stats.t.sf(np.abs(t_stat), df_welch)\n\nci_low = diff - stats.t.ppf(0.975, df_welch)*se\nci_high = diff + stats.t.ppf(0.975, df_welch)*se\n\npooled_sd = np.sqrt(((n1-1)*var1 + (n2-1)*var2)/(n1+n2-2))\nJ = 1 - 3/(4*(n1+n2)-9)\nhedges_g = J * diff / pooled_sd\n\nresult = {\n    \"method\": \"welch_independent\",\n    \"n_left\": int(n1),\n    \"n_right\": int(n2),\n    \"difference\": float(diff),\n    \"se\": float(se),\n    \"df\": float(df_welch),\n    \"p_value\": float(p_value),\n    \"ci_low\": float(ci_low),\n    \"ci_high\": float(ci_high),\n    \"hedges_g\": …",
                "code_truncated": true,
                "checks": [
                  {
                    "input": "original",
                    "passed": true
                  },
                  {
                    "input": "changed",
                    "passed": true
                  }
                ],
                "execution_warnings": false
              }
            },
            "skill": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.044790899999999995,
              "seconds": 16.317282750000004,
              "schema_valid": true,
              "outcome": "Both original and changed-input executions passed.",
              "metrics": {
                "original_correct": 1.0,
                "perturbed_correct": 1.0,
                "executable": 1.0
              },
              "output": {
                "kind": "executed_analysis",
                "value": {
                  "method": "welch_independent",
                  "n_left": 73,
                  "n_right": 73,
                  "difference": 674.6575342465753,
                  "se": 51.39744312894034,
                  "df": 135.69357893530383,
                  "p_value": 6.402319748031793e-26,
                  "ci_low": 573.0139063407537,
                  "ci_high": 776.301162152397,
                  "hedges_g": 2.1613425922546154
                },
                "value_source": "Original-input execution stdout",
                "code_excerpt": "\nimport pandas as pd, numpy as np\nfrom scipy import stats\nimport json\n\ndf = pd.read_csv('data.csv')\ndf = df[df['sex'].isin(['male','female'])]\ndf = df.dropna(subset=['body_mass_g'])\nmale = df[df['sex']=='male']['body_mass_g'].astype(float)\nfemale = df[df['sex']=='female']['body_mass_g'].astype(float)\n\nn1,n2 = len(male), len(female)\nm1,m2 = male.mean(), female.mean()\nv1,v2 = male.var(ddof=1), female.var(ddof=1)\n\ndiff = m1-m2\nse = np.sqrt(v1/n1+v2/n2)\ndf_welch = (v1/n1+v2/n2)**2 / ((v1/n1)**2/(n1-1)+(v2/n2)**2/(n2-1))\nt = diff/se\np = 2*stats.t.sf(abs(t), df_welch)\nci_low = diff - stats.t.ppf(0.975, df_welch)*se\nci_high = diff + stats.t.ppf(0.975, df_welch)*se\n\nsp = np.sqrt(((n1-1)*v1+(n2-1)*v2)/(n1+n2-2))\nd = diff/sp\nJ = 1-3/(4*(n1+n2)-9)\ng = d*J\n\nres = {\n\"method\":\"welch_independent\",\n\"n_left\": n1,\n\"n_right\": n2,\n\"difference\": diff,\n\"se\": se,\n\"df\": df_welch,\n\"p_value\": p,\n\"ci_low\": ci_low,\n\"ci_high\": ci_high,\n\"hedges_g\": g\n}\nprint(json.dumps(res))\n",
                "code_truncated": false,
                "checks": [
                  {
                    "input": "original",
                    "passed": true
                  },
                  {
                    "input": "changed",
                    "passed": true
                  }
                ],
                "execution_warnings": false
              }
            }
          },
          "evidence": {
            "run_id": "four-track-canary-v3",
            "task_id": "analyze-penguin-welch"
          }
        },
        {
          "id": "analyze-norris-calibration",
          "label": "Norris: calibration regression",
          "task_summary": "Fit an intercept and slope to the calibration observations and report classical uncertainty.",
          "source": {
            "dataset": "Norris.dat",
            "label": "Norris: calibration regression",
            "cluster_id": "nist_norris",
            "url": "https://www.itl.nist.gov/div898/strd/lls/data/LINKS/DATA/Norris.dat",
            "synthetic": false
          },
          "arms": {
            "baseline": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.018722000000000003,
              "seconds": 29.300928791999993,
              "schema_valid": true,
              "outcome": "Both original and changed-input executions passed.",
              "metrics": {
                "original_correct": 1.0,
                "perturbed_correct": 1.0,
                "executable": 1.0
              },
              "output": {
                "kind": "executed_analysis",
                "value": {
                  "method": "ols_with_intercept",
                  "n": 36,
                  "intercept": -0.2623230737740414,
                  "slope": 1.0021168180204545,
                  "slope_ci_low": 1.0012433657355782,
                  "slope_ci_high": 1.0029902703053308,
                  "residual_sd": 0.884796396144381,
                  "r_squared": 0.9999937458837117
                },
                "value_source": "Original-input execution stdout",
                "code_excerpt": "\nimport pandas as pd, numpy as np\nfrom scipy import stats\nimport json\n\ndf = pd.read_csv(\"data.csv\")\nx = df['x'].values.astype(float)\ny = df['y'].values.astype(float)\nn = len(x)\n\nxm = x.mean(); ym = y.mean()\nSxx = ((x-xm)**2).sum()\nSxy = ((x-xm)*(y-ym)).sum()\nslope = Sxy/Sxx\nintercept = ym - slope*xm\nyhat = intercept + slope*x\nresid = y - yhat\nss_res = (resid**2).sum()\ndof = n-2\nresid_var = ss_res/dof\nresid_sd = np.sqrt(resid_var)\nse_slope = np.sqrt(resid_var/Sxx)\ntcrit = stats.t.ppf(0.975, dof)\nci_low = slope - tcrit*se_slope\nci_high = slope + tcrit*se_slope\nss_tot = ((y-ym)**2).sum()\nr2 = 1 - ss_res/ss_tot\n\nresult = {\n    \"method\": \"ols_with_intercept\",\n    \"n\": n,\n    \"intercept\": intercept,\n    \"slope\": slope,\n    \"slope_ci_low\": ci_low,\n    \"slope_ci_high\": ci_high,\n    \"residual_sd\": resid_sd,\n    \"r_squared\": r2\n}\nprint(json.dumps(result))\n",
                "code_truncated": false,
                "checks": [
                  {
                    "input": "original",
                    "passed": true
                  },
                  {
                    "input": "changed",
                    "passed": true
                  }
                ],
                "execution_warnings": false
              }
            },
            "checklist": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.0168096,
              "seconds": 15.912146750000005,
              "schema_valid": true,
              "outcome": "Both original and changed-input executions passed.",
              "metrics": {
                "original_correct": 1.0,
                "perturbed_correct": 1.0,
                "executable": 1.0
              },
              "output": {
                "kind": "executed_analysis",
                "value": {
                  "method": "ols_with_intercept",
                  "n": 36,
                  "intercept": -0.2623230737737886,
                  "slope": 1.0021168180204536,
                  "slope_ci_low": 1.0012433657355773,
                  "slope_ci_high": 1.00299027030533,
                  "residual_sd": 0.8847963961443822,
                  "r_squared": 0.9999937458837117
                },
                "value_source": "Original-input execution stdout",
                "code_excerpt": "\nimport pandas as pd, numpy as np\nfrom scipy import stats\nimport json\n\ndf = pd.read_csv(\"data.csv\")\ndf = df.dropna(subset=[\"y\",\"x\"])\nx = df[\"x\"].values.astype(float)\ny = df[\"y\"].values.astype(float)\nn = len(x)\n\nX = np.column_stack([np.ones(n), x])\nbeta, res, rank, sv = np.linalg.lstsq(X, y, rcond=None)\nintercept, slope = beta\nyhat = X @ beta\nresid = y - yhat\ndof = n - 2\nrss = np.sum(resid**2)\nresidual_sd = np.sqrt(rss/dof)\nXtX_inv = np.linalg.inv(X.T @ X)\nse_slope = residual_sd * np.sqrt(XtX_inv[1,1])\ntcrit = stats.t.ppf(0.975, dof)\nci_low = slope - tcrit*se_slope\nci_high = slope + tcrit*se_slope\n\nss_tot = np.sum((y-np.mean(y))**2)\nr_squared = 1 - rss/ss_tot\n\nresult = {\n    \"method\": \"ols_with_intercept\",\n    \"n\": n,\n    \"intercept\": float(intercept),\n    \"slope\": float(slope),\n    \"slope_ci_low\": float(ci_low),\n    \"slope_ci_high\": float(ci_high),\n    \"residual_sd\": float(residual_sd),\n    \"r_squared\": float(r_squared)\n}\nprint(json.dumps(result))\n",
                "code_truncated": false,
                "checks": [
                  {
                    "input": "original",
                    "passed": true
                  },
                  {
                    "input": "changed",
                    "passed": true
                  }
                ],
                "execution_warnings": false
              }
            },
            "skill": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.0194628,
              "seconds": 26.130196208,
              "schema_valid": true,
              "outcome": "Both original and changed-input executions passed.",
              "metrics": {
                "original_correct": 1.0,
                "perturbed_correct": 1.0,
                "executable": 1.0
              },
              "output": {
                "kind": "executed_analysis",
                "value": {
                  "method": "ols_with_intercept",
                  "n": 36,
                  "intercept": -0.2623230737737886,
                  "slope": 1.0021168180204536,
                  "slope_ci_low": 1.0012433657355773,
                  "slope_ci_high": 1.00299027030533,
                  "residual_sd": 0.8847963961443822,
                  "r_squared": 0.9999937458837117
                },
                "value_source": "Original-input execution stdout",
                "code_excerpt": "\nimport pandas as pd, numpy as np\nfrom scipy import stats\nimport json\n\ndf = pd.read_csv(\"data.csv\")\nx = df['x'].values\ny = df['y'].values\nn = len(x)\n\nX = np.column_stack([np.ones(n), x])\nbeta, res, rank, sv = np.linalg.lstsq(X, y, rcond=None)\nintercept, slope = beta\n\nyhat = X @ beta\nresid = y - yhat\ndof = n - 2\nrss = np.sum(resid**2)\nresidual_sd = np.sqrt(rss/dof)\n\nXtX_inv = np.linalg.inv(X.T @ X)\nse = np.sqrt(residual_sd**2 * np.diag(XtX_inv))\nse_slope = se[1]\n\ntcrit = stats.t.ppf(0.975, dof)\nci_low = slope - tcrit*se_slope\nci_high = slope + tcrit*se_slope\n\nss_tot = np.sum((y-np.mean(y))**2)\nr2 = 1 - rss/ss_tot\n\nresult = {\n    \"method\": \"ols_with_intercept\",\n    \"n\": n,\n    \"intercept\": float(intercept),\n    \"slope\": float(slope),\n    \"slope_ci_low\": float(ci_low),\n    \"slope_ci_high\": float(ci_high),\n    \"residual_sd\": float(residual_sd),\n    \"r_squared\": float(r2)\n}\nprint(json.dumps(result))\n",
                "code_truncated": false,
                "checks": [
                  {
                    "input": "original",
                    "passed": true
                  },
                  {
                    "input": "changed",
                    "passed": true
                  }
                ],
                "execution_warnings": false
              }
            }
          },
          "evidence": {
            "run_id": "four-track-canary-v3",
            "task_id": "analyze-norris-calibration"
          }
        },
        {
          "id": "analyze-mavro-blocks",
          "label": "Mavro: ordered measurements",
          "task_summary": "Compare independent-observation uncertainty with five consecutive block means.",
          "source": {
            "dataset": "Mavro.dat",
            "label": "Mavro: ordered measurements",
            "cluster_id": "nist_mavro",
            "url": "https://www.itl.nist.gov/div898/strd/univ/data/Mavro.dat",
            "synthetic": false
          },
          "arms": {
            "baseline": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.019268,
              "seconds": 12.464235292000012,
              "schema_valid": true,
              "outcome": "Both original and changed-input executions passed.",
              "metrics": {
                "original_correct": 1.0,
                "perturbed_correct": 1.0,
                "executable": 1.0
              },
              "output": {
                "kind": "executed_analysis",
                "value": {
                  "method": "ordered_block_mean_t",
                  "n": 50,
                  "mean": 2.0018559999999996,
                  "sample_sd": 0.0004291234540030854,
                  "lag1_rho": 0.9379891834382406,
                  "iid_se": 6.068722085835504e-05,
                  "block_se": 0.0001895415521726317,
                  "block_ci_low": 2.0013297482852392,
                  "block_ci_high": 2.002382251714762
                },
                "value_source": "Original-input execution stdout",
                "code_excerpt": "\nimport pandas as pd, numpy as np\nfrom scipy import stats\nimport json\n\ndf = pd.read_csv('data.csv')\ndf = df.sort_values('sequence')\ny = df['transmittance'].values\nn = len(y)\nmean = y.mean()\nsample_sd = y.std(ddof=1)\niid_se = sample_sd/np.sqrt(n)\nnum = np.sum((y[1:]-mean)*(y[:-1]-mean))\nden = np.sum((y-mean)**2)\nlag1_rho = num/den\n\nblocks = y[:50].reshape(5,10)\nblock_means = blocks.mean(axis=1)\nblock_sd = block_means.std(ddof=1)\nblock_se = block_sd/np.sqrt(5)\ntcrit = stats.t.ppf(0.975,4)\nbm = block_means.mean()\nci_low = bm - tcrit*block_se\nci_high = bm + tcrit*block_se\n\nres = {\n \"method\":\"ordered_block_mean_t\",\n \"n\": int(n),\n \"mean\": float(mean),\n \"sample_sd\": float(sample_sd),\n \"lag1_rho\": float(lag1_rho),\n \"iid_se\": float(iid_se),\n \"block_se\": float(block_se),\n \"block_ci_low\": float(ci_low),\n \"block_ci_high\": float(ci_high)\n}\nprint(json.dumps(res))\n",
                "code_truncated": false,
                "checks": [
                  {
                    "input": "original",
                    "passed": true
                  },
                  {
                    "input": "changed",
                    "passed": true
                  }
                ],
                "execution_warnings": false
              }
            },
            "checklist": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.0079828,
              "seconds": 8.296932333,
              "schema_valid": true,
              "outcome": "Both original and changed-input executions passed.",
              "metrics": {
                "original_correct": 1.0,
                "perturbed_correct": 1.0,
                "executable": 1.0
              },
              "output": {
                "kind": "executed_analysis",
                "value": {
                  "method": "ordered_block_mean_t",
                  "n": 50,
                  "mean": 2.0018559999999996,
                  "sample_sd": 0.0004291234540030854,
                  "lag1_rho": 0.9379891834382406,
                  "iid_se": 6.068722085835504e-05,
                  "block_se": 0.0001895415521726317,
                  "block_ci_low": 2.0013297482852392,
                  "block_ci_high": 2.002382251714762
                },
                "value_source": "Original-input execution stdout",
                "code_excerpt": "\nimport pandas as pd, numpy as np, json\nfrom scipy import stats\n\ndf = pd.read_csv(\"data.csv\").sort_values(\"sequence\")\ny = df[\"transmittance\"].to_numpy(dtype=float)\nn = len(y)\nmean = y.mean()\nsample_sd = y.std(ddof=1)\niid_se = sample_sd/np.sqrt(n)\n\nnum = np.sum((y[1:]-mean)*(y[:-1]-mean))\nden = np.sum((y-mean)**2)\nlag1_rho = num/den\n\nnblocks = 5\nbsize = n//nblocks\nblocks = y[:nblocks*bsize].reshape(nblocks,bsize)\nblock_means = blocks.mean(axis=1)\nbm_mean = block_means.mean()\nbm_sd = block_means.std(ddof=1)\nblock_se = bm_sd/np.sqrt(nblocks)\ntcrit = stats.t.ppf(0.975, 4)\nblock_ci_low = bm_mean - tcrit*block_se\nblock_ci_high = bm_mean + tcrit*block_se\n\nresult = {\n    \"method\": \"ordered_block_mean_t\",\n    \"n\": int(n),\n    \"mean\": float(mean),\n    \"sample_sd\": float(sample_sd),\n    \"lag1_rho\": float(lag1_rho),\n    \"iid_se\": float(iid_se),\n    \"block_se\": float(block_se),\n    \"block_ci_low\": float(block_ci_low),\n    \"block_ci_high\": float(block_ci_high)\n}\nprint(json.dumps(result))\n",
                "code_truncated": false,
                "checks": [
                  {
                    "input": "original",
                    "passed": true
                  },
                  {
                    "input": "changed",
                    "passed": true
                  }
                ],
                "execution_warnings": false
              }
            },
            "skill": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.019770799999999998,
              "seconds": 14.647173083999974,
              "schema_valid": true,
              "outcome": "Both original and changed-input executions passed.",
              "metrics": {
                "original_correct": 1.0,
                "perturbed_correct": 1.0,
                "executable": 1.0
              },
              "output": {
                "kind": "executed_analysis",
                "value": {
                  "method": "ordered_block_mean_t",
                  "n": 50,
                  "mean": 2.0018559999999996,
                  "sample_sd": 0.0004291234540030854,
                  "lag1_rho": 0.9379891834382406,
                  "iid_se": 6.068722085835504e-05,
                  "block_se": 0.0001895415521726317,
                  "block_ci_low": 2.0013297482852392,
                  "block_ci_high": 2.002382251714762
                },
                "value_source": "Original-input execution stdout",
                "code_excerpt": "\nimport pandas as pd, numpy as np\nfrom scipy import stats\nimport json\n\ndf = pd.read_csv(\"data.csv\").sort_values(\"sequence\")\ny = df[\"transmittance\"].values\nn = len(y)\nmean = y.mean()\nsample_sd = y.std(ddof=1)\niid_se = sample_sd/np.sqrt(n)\n\nnum = np.sum((y[1:]-mean)*(y[:-1]-mean))\nden = np.sum((y-mean)**2)\nlag1_rho = num/den\n\nblocks = y.reshape(5,10)\nblock_means = blocks.mean(axis=1)\nblock_sd = block_means.std(ddof=1)\nblock_se = block_sd/np.sqrt(5)\ntcrit = stats.t.ppf(0.975,4)\nblock_ci_low = block_means.mean() - tcrit*block_se\nblock_ci_high = block_means.mean() + tcrit*block_se\n\nres = {\n    \"method\": \"ordered_block_mean_t\",\n    \"n\": int(n),\n    \"mean\": float(mean),\n    \"sample_sd\": float(sample_sd),\n    \"lag1_rho\": float(lag1_rho),\n    \"iid_se\": float(iid_se),\n    \"block_se\": float(block_se),\n    \"block_ci_low\": float(block_ci_low),\n    \"block_ci_high\": float(block_ci_high),\n}\nprint(json.dumps(res, indent=2))\n",
                "code_truncated": false,
                "checks": [
                  {
                    "input": "original",
                    "passed": true
                  },
                  {
                    "input": "changed",
                    "passed": true
                  }
                ],
                "execution_warnings": false
              }
            }
          },
          "evidence": {
            "run_id": "four-track-canary-v3",
            "task_id": "analyze-mavro-blocks"
          }
        },
        {
          "id": "analyze-silver-instruments",
          "label": "Silver: instrument comparison",
          "task_summary": "Compare two independent instruments while preserving very small atomic-weight differences.",
          "source": {
            "dataset": "AtmWtAg.dat",
            "label": "Silver: instrument comparison",
            "cluster_id": "nist_silver",
            "url": "https://www.itl.nist.gov/div898/strd/anova/AtmWtAg.dat",
            "synthetic": false
          },
          "arms": {
            "baseline": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.009892,
              "seconds": 8.844608707999981,
              "schema_valid": true,
              "outcome": "Both original and changed-input executions passed.",
              "metrics": {
                "original_correct": 1.0,
                "perturbed_correct": 1.0,
                "executable": 1.0
              },
              "output": {
                "kind": "executed_analysis",
                "value": {
                  "method": "welch_independent",
                  "n_left": 24,
                  "n_right": 24,
                  "difference": -1.7412500028513023e-05,
                  "se": 4.360389250340865e-06,
                  "df": 43.25183428311259,
                  "p_value": 0.00024855676056512177,
                  "ci_low": -2.620458259250799e-05,
                  "ci_high": -8.620417464518057e-06,
                  "hedges_g": -1.1338788698859152
                },
                "value_source": "Original-input execution stdout",
                "code_excerpt": "\nimport pandas as pd, numpy as np\nfrom scipy import stats\nimport json\n\ndf = pd.read_csv(\"data.csv\")\ng1 = df[df.instrument==2][\"atomic_weight\"].values  # left\ng2 = df[df.instrument==1][\"atomic_weight\"].values  # right\n\nn1, n2 = len(g1), len(g2)\nm1, m2 = g1.mean(), g2.mean()\nv1, v2 = g1.var(ddof=1), g2.var(ddof=1)\n\ndiff = m1 - m2\nse = np.sqrt(v1/n1 + v2/n2)\ndf_welch = (v1/n1 + v2/n2)**2 / ((v1/n1)**2/(n1-1) + (v2/n2)**2/(n2-1))\n\nt_stat = diff/se\np_value = 2*stats.t.sf(abs(t_stat), df_welch)\n\nci_low = diff - stats.t.ppf(0.975, df_welch)*se\nci_high = diff + stats.t.ppf(0.975, df_welch)*se\n\nsp = np.sqrt(((n1-1)*v1+(n2-1)*v2)/(n1+n2-2))\nJ = 1 - 3/(4*(n1+n2)-9)\nhedges_g = J*diff/sp\n\nresult = {\n    \"method\": \"welch_independent\",\n    \"n_left\": n1,\n    \"n_right\": n2,\n    \"difference\": float(diff),\n    \"se\": float(se),\n    \"df\": float(df_welch),\n    \"p_value\": float(p_value),\n    \"ci_low\": float(ci_low),\n    \"ci_high\": float(ci_high),\n    \"hedges_g\": float(hedges_g)\n}\n\nprint(json.dumps(result))\n",
                "code_truncated": false,
                "checks": [
                  {
                    "input": "original",
                    "passed": true
                  },
                  {
                    "input": "changed",
                    "passed": true
                  }
                ],
                "execution_warnings": false
              }
            },
            "checklist": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.0091668,
              "seconds": 9.717503459,
              "schema_valid": true,
              "outcome": "Both original and changed-input executions passed.",
              "metrics": {
                "original_correct": 1.0,
                "perturbed_correct": 1.0,
                "executable": 1.0
              },
              "output": {
                "kind": "executed_analysis",
                "value": {
                  "method": "welch_independent",
                  "n_left": 24,
                  "n_right": 24,
                  "difference": -1.7412500028513023e-05,
                  "se": 4.360389250340865e-06,
                  "df": 43.25183428311259,
                  "p_value": 0.00024855676056512177,
                  "ci_low": -2.620458259250799e-05,
                  "ci_high": -8.620417464518057e-06,
                  "hedges_g": -1.133878869885915
                },
                "value_source": "Original-input execution stdout",
                "code_excerpt": "\nimport pandas as pd, numpy as np\nfrom scipy import stats\nimport json\n\ndf = pd.read_csv(\"data.csv\")\ndf = df.dropna(subset=[\"instrument\",\"atomic_weight\"])\n\ng1 = df.loc[df.instrument==1, \"atomic_weight\"].astype(float)\ng2 = df.loc[df.instrument==2, \"atomic_weight\"].astype(float)\n\nn_left = len(g2)\nn_right = len(g1)\n\nmean1, mean2 = g1.mean(), g2.mean()\nvar1, var2 = g1.var(ddof=1), g2.var(ddof=1)\n\ndifference = mean2 - mean1\n\nse = np.sqrt(var1/n_right + var2/n_left)\n\ndf_welch = (var1/n_right + var2/n_left)**2 / (\n    (var1/n_right)**2/(n_right-1) + (var2/n_left)**2/(n_left-1)\n)\n\nt_stat = difference / se\np_value = 2*stats.t.sf(np.abs(t_stat), df_welch)\n\ntcrit = stats.t.ppf(0.975, df_welch)\nci_low = difference - tcrit*se\nci_high = difference + tcrit*se\n\n# pooled SD for hedges g\npooled_sd = np.sqrt(((n_right-1)*var1 + (n_left-1)*var2) / (n_right + n_left - 2))\nJ = 1 - 3/(4*(n_left+n_right)-9)\nhedges_g = J * (difference / pooled_sd)\n\nresult = {\n    \"method\": \"welch_independent\",\n    \"n_left\": int(n_left),\n    \"n_right\": int(n_right),\n    \"difference\": float(difference),\n    \"se\": float(se),\n    \"df\": float(df_welch),\n    \"p_value\": float(p_value),\n    \"ci_low\": float(ci_low),\n    \"ci_high\": f…",
                "code_truncated": true,
                "checks": [
                  {
                    "input": "original",
                    "passed": true
                  },
                  {
                    "input": "changed",
                    "passed": true
                  }
                ],
                "execution_warnings": false
              }
            },
            "skill": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.0219308,
              "seconds": 17.30649954200001,
              "schema_valid": true,
              "outcome": "Both original and changed-input executions passed.",
              "metrics": {
                "original_correct": 1.0,
                "perturbed_correct": 1.0,
                "executable": 1.0
              },
              "output": {
                "kind": "executed_analysis",
                "value": {
                  "method": "welch_independent",
                  "n_left": 24,
                  "n_right": 24,
                  "difference": -1.7412500028513023e-05,
                  "se": 4.360389250340865e-06,
                  "df": 43.25183428311259,
                  "p_value": 0.00024855676056512177,
                  "ci_low": -2.620458259250799e-05,
                  "ci_high": -8.620417464518057e-06,
                  "hedges_g": -1.133878869885915
                },
                "value_source": "Original-input execution stdout",
                "code_excerpt": "\nimport pandas as pd, numpy as np\nfrom scipy import stats\nimport json\n\ndf = pd.read_csv('data.csv')\ng1 = df[df.instrument==2]['atomic_weight'].values  # left = instrument 2\ng2 = df[df.instrument==1]['atomic_weight'].values  # right = instrument 1\n\nn_left = len(g1)\nn_right = len(g2)\nm1, m2 = g1.mean(), g2.mean()\nv1, v2 = g1.var(ddof=1), g2.var(ddof=1)\n\ndiff = m1 - m2\nse = np.sqrt(v1/n_left + v2/n_right)\ndf_welch = (v1/n_left + v2/n_right)**2 / ((v1/n_left)**2/(n_left-1) + (v2/n_right)**2/(n_right-1))\nt = diff/se\np = 2*stats.t.sf(np.abs(t), df_welch)\ntcrit = stats.t.ppf(0.975, df_welch)\nci_low = diff - tcrit*se\nci_high = diff + tcrit*se\n\nsp = np.sqrt(((n_left-1)*v1 + (n_right-1)*v2)/(n_left+n_right-2))\nd = diff/sp\nJ = 1 - 3/(4*(n_left+n_right)-9)\ng = d*J\n\nresult = {\n    \"method\": \"welch_independent\",\n    \"n_left\": n_left,\n    \"n_right\": n_right,\n    \"difference\": float(diff),\n    \"se\": float(se),\n    \"df\": float(df_welch),\n    \"p_value\": float(p),\n    \"ci_low\": float(ci_low),\n    \"ci_high\": float(ci_high),\n    \"hedges_g\": float(g)\n}\nprint(json.dumps(result))\n",
                "code_truncated": false,
                "checks": [
                  {
                    "input": "original",
                    "passed": true
                  },
                  {
                    "input": "changed",
                    "passed": true
                  }
                ],
                "execution_warnings": false
              }
            }
          },
          "evidence": {
            "run_id": "four-track-canary-v3",
            "task_id": "analyze-silver-instruments"
          }
        },
        {
          "id": "analyze-paired-missing",
          "label": "Synthetic data: incomplete pairs",
          "task_summary": "Compute after-minus-before changes using complete matched subjects only.",
          "source": {
            "dataset": "paired_synthetic.csv",
            "label": "Synthetic data: incomplete pairs",
            "cluster_id": "synthetic_paired",
            "url": null,
            "synthetic": true
          },
          "arms": {
            "baseline": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.007876,
              "seconds": 7.137547291999994,
              "schema_valid": true,
              "outcome": "Both original and changed-input executions passed.",
              "metrics": {
                "original_correct": 1.0,
                "perturbed_correct": 1.0,
                "executable": 1.0
              },
              "output": {
                "kind": "executed_analysis",
                "value": {
                  "method": "paired_complete_cases",
                  "n_pairs": 12,
                  "excluded_subjects": 2,
                  "mean_change": 1.508333333333334,
                  "sd_change": 1.1843358962522066,
                  "p_value": 0.001042637825800151,
                  "ci_low": 0.7558422052400875,
                  "ci_high": 2.2608244614265804,
                  "cohen_dz": 1.2735688735825765
                },
                "value_source": "Original-input execution stdout",
                "code_excerpt": "\nimport pandas as pd, numpy as np, json\nfrom scipy import stats\n\ndf = pd.read_csv('data.csv')\ntotal = len(df)\ncomplete = df.dropna(subset=['before','after'])\nexcluded = total - len(complete)\n\ndiff = complete['after'] - complete['before']\nn = len(diff)\nmean_change = diff.mean()\nsd_change = diff.std(ddof=1)\nse = sd_change/np.sqrt(n)\ntstat = mean_change/se\np_value = 2*stats.t.sf(abs(tstat), df=n-1)\ntcrit = stats.t.ppf(0.975, df=n-1)\nci_low = mean_change - tcrit*se\nci_high = mean_change + tcrit*se\ncohen_dz = mean_change/sd_change\n\nresult = {\n    \"method\": \"paired_complete_cases\",\n    \"n_pairs\": int(n),\n    \"excluded_subjects\": int(excluded),\n    \"mean_change\": float(mean_change),\n    \"sd_change\": float(sd_change),\n    \"p_value\": float(p_value),\n    \"ci_low\": float(ci_low),\n    \"ci_high\": float(ci_high),\n    \"cohen_dz\": float(cohen_dz)\n}\n\nprint(json.dumps(result))\n",
                "code_truncated": false,
                "checks": [
                  {
                    "input": "original",
                    "passed": true
                  },
                  {
                    "input": "changed",
                    "passed": true
                  }
                ],
                "execution_warnings": false
              }
            },
            "checklist": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.0065308,
              "seconds": 9.146532292000003,
              "schema_valid": true,
              "outcome": "Both original and changed-input executions passed.",
              "metrics": {
                "original_correct": 1.0,
                "perturbed_correct": 1.0,
                "executable": 1.0
              },
              "output": {
                "kind": "executed_analysis",
                "value": {
                  "method": "paired_complete_cases",
                  "n_pairs": 12,
                  "excluded_subjects": 2,
                  "mean_change": 1.508333333333334,
                  "sd_change": 1.1843358962522066,
                  "p_value": 0.001042637825800151,
                  "ci_low": 0.7558422052400875,
                  "ci_high": 2.2608244614265804,
                  "cohen_dz": 1.2735688735825765
                },
                "value_source": "Original-input execution stdout",
                "code_excerpt": "\nimport pandas as pd, numpy as np, json\nfrom scipy import stats\n\ndf = pd.read_csv(\"data.csv\")\ntotal_subjects = len(df)\ncomplete = df.dropna(subset=[\"before\",\"after\"])\nexcluded_subjects = total_subjects - len(complete)\n\ndiffs = complete[\"after\"] - complete[\"before\"]\nn = len(diffs)\nmean_change = diffs.mean()\nsd_change = diffs.std(ddof=1)\nse = sd_change / np.sqrt(n)\ntstat = mean_change / se\np_value = 2*stats.t.sf(abs(tstat), df=n-1)\ntcrit = stats.t.ppf(0.975, df=n-1)\nci_low = mean_change - tcrit*se\nci_high = mean_change + tcrit*se\ncohen_dz = mean_change / sd_change\n\nresult = {\n    \"method\": \"paired_complete_cases\",\n    \"n_pairs\": int(n),\n    \"excluded_subjects\": int(excluded_subjects),\n    \"mean_change\": float(mean_change),\n    \"sd_change\": float(sd_change),\n    \"p_value\": float(p_value),\n    \"ci_low\": float(ci_low),\n    \"ci_high\": float(ci_high),\n    \"cohen_dz\": float(cohen_dz)\n}\n\nprint(json.dumps(result))\n",
                "code_truncated": false,
                "checks": [
                  {
                    "input": "original",
                    "passed": true
                  },
                  {
                    "input": "changed",
                    "passed": true
                  }
                ],
                "execution_warnings": false
              }
            },
            "skill": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.0167928,
              "seconds": 16.307344541,
              "schema_valid": true,
              "outcome": "Both original and changed-input executions passed.",
              "metrics": {
                "original_correct": 1.0,
                "perturbed_correct": 1.0,
                "executable": 1.0
              },
              "output": {
                "kind": "executed_analysis",
                "value": {
                  "method": "paired_complete_cases",
                  "n_pairs": 12,
                  "excluded_subjects": 2,
                  "mean_change": 1.508333333333334,
                  "sd_change": 1.1843358962522066,
                  "p_value": 0.001042637825800151,
                  "ci_low": 0.7558422052400875,
                  "ci_high": 2.2608244614265804,
                  "cohen_dz": 1.2735688735825765
                },
                "value_source": "Original-input execution stdout",
                "code_excerpt": "\nimport pandas as pd, numpy as np\nfrom scipy import stats\nimport json\n\ndf = pd.read_csv('data.csv')\ntotal = len(df)\ncomplete = df.dropna(subset=['before','after'])\nexcluded = total - len(complete)\ndiff = complete['after'] - complete['before']\nn = len(diff)\nmean_change = diff.mean()\nsd_change = diff.std(ddof=1)\nse = sd_change/np.sqrt(n)\nt_stat = mean_change/se\np_value = 2*stats.t.sf(abs(t_stat), df=n-1)\nci = stats.t.interval(0.95, n-1, loc=mean_change, scale=se)\ncohen_dz = mean_change/sd_change\n\nresult = {\n    \"method\": \"paired_complete_cases\",\n    \"n_pairs\": int(n),\n    \"excluded_subjects\": int(excluded),\n    \"mean_change\": float(mean_change),\n    \"sd_change\": float(sd_change),\n    \"p_value\": float(p_value),\n    \"ci_low\": float(ci[0]),\n    \"ci_high\": float(ci[1]),\n    \"cohen_dz\": float(cohen_dz)\n}\nprint(json.dumps(result))\n",
                "code_truncated": false,
                "checks": [
                  {
                    "input": "original",
                    "passed": true
                  },
                  {
                    "input": "changed",
                    "passed": true
                  }
                ],
                "execution_warnings": false
              }
            }
          },
          "evidence": {
            "run_id": "new-sources-and-analysis-v3",
            "task_id": "analyze-paired-missing"
          }
        },
        {
          "id": "analyze-penguin-adjusted",
          "label": "Penguins: species-adjusted regression",
          "task_summary": "Regress body mass on flipper length and species, retaining rows with missing non-model variables.",
          "source": {
            "dataset": "penguins.csv",
            "label": "Penguins: species-adjusted regression",
            "cluster_id": "penguins",
            "url": "https://raw.githubusercontent.com/allisonhorst/palmerpenguins/main/inst/extdata/penguins.csv",
            "synthetic": false
          },
          "arms": {
            "baseline": {
              "score": 0.0,
              "status": "failed",
              "cost_usd": 0.13527999999999998,
              "seconds": 71.130589625,
              "schema_valid": null,
              "outcome": "A tool call exhausted its output allowance and omitted the code argument; no numerical answer was graded.",
              "metrics": {},
              "output": null
            },
            "checklist": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.0287128,
              "seconds": 15.208112999999997,
              "schema_valid": true,
              "outcome": "Both original and changed-input executions passed.",
              "metrics": {
                "original_correct": 1.0,
                "perturbed_correct": 1.0,
                "executable": 1.0
              },
              "output": {
                "kind": "executed_analysis",
                "value": {
                  "method": "ols_species_adjusted",
                  "n": 342,
                  "intercept": -4031.476890693633,
                  "flipper_slope": 40.705400777284765,
                  "chinstrap_offset": -206.51012033971995,
                  "gentoo_offset": 266.80960317921796,
                  "slope_ci_low": 34.664682514400155,
                  "slope_ci_high": 46.746119040169376,
                  "residual_sd": 375.53507478218387
                },
                "value_source": "Original-input execution stdout",
                "code_excerpt": "\nimport pandas as pd\nimport numpy as np\nfrom scipy import stats\nimport json\n\ndf = pd.read_csv('data.csv', na_values=['NA'])\n\nneeded = ['body_mass_g','flipper_length_mm','species']\nd = df.dropna(subset=needed).copy()\n\ny = d['body_mass_g'].astype(float).values\nflipper = d['flipper_length_mm'].astype(float).values\nspecies = d['species'].astype(str)\n\nn = len(d)\nchin = (species == 'Chinstrap').astype(float).values\ngent = (species == 'Gentoo').astype(float).values\n\nX = np.column_stack([np.ones(n), flipper, chin, gent])\n\nbeta, residuals, rank, sv = np.linalg.lstsq(X, y, rcond=None)\n\nfitted = X @ beta\nresid = y - fitted\np = X.shape[1]\ndof = n - p\nsigma2 = np.sum(resid**2) / dof\nresidual_sd = np.sqrt(sigma2)\n\nXtX_inv = np.linalg.inv(X.T @ X)\ncov_beta = sigma2 * XtX_inv\nse = np.sqrt(np.diag(cov_beta))\n\ntcrit = stats.t.ppf(0.975, dof)\n\nintercept = beta[0]\nflipper_slope = beta[1]\nchinstrap_offset = beta[2]\ngentoo_offset = beta[3]\n\nslope_se = se[1]\nslope_ci_low = flipper_slope - tcrit*slope_se\nslope_ci_high = flipper_slope + tcrit*slope_se\n\nresult = {\n    \"method\": \"ols_species_adjusted\",\n    \"n\": int(n),\n    \"intercept\": float(intercept),\n    \"flipper_slope\": float(flipper_slope),\n    \"chinstr…",
                "code_truncated": true,
                "checks": [
                  {
                    "input": "original",
                    "passed": true
                  },
                  {
                    "input": "changed",
                    "passed": true
                  }
                ],
                "execution_warnings": true
              }
            },
            "skill": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.0622148,
              "seconds": 18.594054125,
              "schema_valid": true,
              "outcome": "Both original and changed-input executions passed.",
              "metrics": {
                "original_correct": 1.0,
                "perturbed_correct": 1.0,
                "executable": 1.0
              },
              "output": {
                "kind": "executed_analysis",
                "value": {
                  "method": "ols_species_adjusted",
                  "n": 342,
                  "intercept": -4031.476890693633,
                  "flipper_slope": 40.705400777284765,
                  "chinstrap_offset": -206.51012033971995,
                  "gentoo_offset": 266.80960317921796,
                  "slope_ci_low": 34.664682514400155,
                  "slope_ci_high": 46.746119040169376,
                  "residual_sd": 375.5350747821838
                },
                "value_source": "Original-input execution stdout",
                "code_excerpt": "\nimport pandas as pd, numpy as np, json\nfrom scipy import stats\n\ndf = pd.read_csv(\"data.csv\")\ndf2 = df.dropna(subset=[\"body_mass_g\",\"flipper_length_mm\",\"species\"])\ndf2 = df2[df2['species'].notna()]\n\ny = df2['body_mass_g'].astype(float).values\nsp = df2['species']\nX = pd.DataFrame({\n    'const':1.0,\n    'flipper': df2['flipper_length_mm'].astype(float).values,\n    'chinstrap': (sp=='Chinstrap').astype(float).values,\n    'gentoo': (sp=='Gentoo').astype(float).values,\n})\nXm = X.values\nn,k = Xm.shape\nbeta, _, _, _ = np.linalg.lstsq(Xm, y, rcond=None)\nresid = y - Xm@beta\ndof = n-k\nsigma2 = (resid@resid)/dof\nXtX_inv = np.linalg.inv(Xm.T@Xm)\ncov = sigma2*XtX_inv\nse = np.sqrt(np.diag(cov))\n\ntcrit = stats.t.ppf(0.975, dof)\n\nslope = beta[1]\nslope_se = se[1]\nci_low = slope - tcrit*slope_se\nci_high = slope + tcrit*slope_se\n\nresidual_sd = np.sqrt(sigma2)\n\nresult = {\n    \"method\": \"ols_species_adjusted\",\n    \"n\": int(n),\n    \"intercept\": float(beta[0]),\n    \"flipper_slope\": float(slope),\n    \"chinstrap_offset\": float(beta[2]),\n    \"gentoo_offset\": float(beta[3]),\n    \"slope_ci_low\": float(ci_low),\n    \"slope_ci_high\": float(ci_high),\n    \"residual_sd\": float(residual_sd),\n}\nprint(json.dumps(resul…",
                "code_truncated": true,
                "checks": [
                  {
                    "input": "original",
                    "passed": true
                  },
                  {
                    "input": "changed",
                    "passed": true
                  }
                ],
                "execution_warnings": true
              }
            }
          },
          "evidence": {
            "run_id": "new-sources-and-analysis-v3",
            "task_id": "analyze-penguin-adjusted"
          }
        },
        {
          "id": "analyze-norris-holdout",
          "label": "Norris: held-out prediction",
          "task_summary": "Fit only the training rows, then calculate prediction error on held-out rows.",
          "source": {
            "dataset": "Norris.dat",
            "label": "Norris: held-out prediction",
            "cluster_id": "nist_norris",
            "url": "https://www.itl.nist.gov/div898/strd/lls/data/LINKS/DATA/Norris.dat",
            "synthetic": false
          },
          "arms": {
            "baseline": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.01969,
              "seconds": 15.062335791999999,
              "schema_valid": true,
              "outcome": "Both original and changed-input executions passed.",
              "metrics": {
                "original_correct": 1.0,
                "perturbed_correct": 1.0,
                "executable": 1.0
              },
              "output": {
                "kind": "executed_analysis",
                "value": {
                  "method": "train_only_ols_holdout",
                  "n_train": 24,
                  "n_test": 12,
                  "intercept": -0.25917168496622917,
                  "slope": 1.0029666153104575,
                  "test_rmse": 1.4837478210602169,
                  "test_mae": 1.1711495352409773
                },
                "value_source": "Original-input execution stdout",
                "code_excerpt": "\nimport pandas as pd, numpy as np, json\n\ndf = pd.read_csv(\"data.csv\")\ntrain = df[df.split==\"train\"]\ntest = df[df.split==\"test\"]\n\nx = train.x.values\ny = train.y.values\nn = len(x)\nslope, intercept = np.polyfit(x, y, 1)\n\nxt = test.x.values\nyt = test.y.values\npred = intercept + slope*xt\nresid = yt - pred\nrmse = np.sqrt(np.mean(resid**2))\nmae = np.mean(np.abs(resid))\n\nresult = {\n    \"method\": \"train_only_ols_holdout\",\n    \"n_train\": int(n),\n    \"n_test\": int(len(xt)),\n    \"intercept\": float(intercept),\n    \"slope\": float(slope),\n    \"test_rmse\": float(rmse),\n    \"test_mae\": float(mae)\n}\nprint(json.dumps(result))\n",
                "code_truncated": false,
                "checks": [
                  {
                    "input": "original",
                    "passed": true
                  },
                  {
                    "input": "changed",
                    "passed": true
                  }
                ],
                "execution_warnings": false
              }
            },
            "checklist": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.0060708,
              "seconds": 9.106120000000004,
              "schema_valid": true,
              "outcome": "Both original and changed-input executions passed.",
              "metrics": {
                "original_correct": 1.0,
                "perturbed_correct": 1.0,
                "executable": 1.0
              },
              "output": {
                "kind": "executed_analysis",
                "value": {
                  "method": "train_only_ols_holdout",
                  "n_train": 24,
                  "n_test": 12,
                  "intercept": -0.25917168496622917,
                  "slope": 1.0029666153104575,
                  "test_rmse": 1.4837478210602169,
                  "test_mae": 1.1711495352409773
                },
                "value_source": "Original-input execution stdout",
                "code_excerpt": "\nimport pandas as pd, numpy as np, json\n\ndf = pd.read_csv(\"data.csv\")\ntrain = df[df.split==\"train\"].dropna(subset=[\"x\",\"y\"])\ntest = df[df.split==\"test\"].dropna(subset=[\"x\",\"y\"])\n\nx_tr = train[\"x\"].values\ny_tr = train[\"y\"].values\nx_te = test[\"x\"].values\ny_te = test[\"y\"].values\n\nn_train = len(x_tr)\nn_test = len(x_te)\n\nslope, intercept = np.polyfit(x_tr, y_tr, 1)\n\npred = intercept + slope * x_te\nresid = y_te - pred\ntest_rmse = float(np.sqrt(np.mean(resid**2)))\ntest_mae = float(np.mean(np.abs(resid)))\n\nresult = {\n    \"method\": \"train_only_ols_holdout\",\n    \"n_train\": int(n_train),\n    \"n_test\": int(n_test),\n    \"intercept\": float(intercept),\n    \"slope\": float(slope),\n    \"test_rmse\": test_rmse,\n    \"test_mae\": test_mae\n}\n\nprint(json.dumps(result))\n",
                "code_truncated": false,
                "checks": [
                  {
                    "input": "original",
                    "passed": true
                  },
                  {
                    "input": "changed",
                    "passed": true
                  }
                ],
                "execution_warnings": false
              }
            },
            "skill": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.014778799999999998,
              "seconds": 15.470077417000027,
              "schema_valid": true,
              "outcome": "Both original and changed-input executions passed.",
              "metrics": {
                "original_correct": 1.0,
                "perturbed_correct": 1.0,
                "executable": 1.0
              },
              "output": {
                "kind": "executed_analysis",
                "value": {
                  "method": "train_only_ols_holdout",
                  "n_train": 24,
                  "n_test": 12,
                  "intercept": -0.25917168496622917,
                  "slope": 1.0029666153104575,
                  "test_rmse": 1.4837478210602169,
                  "test_mae": 1.1711495352409773
                },
                "value_source": "Original-input execution stdout",
                "code_excerpt": "\nimport pandas as pd, numpy as np, json\n\ndf = pd.read_csv('data.csv')\ntrain = df[df.split=='train']\ntest = df[df.split=='test']\n\nx = train.x.values\ny = train.y.values\nslope, intercept = np.polyfit(x, y, 1)\n\npred = intercept + slope*test.x.values\nresid = test.y.values - pred\nrmse = np.sqrt(np.mean(resid**2))\nmae = np.mean(np.abs(resid))\n\nresult = {\n    \"method\": \"train_only_ols_holdout\",\n    \"n_train\": int(len(train)),\n    \"n_test\": int(len(test)),\n    \"intercept\": float(intercept),\n    \"slope\": float(slope),\n    \"test_rmse\": float(rmse),\n    \"test_mae\": float(mae)\n}\nprint(json.dumps(result))\n",
                "code_truncated": false,
                "checks": [
                  {
                    "input": "original",
                    "passed": true
                  },
                  {
                    "input": "changed",
                    "passed": true
                  }
                ],
                "execution_warnings": false
              }
            }
          },
          "evidence": {
            "run_id": "new-sources-and-analysis-v3",
            "task_id": "analyze-norris-holdout"
          }
        },
        {
          "id": "analyze-penguin-multiplicity",
          "label": "Penguins: four simultaneous tests",
          "task_summary": "Compare 2009 with 2007 for four measurements and apply Benjamini-Hochberg adjustment.",
          "source": {
            "dataset": "penguins.csv",
            "label": "Penguins: four simultaneous tests",
            "cluster_id": "penguins",
            "url": "https://raw.githubusercontent.com/allisonhorst/palmerpenguins/main/inst/extdata/penguins.csv",
            "synthetic": false
          },
          "arms": {
            "baseline": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.010026,
              "seconds": 8.92177700000002,
              "schema_valid": true,
              "outcome": "Both original and changed-input executions passed.",
              "metrics": {
                "original_correct": 1.0,
                "perturbed_correct": 1.0,
                "executable": 1.0
              },
              "output": {
                "kind": "executed_analysis",
                "value": {
                  "method": "welch_bh_four_tests",
                  "bill_length_mm": {
                    "difference": -0.5150349650349781,
                    "p_value": 0.3731192317432984,
                    "q_value": 0.4657145087383828,
                    "reject_fdr_05": false
                  },
                  "bill_depth_mm": {
                    "difference": -0.7772727272727238,
                    "p_value": 0.004087705017456997,
                    "q_value": 0.01635082006982799,
                    "reject_fdr_05": true
                  },
                  "flipper_length_mm": {
                    "difference": 4.11888111888112,
                    "p_value": 0.018363579336678298,
                    "q_value": 0.036727158673356596,
                    "reject_fdr_05": true
                  },
                  "body_mass_g": {
                    "difference": -55.157342657342724,
                    "p_value": 0.4657145087383828,
                    "q_value": 0.4657145087383828,
                    "reject_fdr_05": false
                  }
                },
                "value_source": "Original-input execution stdout",
                "code_excerpt": "\nimport pandas as pd, numpy as np\nfrom scipy import stats\nimport json\n\ndf = pd.read_csv('data.csv')\nvars_ = ['bill_length_mm','bill_depth_mm','flipper_length_mm','body_mass_g']\n\nresults = {}\npvals = []\ndiffs = []\nfor v in vars_:\n    g2009 = df.loc[df['year']==2009, v].dropna()\n    g2007 = df.loc[df['year']==2007, v].dropna()\n    t, p = stats.ttest_ind(g2009, g2007, equal_var=False)\n    diff = g2009.mean() - g2007.mean()\n    diffs.append(diff)\n    pvals.append(p)\n\npvals = np.array(pvals)\nn = len(pvals)\norder = np.argsort(pvals)\nranked = pvals[order]\nq = ranked * n / (np.arange(1, n+1))\n# enforce monotonicity from the end\nq_mono = np.minimum.accumulate(q[::-1])[::-1]\nq_final = np.empty(n)\nq_final[order] = np.minimum(q_mono, 1.0)\n\nout = {\"method\": \"welch_bh_four_tests\"}\nfor i, v in enumerate(vars_):\n    out[v] = {\n        \"difference\": float(diffs[i]),\n        \"p_value\": float(pvals[i]),\n        \"q_value\": float(q_final[i]),\n        \"reject_fdr_05\": bool(q_final[i] <= 0.05)\n    }\n\nprint(json.dumps(out))\n",
                "code_truncated": false,
                "checks": [
                  {
                    "input": "original",
                    "passed": true
                  },
                  {
                    "input": "changed",
                    "passed": true
                  }
                ],
                "execution_warnings": false
              }
            },
            "checklist": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.045180399999999996,
              "seconds": 27.555874208000034,
              "schema_valid": true,
              "outcome": "Both original and changed-input executions passed.",
              "metrics": {
                "original_correct": 1.0,
                "perturbed_correct": 1.0,
                "executable": 1.0
              },
              "output": {
                "kind": "executed_analysis",
                "value": {
                  "method": "welch_bh_four_tests",
                  "bill_length_mm": {
                    "difference": -0.5150349650349781,
                    "p_value": 0.3731192317432984,
                    "q_value": 0.4657145087383828,
                    "reject_fdr_05": false
                  },
                  "bill_depth_mm": {
                    "difference": -0.7772727272727238,
                    "p_value": 0.004087705017456997,
                    "q_value": 0.01635082006982799,
                    "reject_fdr_05": true
                  },
                  "flipper_length_mm": {
                    "difference": 4.11888111888112,
                    "p_value": 0.018363579336678298,
                    "q_value": 0.036727158673356596,
                    "reject_fdr_05": true
                  },
                  "body_mass_g": {
                    "difference": -55.157342657342724,
                    "p_value": 0.4657145087383828,
                    "q_value": 0.4657145087383828,
                    "reject_fdr_05": false
                  }
                },
                "value_source": "Original-input execution stdout",
                "code_excerpt": "\nimport pandas as pd, numpy as np\nfrom scipy import stats\nimport json\n\ndf = pd.read_csv(\"data.csv\")\n\nvars_ = [\"bill_length_mm\",\"bill_depth_mm\",\"flipper_length_mm\",\"body_mass_g\"]\npvals = []\ndiffs = {}\nfor v in vars_:\n    sub = df[[\"year\", v]].dropna()\n    g2009 = sub[sub.year==2009][v].values\n    g2007 = sub[sub.year==2007][v].values\n    diff = float(np.mean(g2009) - np.mean(g2007))\n    t, p = stats.ttest_ind(g2009, g2007, equal_var=False)\n    diffs[v] = diff\n    pvals.append(float(p))\n\nm = len(pvals)\norder = np.argsort(pvals)\nranked = np.array(pvals)[order]\nq = np.empty(m)\nprev = 1.0\nfor i in range(m-1, -1, -1):\n    rank = i+1\n    val = ranked[i]*m/rank\n    prev = min(prev, val)\n    q[i] = prev\nq_final = np.empty(m)\nq_final[order] = q\n\nout = {\"method\": \"welch_bh_four_tests\"}\nfor idx, v in enumerate(vars_):\n    out[v] = {\n        \"difference\": diffs[v],\n        \"p_value\": pvals[idx],\n        \"q_value\": float(q_final[idx]),\n        \"reject_fdr_05\": bool(q_final[idx] <= 0.05)\n    }\n\nprint(json.dumps(out))\n",
                "code_truncated": false,
                "checks": [
                  {
                    "input": "original",
                    "passed": true
                  },
                  {
                    "input": "changed",
                    "passed": true
                  }
                ],
                "execution_warnings": false
              }
            },
            "skill": {
              "score": 1.0,
              "status": "completed",
              "cost_usd": 0.021802799999999997,
              "seconds": 14.138100000000009,
              "schema_valid": true,
              "outcome": "Both original and changed-input executions passed.",
              "metrics": {
                "original_correct": 1.0,
                "perturbed_correct": 1.0,
                "executable": 1.0
              },
              "output": {
                "kind": "executed_analysis",
                "value": {
                  "method": "welch_bh_four_tests",
                  "bill_length_mm": {
                    "difference": -0.5150349650349781,
                    "p_value": 0.3731192317432984,
                    "q_value": 0.4657145087383828,
                    "reject_fdr_05": false
                  },
                  "bill_depth_mm": {
                    "difference": -0.7772727272727238,
                    "p_value": 0.004087705017456997,
                    "q_value": 0.01635082006982799,
                    "reject_fdr_05": true
                  },
                  "flipper_length_mm": {
                    "difference": 4.11888111888112,
                    "p_value": 0.018363579336678298,
                    "q_value": 0.036727158673356596,
                    "reject_fdr_05": true
                  },
                  "body_mass_g": {
                    "difference": -55.157342657342724,
                    "p_value": 0.4657145087383828,
                    "q_value": 0.4657145087383828,
                    "reject_fdr_05": false
                  }
                },
                "value_source": "Original-input execution stdout",
                "code_excerpt": "\nimport pandas as pd, numpy as np\nfrom scipy import stats\nimport json\n\ndf = pd.read_csv('data.csv')\nvars_ = ['bill_length_mm','bill_depth_mm','flipper_length_mm','body_mass_g']\npvals = []\ndiffs = []\nfor v in vars_:\n    g09 = df[df.year==2009][v].dropna()\n    g07 = df[df.year==2007][v].dropna()\n    t,p = stats.ttest_ind(g09, g07, equal_var=False)\n    diff = g09.mean()-g07.mean()\n    diffs.append(diff)\n    pvals.append(p)\n\npvals = np.array(pvals)\nn = len(pvals)\norder = np.argsort(pvals)\nranked = pvals[order]\nq = ranked * n / (np.arange(n)+1)\nq = np.minimum.accumulate(q[::-1])[::-1]\nqvals = np.empty(n)\nqvals[order] = np.minimum(q,1)\n\nout = {\"method\":\"welch_bh_four_tests\"}\nfor i,v in enumerate(vars_):\n    out[v] = {\n        \"difference\": float(diffs[i]),\n        \"p_value\": float(pvals[i]),\n        \"q_value\": float(qvals[i]),\n        \"reject_fdr_05\": bool(qvals[i]<=0.05)\n    }\n\nprint(json.dumps(out))\n",
                "code_truncated": false,
                "checks": [
                  {
                    "input": "original",
                    "passed": true
                  },
                  {
                    "input": "changed",
                    "passed": true
                  }
                ],
                "execution_warnings": false
              }
            }
          },
          "evidence": {
            "run_id": "new-sources-and-analysis-v3",
            "task_id": "analyze-penguin-multiplicity"
          }
        }
      ]
    }
  ]
}