{
  "assetVersion": "dev",
  "costComparisons": [
    {
      "interfaces": [
        {
          "cost": 265.2709,
          "costBasis": "billed",
          "costPerSuccess": 0.211371,
          "criteria": [
            {
              "description": "Diagnostic. The script used the intended estimator, target, and inference rule.",
              "key": "specification_correctness",
              "label": "Specification correctness",
              "passed": 1488,
              "pct": 98.4,
              "total": 1512
            },
            {
              "description": "Diagnostic. The submitted script completed far enough for scoring.",
              "key": "executability",
              "label": "Executability",
              "passed": 1366,
              "pct": 90.3,
              "total": 1512
            },
            {
              "description": "The required result file was valid and matched the reference output within tolerance.",
              "key": "output_correctness",
              "label": "Output correctness",
              "passed": 1247,
              "pct": 82.5,
              "total": 1512
            },
            {
              "description": "The run avoided input-hash, hidden-output, and shortcut problems.",
              "key": "submission_validity",
              "label": "Submission validity",
              "passed": 1509,
              "pct": 99.8,
              "total": 1512
            },
            {
              "description": "The reported values matched the reference output within the declared tolerances and the integrity checks passed.",
              "key": "task_success",
              "label": "Task success",
              "passed": 1256,
              "pct": 83.1,
              "total": 1512
            }
          ],
          "detail": "one script, no execution",
          "id": "no_feedback",
          "label": "Chatbot",
          "meanCost": 0.175909,
          "n": 1512,
          "nPriced": 1508,
          "records": 1512,
          "success": 83.1,
          "taskSuccessPassed": 1256
        },
        {
          "cost": 337.4788,
          "costBasis": "billed",
          "costPerSuccess": 0.239008,
          "criteria": [
            {
              "description": "Diagnostic. The script used the intended estimator, target, and inference rule.",
              "key": "specification_correctness",
              "label": "Specification correctness",
              "passed": 1491,
              "pct": 98.6,
              "total": 1512
            },
            {
              "description": "Diagnostic. The submitted script completed far enough for scoring.",
              "key": "executability",
              "label": "Executability",
              "passed": 1489,
              "pct": 98.5,
              "total": 1512
            },
            {
              "description": "The required result file was valid and matched the reference output within tolerance.",
              "key": "output_correctness",
              "label": "Output correctness",
              "passed": 1407,
              "pct": 93.1,
              "total": 1512
            },
            {
              "description": "The run avoided input-hash, hidden-output, and shortcut problems.",
              "key": "submission_validity",
              "label": "Submission validity",
              "passed": 1508,
              "pct": 99.7,
              "total": 1512
            },
            {
              "description": "The reported values matched the reference output within the declared tolerances and the integrity checks passed.",
              "key": "task_success",
              "label": "Task success",
              "passed": 1412,
              "pct": 93.4,
              "total": 1512
            }
          ],
          "detail": "revises after execution feedback",
          "id": "self_repair",
          "label": "Self-repair",
          "meanCost": 0.223792,
          "n": 1512,
          "nPriced": 1508,
          "records": 1512,
          "success": 93.4,
          "taskSuccessPassed": 1412
        },
        {
          "cost": 395.0629,
          "costBasis": "billed",
          "costPerSuccess": 0.272457,
          "criteria": [
            {
              "description": "Diagnostic. The script used the intended estimator, target, and inference rule.",
              "key": "specification_correctness",
              "label": "Specification correctness",
              "passed": 1505,
              "pct": 99.5,
              "total": 1512
            },
            {
              "description": "Diagnostic. The submitted script completed far enough for scoring.",
              "key": "executability",
              "label": "Executability",
              "passed": 1511,
              "pct": 99.9,
              "total": 1512
            },
            {
              "description": "The required result file was valid and matched the reference output within tolerance.",
              "key": "output_correctness",
              "label": "Output correctness",
              "passed": 1450,
              "pct": 95.9,
              "total": 1512
            },
            {
              "description": "The run avoided input-hash, hidden-output, and shortcut problems.",
              "key": "submission_validity",
              "label": "Submission validity",
              "passed": 1512,
              "pct": 100.0,
              "total": 1512
            },
            {
              "description": "The reported values matched the reference output within the declared tolerances and the integrity checks passed.",
              "key": "task_success",
              "label": "Task success",
              "passed": 1450,
              "pct": 95.9,
              "total": 1512
            }
          ],
          "detail": "capped shell-execution loop",
          "id": "tool_loop",
          "label": "Constrained agent",
          "meanCost": 0.261458,
          "n": 1512,
          "nPriced": 1511,
          "records": 1512,
          "success": 95.9,
          "taskSuccessPassed": 1450
        }
      ],
      "label": "Claude Sonnet 4.6"
    },
    {
      "interfaces": [
        {
          "cost": 46.9245,
          "costBasis": "imputed",
          "criteria": [
            {
              "description": "Diagnostic. The script used the intended estimator, target, and inference rule.",
              "key": "specification_correctness",
              "label": "Specification correctness",
              "passed": 1475,
              "pct": 97.6,
              "total": 1512
            },
            {
              "description": "Diagnostic. The submitted script completed far enough for scoring.",
              "key": "executability",
              "label": "Executability",
              "passed": 1291,
              "pct": 85.4,
              "total": 1512
            },
            {
              "description": "The required result file was valid and matched the reference output within tolerance.",
              "key": "output_correctness",
              "label": "Output correctness",
              "passed": 1182,
              "pct": 78.2,
              "total": 1512
            },
            {
              "description": "The run avoided input-hash, hidden-output, and shortcut problems.",
              "key": "submission_validity",
              "label": "Submission validity",
              "passed": 1512,
              "pct": 100.0,
              "total": 1512
            },
            {
              "description": "The reported values matched the reference output within the declared tolerances and the integrity checks passed.",
              "key": "task_success",
              "label": "Task success",
              "passed": 1237,
              "pct": 81.8,
              "total": 1512
            }
          ],
          "detail": "one script, no execution",
          "id": "no_feedback",
          "label": "Chatbot",
          "meanCost": 0.031035,
          "n": 1512,
          "nPriced": 1512,
          "records": 1512,
          "success": 81.8,
          "taskSuccessPassed": 1237
        },
        {
          "cost": 80.8785,
          "costBasis": "imputed",
          "criteria": [
            {
              "description": "Diagnostic. The script used the intended estimator, target, and inference rule.",
              "key": "specification_correctness",
              "label": "Specification correctness",
              "passed": 1475,
              "pct": 97.6,
              "total": 1512
            },
            {
              "description": "Diagnostic. The submitted script completed far enough for scoring.",
              "key": "executability",
              "label": "Executability",
              "passed": 1401,
              "pct": 92.7,
              "total": 1512
            },
            {
              "description": "The required result file was valid and matched the reference output within tolerance.",
              "key": "output_correctness",
              "label": "Output correctness",
              "passed": 1343,
              "pct": 88.8,
              "total": 1512
            },
            {
              "description": "The run avoided input-hash, hidden-output, and shortcut problems.",
              "key": "submission_validity",
              "label": "Submission validity",
              "passed": 1512,
              "pct": 100.0,
              "total": 1512
            },
            {
              "description": "The reported values matched the reference output within the declared tolerances and the integrity checks passed.",
              "key": "task_success",
              "label": "Task success",
              "passed": 1374,
              "pct": 90.9,
              "total": 1512
            }
          ],
          "detail": "revises after execution feedback",
          "id": "self_repair",
          "label": "Self-repair",
          "meanCost": 0.053491,
          "n": 1512,
          "nPriced": 1512,
          "records": 1512,
          "success": 90.9,
          "taskSuccessPassed": 1374
        },
        {
          "cost": 144.9365,
          "costBasis": "imputed",
          "criteria": [
            {
              "description": "Diagnostic. The script used the intended estimator, target, and inference rule.",
              "key": "specification_correctness",
              "label": "Specification correctness",
              "passed": 1475,
              "pct": 97.6,
              "total": 1512
            },
            {
              "description": "Diagnostic. The submitted script completed far enough for scoring.",
              "key": "executability",
              "label": "Executability",
              "passed": 1512,
              "pct": 100.0,
              "total": 1512
            },
            {
              "description": "The required result file was valid and matched the reference output within tolerance.",
              "key": "output_correctness",
              "label": "Output correctness",
              "passed": 1471,
              "pct": 97.3,
              "total": 1512
            },
            {
              "description": "The run avoided input-hash, hidden-output, and shortcut problems.",
              "key": "submission_validity",
              "label": "Submission validity",
              "passed": 1512,
              "pct": 100.0,
              "total": 1512
            },
            {
              "description": "The reported values matched the reference output within the declared tolerances and the integrity checks passed.",
              "key": "task_success",
              "label": "Task success",
              "passed": 1471,
              "pct": 97.3,
              "total": 1512
            }
          ],
          "detail": "capped shell-execution loop",
          "id": "tool_loop",
          "label": "Constrained agent",
          "meanCost": 0.095857,
          "n": 1512,
          "nPriced": 1512,
          "records": 1512,
          "success": 97.3,
          "taskSuccessPassed": 1471
        }
      ],
      "label": "GPT-5.4"
    }
  ],
  "execPlatform": [
    {
      "execMode": "tool_loop",
      "id": "tool_loop__stata",
      "label": "Constrained agent \u00d7 Stata",
      "meanCost": 0,
      "n": 1008,
      "platform": "stata",
      "records": 1008,
      "success": 96.7
    },
    {
      "execMode": "tool_loop",
      "id": "tool_loop__python",
      "label": "Constrained agent \u00d7 Python",
      "meanCost": 0,
      "n": 1008,
      "platform": "python",
      "records": 1008,
      "success": 96.3
    },
    {
      "execMode": "self_repair",
      "id": "self_repair__python",
      "label": "Self-repair \u00d7 Python",
      "meanCost": 0,
      "n": 1008,
      "platform": "python",
      "records": 1008,
      "success": 95.7
    },
    {
      "execMode": "self_repair",
      "id": "self_repair__r",
      "label": "Self-repair \u00d7 R",
      "meanCost": 0,
      "n": 1008,
      "platform": "r",
      "records": 1008,
      "success": 96.1
    },
    {
      "execMode": "tool_loop",
      "id": "tool_loop__r",
      "label": "Constrained agent \u00d7 R",
      "meanCost": 0,
      "n": 1008,
      "platform": "r",
      "records": 1008,
      "success": 96.7
    },
    {
      "execMode": "no_feedback",
      "id": "no_feedback__python",
      "label": "Chatbot \u00d7 Python",
      "meanCost": 0,
      "n": 1008,
      "platform": "python",
      "records": 1008,
      "success": 93.8
    },
    {
      "execMode": "self_repair",
      "id": "self_repair__stata",
      "label": "Self-repair \u00d7 Stata",
      "meanCost": 0,
      "n": 1008,
      "platform": "stata",
      "records": 1008,
      "success": 84.5
    },
    {
      "execMode": "no_feedback",
      "id": "no_feedback__r",
      "label": "Chatbot \u00d7 R",
      "meanCost": 0,
      "n": 1008,
      "platform": "r",
      "records": 1008,
      "success": 91.0
    },
    {
      "execMode": "no_feedback",
      "id": "no_feedback__stata",
      "label": "Chatbot \u00d7 Stata",
      "meanCost": 0,
      "n": 1008,
      "platform": "stata",
      "records": 1008,
      "success": 62.6
    }
  ],
  "interfaces": [
    {
      "criteria": [
        {
          "key": "specification_correctness",
          "label": "Specification correctness",
          "passed": 2963,
          "total": 3024
        },
        {
          "key": "executability",
          "label": "Executability",
          "passed": 2657,
          "total": 3024
        },
        {
          "key": "output_correctness",
          "label": "Output correctness",
          "passed": 2429,
          "total": 3024
        },
        {
          "key": "submission_validity",
          "label": "Submission validity",
          "passed": 3021,
          "total": 3024
        }
      ],
      "id": "no_feedback",
      "label": "Chatbot",
      "meanCost": 0,
      "n": 3024,
      "records": 3024,
      "success": 82.4
    },
    {
      "criteria": [
        {
          "key": "specification_correctness",
          "label": "Specification correctness",
          "passed": 2966,
          "total": 3024
        },
        {
          "key": "executability",
          "label": "Executability",
          "passed": 2890,
          "total": 3024
        },
        {
          "key": "output_correctness",
          "label": "Output correctness",
          "passed": 2750,
          "total": 3024
        },
        {
          "key": "submission_validity",
          "label": "Submission validity",
          "passed": 3020,
          "total": 3024
        }
      ],
      "id": "self_repair",
      "label": "Self-repair",
      "meanCost": 0,
      "n": 3024,
      "records": 3024,
      "success": 92.1
    },
    {
      "criteria": [
        {
          "key": "specification_correctness",
          "label": "Specification correctness",
          "passed": 2980,
          "total": 3024
        },
        {
          "key": "executability",
          "label": "Executability",
          "passed": 3023,
          "total": 3024
        },
        {
          "key": "output_correctness",
          "label": "Output correctness",
          "passed": 2921,
          "total": 3024
        },
        {
          "key": "submission_validity",
          "label": "Submission validity",
          "passed": 3024,
          "total": 3024
        }
      ],
      "id": "tool_loop",
      "label": "Constrained agent",
      "meanCost": 0,
      "n": 3024,
      "records": 3024,
      "success": 96.6
    }
  ],
  "models": [
    {
      "id": "sonnet",
      "label": "Claude Sonnet 4.6",
      "meanCost": 0.220414,
      "n": 4536,
      "records": 4536,
      "success": 90.8
    },
    {
      "id": "codex",
      "label": "GPT-5.4",
      "meanCost": 0.060128,
      "n": 4536,
      "records": 4536,
      "success": 90.0
    }
  ],
  "narrative": {
    "headline": {
      "model": "Claude Sonnet 4.6 and GPT-5.4",
      "overallLabel": "Overall 90.4%",
      "pct": "90.4%",
      "percent": "90.4 percent",
      "runs": "9,072",
      "runsLabel": "n=9,072 Claude Sonnet 4.6 and GPT-5.4 runs"
    },
    "interfaces": {
      "no_feedback": {
        "label": "Chatbot",
        "n": "3,024",
        "nLabel": "n=3,024",
        "pct": "82.4%",
        "percent": "82.4 percent",
        "value": 82.4
      },
      "self_repair": {
        "label": "Self-repair",
        "n": "3,024",
        "nLabel": "n=3,024",
        "pct": "92.1%",
        "percent": "92.1 percent",
        "value": 92.1
      },
      "tool_loop": {
        "label": "Constrained agent",
        "n": "3,024",
        "nLabel": "n=3,024",
        "pct": "96.6%",
        "percent": "96.6 percent",
        "value": 96.6
      }
    },
    "models": {
      "codex": {
        "label": "GPT-5.4",
        "n": "4,536",
        "nLabel": "n=4,536",
        "pct": "90.0%",
        "percent": "90.0 percent",
        "value": 90.0
      },
      "sonnet": {
        "label": "Claude Sonnet 4.6",
        "n": "4,536",
        "nLabel": "n=4,536",
        "pct": "90.8%",
        "percent": "90.8 percent",
        "value": 90.8
      }
    },
    "outcome": {
      "definition": "A run counts as a task success when its reported estimates match the stored reference output within the declared tolerances and pass the integrity checks.",
      "diagnostics": "Execution status, result-file shape, and method evidence stay diagnostics and do not enter the primary outcome.",
      "integrity": "The integrity checks require the run to solve the task from the declared inputs.",
      "repair": "The comparison uses the submitted result file after a mechanical, model-blind repair that keeps every reported digit."
    },
    "platforms": {
      "python": {
        "label": "Python",
        "n": "3,024",
        "nLabel": "n=3,024",
        "pct": "95.3%",
        "percent": "95.3 percent",
        "value": 95.3
      },
      "r": {
        "label": "R",
        "n": "3,024",
        "nLabel": "n=3,024",
        "pct": "94.6%",
        "percent": "94.6 percent",
        "value": 94.6
      },
      "stata": {
        "label": "Stata",
        "n": "3,024",
        "nLabel": "n=3,024",
        "pct": "81.3%",
        "percent": "81.3 percent",
        "value": 81.3
      }
    },
    "prompts": {
      "few_shot": {
        "label": "Few-shot",
        "n": "4,536",
        "nLabel": "n=4,536",
        "pct": "92.7%",
        "percent": "92.7 percent",
        "value": 92.7
      },
      "zero_shot": {
        "label": "Zero-shot",
        "n": "4,536",
        "nLabel": "n=4,536",
        "pct": "88.0%",
        "percent": "88.0 percent",
        "value": 88.0
      }
    },
    "sample": {
      "designLine": "3 platforms \u00d7 2 prompts \u00d7 3 agency conditions \u00b7 21 tasks \u00b7 12 repetitions \u00b7 2 models \u00b7 N=9,072 runs",
      "interfaceCount": "3",
      "modelArmCount": "2",
      "modelArmLine": "The main estimates use Claude Sonnet 4.6 and GPT-5.4 \u00b7 9,072 runs in this summary",
      "modelArmWord": "two",
      "modelArmsPhrase": "two models",
      "modelLabels": "Claude Sonnet 4.6, GPT-5.4",
      "platformCount": "3",
      "promptCount": "2",
      "repCount": "12",
      "runs": "9,072",
      "runsPerArm": "2,268",
      "runsPhrase": "9,072 scored runs",
      "runsTasksLine": "N=9,072 runs across 21 tasks",
      "scopeLabel": "Paper sample",
      "tasksCount": "21",
      "tasksHeading": "Twenty-one analyzed econometric tasks",
      "tasksPhrase": "21 econometric tasks",
      "totalLabel": "N=9,072"
    },
    "sentences": {
      "cost": "Costs are reported separately by model. Claude costs are billed amounts. GPT-5.4 costs use API list prices.",
      "design": "We cross three software platforms, two prompt conditions, and three agency conditions across 21 econometric tasks, and repeat the whole design on two models.",
      "hero": "Task success for Claude Sonnet 4.6 and GPT-5.4 is 90.4 percent. The rate goes from 82.4 percent under the chatbot to 96.6 percent under the constrained agent.",
      "interface": "Task success is 82.4% for Chatbot, 92.1% for Self-repair, and 96.6% for Constrained agent.",
      "model": "Task success is 90.8% for Claude Sonnet 4.6 and 90.0% for GPT-5.4.",
      "platform": "Task success is 95.3% for Python, 94.6% for R, and 81.3% for Stata.",
      "platformGap": "The highest platform leads the lowest by 14.0 percentage points.",
      "prompt": "Task success is 88.0% for Zero-shot and 92.7% for Few-shot.",
      "taskSpread": "The lowest rates are 63.9% for Arellano-Bond GMM, 80.1% for Two-way clustered OLS, and 83.6% for Chi-square independence test."
    },
    "taskHighlights": {
      "discretion": "Arellano-Bond GMM 63.9% \u00b7 Two-way clustered OLS 80.1% \u00b7 Chi-square independence test 83.6%",
      "routine": "Probit regression 97.0% \u00b7 Poisson regression 97.0% \u00b7 Logit regression 96.8% \u00b7 Cox regression 96.1% \u00b7 Cook's distance 95.6%"
    }
  },
  "platforms": [
    {
      "criteria": [],
      "id": "python",
      "label": "Python",
      "meanCost": 0,
      "n": 3024,
      "records": 3024,
      "success": 95.3
    },
    {
      "criteria": [],
      "id": "r",
      "label": "R",
      "meanCost": 0,
      "n": 3024,
      "records": 3024,
      "success": 94.6
    },
    {
      "criteria": [],
      "id": "stata",
      "label": "Stata",
      "meanCost": 0,
      "n": 3024,
      "records": 3024,
      "success": 81.3
    }
  ],
  "promptModes": [
    {
      "id": "few_shot",
      "label": "Few-shot",
      "meanCost": 0,
      "n": 4536,
      "records": 4536,
      "success": 92.7
    },
    {
      "id": "zero_shot",
      "label": "Zero-shot",
      "meanCost": 0,
      "n": 4536,
      "records": 4536,
      "success": 88.0
    }
  ],
  "stats": [
    {
      "label": "analyzed econometric tasks",
      "value": "21"
    },
    {
      "label": "scored runs (2 models, 2 prompt conditions)",
      "value": "9,072"
    },
    {
      "label": "task success, paper sample",
      "value": "90.4%"
    }
  ],
  "tasks": [
    {
      "group": "Binary regression",
      "id": "probit",
      "level": "strong",
      "meanCost": 0.0,
      "records": 432,
      "success": 97.0,
      "title": "Probit regression"
    },
    {
      "group": "Binary regression",
      "id": "logit",
      "level": "strong",
      "meanCost": 0.0,
      "records": 432,
      "success": 96.8,
      "title": "Logit regression"
    },
    {
      "group": "OLS",
      "id": "ols_robust",
      "level": "strong",
      "meanCost": 0.0,
      "records": 432,
      "success": 91.2,
      "title": "OLS with robust standard errors"
    },
    {
      "group": "OLS",
      "id": "cluster_ols",
      "level": "strong",
      "meanCost": 0.0,
      "records": 432,
      "success": 95.4,
      "title": "One-way clustered OLS"
    },
    {
      "group": "OLS",
      "id": "two_way_cluster",
      "level": "strong",
      "meanCost": 0.0,
      "records": 432,
      "success": 80.1,
      "title": "Two-way clustered OLS"
    },
    {
      "group": "Difference-in-differences",
      "id": "staggered_did",
      "level": "strong",
      "meanCost": 0.0,
      "records": 432,
      "success": 95.1,
      "title": "Staggered DiD"
    },
    {
      "group": "Difference-in-differences",
      "id": "twoway_fe",
      "level": "strong",
      "meanCost": 0.0,
      "records": 432,
      "success": 92.6,
      "title": "Two-way fixed effects"
    },
    {
      "group": "Instrumental variables",
      "id": "iv2sls_robust",
      "level": "strong",
      "meanCost": 0.0,
      "records": 432,
      "success": 93.5,
      "title": "Two-stage least squares"
    },
    {
      "group": "Regression discontinuity",
      "id": "rdd_sharp",
      "level": "strong",
      "meanCost": 0.0,
      "records": 432,
      "success": 94.7,
      "title": "Regression discontinuity"
    },
    {
      "group": "Time series",
      "id": "arima_forecast",
      "level": "strong",
      "meanCost": 0.0,
      "records": 432,
      "success": 88.0,
      "title": "ARIMA forecast"
    },
    {
      "group": "Time series",
      "id": "var_granger",
      "level": "strong",
      "meanCost": 0.0,
      "records": 432,
      "success": 94.9,
      "title": "Granger causality"
    },
    {
      "group": "Time series",
      "id": "var_irf",
      "level": "strong",
      "meanCost": 0.0,
      "records": 432,
      "success": 86.8,
      "title": "VAR impulse response"
    },
    {
      "group": "Time series",
      "id": "lp_irf",
      "level": "strong",
      "meanCost": 0.0,
      "records": 432,
      "success": 88.4,
      "title": "Local-projection impulse response"
    },
    {
      "group": "Poisson regression",
      "id": "poisson_offset",
      "level": "strong",
      "meanCost": 0.0,
      "records": 432,
      "success": 97.0,
      "title": "Poisson regression"
    },
    {
      "group": "Other",
      "id": "cooks_distance",
      "level": "strong",
      "meanCost": 0.0,
      "records": 432,
      "success": 95.6,
      "title": "Cook's distance"
    },
    {
      "group": "Other",
      "id": "vif",
      "level": "strong",
      "meanCost": 0.0,
      "records": 432,
      "success": 93.1,
      "title": "Variance inflation factors"
    },
    {
      "group": "Other",
      "id": "dynpanel_ab",
      "level": "mixed",
      "meanCost": 0.0,
      "records": 432,
      "success": 63.9,
      "title": "Arellano-Bond GMM"
    },
    {
      "group": "Other",
      "id": "ttest_welch",
      "level": "strong",
      "meanCost": 0.0,
      "records": 432,
      "success": 90.7,
      "title": "Welch t-test"
    },
    {
      "group": "Other",
      "id": "anova_oneway",
      "level": "strong",
      "meanCost": 0.0,
      "records": 432,
      "success": 83.8,
      "title": "One-way ANOVA"
    },
    {
      "group": "Other",
      "id": "chisq_indep",
      "level": "strong",
      "meanCost": 0.0,
      "records": 432,
      "success": 83.6,
      "title": "Chi-square independence test"
    },
    {
      "group": "Other",
      "id": "cox_hazard",
      "level": "strong",
      "meanCost": 0.0,
      "records": 432,
      "success": 96.1,
      "title": "Cox regression"
    }
  ]
}
