{
 "generatedAt": "2026-10-06T11:55:52.900Z",
 "setup": {
  "codex": "0.160.1 (codex exec --json, code mode)",
  "model": "gpt-6-astra, reasoning low",
  "otherMcpServersAndPlugins": "disabled with -c for both arms",
  "runs": "each task twice per arm; arms A and B of the same task and run concurrently",
  "logs": "apps/studio/.wrangler/efficiency/runs/<id>/ (git-ignored; copied summaries in docs/verification/mcp-efficiency/data/)"
 },
 "facts": [
  {
   "id": "T1.base.A.inputTokens",
   "value": 94698,
   "unit": "tokens",
   "what": {
    "en": "T1, Codex alone: input tokens, mean of 2 runs",
    "ja": "T1・Codexのみ：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T1-A-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T1-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "T1.base.A.inputTokensRange",
   "value": [
    94359,
    95036
   ],
   "unit": "tokens",
   "what": {
    "en": "T1, Codex alone: input tokens, lowest and highest run",
    "ja": "T1・Codexのみ：入力トークンの最小と最大"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T1-A-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T1-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "T1.base.A.uncachedInputTokens",
   "value": 13674,
   "unit": "tokens",
   "what": {
    "en": "T1, Codex alone: uncached input tokens, mean",
    "ja": "T1・Codexのみ：キャッシュされない入力（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T1-A-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T1-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "T1.base.A.outputTokens",
   "value": 4578,
   "unit": "tokens",
   "what": {
    "en": "T1, Codex alone: output tokens, mean",
    "ja": "T1・Codexのみ：出力トークン（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T1-A-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T1-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "T1.base.A.requests",
   "value": 4.0,
   "unit": "model requests",
   "what": {
    "en": "T1, Codex alone: model requests, mean",
    "ja": "T1・Codexのみ：モデルへの要求回数（平均）"
   },
   "source": [
    "apps/studio/.wrangler/efficiency/runs/base-T1-A-r1/rollout.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T1-A-r2/rollout.jsonl"
   ]
  },
  {
   "id": "T1.base.A.seconds",
   "value": 125,
   "unit": "s",
   "what": {
    "en": "T1, Codex alone: wall time, mean",
    "ja": "T1・Codexのみ：所要時間（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T1-A-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T1-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "T1.base.A.passed",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "T1, Codex alone: runs passing the scripted checks",
    "ja": "T1・Codexのみ：採点スクリプトの合格数"
   },
   "source": [
    "apps/studio/.wrangler/efficiency/runs/base-T1-A-r1/score.json",
    "apps/studio/.wrangler/efficiency/runs/base-T1-A-r2/score.json"
   ]
  },
  {
   "id": "T1.base.A.apiCostUsd",
   "value": 0.447,
   "unit": "USD (estimate)",
   "what": {
    "en": "T1, Codex alone: API list-price estimate (gpt-6-astra Standard: $10 input, $1 cached, $50 output per 1M); the runs used a ChatGPT sign-in, not API billing",
    "ja": "T1・Codexのみ：API定価での推定費用（gpt-6-astra標準：入力$10・キャッシュ$1・出力$50／100万）。実行はChatGPTのサインインでAPI課金ではない"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T1-A-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T1-A-r2/codex.jsonl",
    "https://developers.openai.com/api/docs/pricing (read 2026-10-06)"
   ]
  },
  {
   "id": "T1.base.B.inputTokens",
   "value": 481018,
   "unit": "tokens",
   "what": {
    "en": "T1, Codex + Laydyne MCP, before: input tokens, mean of 2 runs",
    "ja": "T1・Codex＋Laydyne（変更前）：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T1-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T1-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "T1.base.B.inputTokensRange",
   "value": [
    447265,
    514771
   ],
   "unit": "tokens",
   "what": {
    "en": "T1, Codex + Laydyne MCP, before: input tokens, lowest and highest run",
    "ja": "T1・Codex＋Laydyne（変更前）：入力トークンの最小と最大"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T1-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T1-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "T1.base.B.uncachedInputTokens",
   "value": 49850,
   "unit": "tokens",
   "what": {
    "en": "T1, Codex + Laydyne MCP, before: uncached input tokens, mean",
    "ja": "T1・Codex＋Laydyne（変更前）：キャッシュされない入力（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T1-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T1-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "T1.base.B.outputTokens",
   "value": 2214,
   "unit": "tokens",
   "what": {
    "en": "T1, Codex + Laydyne MCP, before: output tokens, mean",
    "ja": "T1・Codex＋Laydyne（変更前）：出力トークン（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T1-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T1-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "T1.base.B.requests",
   "value": 11.0,
   "unit": "model requests",
   "what": {
    "en": "T1, Codex + Laydyne MCP, before: model requests, mean",
    "ja": "T1・Codex＋Laydyne（変更前）：モデルへの要求回数（平均）"
   },
   "source": [
    "apps/studio/.wrangler/efficiency/runs/base-T1-B-r1/rollout.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T1-B-r2/rollout.jsonl"
   ]
  },
  {
   "id": "T1.base.B.seconds",
   "value": 114,
   "unit": "s",
   "what": {
    "en": "T1, Codex + Laydyne MCP, before: wall time, mean",
    "ja": "T1・Codex＋Laydyne（変更前）：所要時間（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T1-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T1-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "T1.base.B.passed",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "T1, Codex + Laydyne MCP, before: runs passing the scripted checks",
    "ja": "T1・Codex＋Laydyne（変更前）：採点スクリプトの合格数"
   },
   "source": [
    "apps/studio/.wrangler/efficiency/runs/base-T1-B-r1/score.json",
    "apps/studio/.wrangler/efficiency/runs/base-T1-B-r2/score.json"
   ]
  },
  {
   "id": "T1.base.B.apiCostUsd",
   "value": 1.04,
   "unit": "USD (estimate)",
   "what": {
    "en": "T1, Codex + Laydyne MCP, before: API list-price estimate (gpt-6-astra Standard: $10 input, $1 cached, $50 output per 1M); the runs used a ChatGPT sign-in, not API billing",
    "ja": "T1・Codex＋Laydyne（変更前）：API定価での推定費用（gpt-6-astra標準：入力$10・キャッシュ$1・出力$50／100万）。実行はChatGPTのサインインでAPI課金ではない"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T1-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T1-B-r2/codex.jsonl",
    "https://developers.openai.com/api/docs/pricing (read 2026-10-06)"
   ]
  },
  {
   "id": "T1.light.B.inputTokens",
   "value": 463740,
   "unit": "tokens",
   "what": {
    "en": "T1, Codex + Laydyne MCP, lighter: input tokens, mean of 2 runs",
    "ja": "T1・Codex＋Laydyne（軽量化後）：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/light-T1-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/light-T1-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "T1.light.B.inputTokensRange",
   "value": [
    460378,
    467103
   ],
   "unit": "tokens",
   "what": {
    "en": "T1, Codex + Laydyne MCP, lighter: input tokens, lowest and highest run",
    "ja": "T1・Codex＋Laydyne（軽量化後）：入力トークンの最小と最大"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/light-T1-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/light-T1-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "T1.light.B.uncachedInputTokens",
   "value": 50684,
   "unit": "tokens",
   "what": {
    "en": "T1, Codex + Laydyne MCP, lighter: uncached input tokens, mean",
    "ja": "T1・Codex＋Laydyne（軽量化後）：キャッシュされない入力（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/light-T1-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/light-T1-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "T1.light.B.outputTokens",
   "value": 1868,
   "unit": "tokens",
   "what": {
    "en": "T1, Codex + Laydyne MCP, lighter: output tokens, mean",
    "ja": "T1・Codex＋Laydyne（軽量化後）：出力トークン（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/light-T1-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/light-T1-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "T1.light.B.requests",
   "value": 10.0,
   "unit": "model requests",
   "what": {
    "en": "T1, Codex + Laydyne MCP, lighter: model requests, mean",
    "ja": "T1・Codex＋Laydyne（軽量化後）：モデルへの要求回数（平均）"
   },
   "source": [
    "apps/studio/.wrangler/efficiency/runs/light-T1-B-r1/rollout.jsonl",
    "apps/studio/.wrangler/efficiency/runs/light-T1-B-r2/rollout.jsonl"
   ]
  },
  {
   "id": "T1.light.B.seconds",
   "value": 116,
   "unit": "s",
   "what": {
    "en": "T1, Codex + Laydyne MCP, lighter: wall time, mean",
    "ja": "T1・Codex＋Laydyne（軽量化後）：所要時間（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/light-T1-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/light-T1-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "T1.light.B.passed",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "T1, Codex + Laydyne MCP, lighter: runs passing the scripted checks",
    "ja": "T1・Codex＋Laydyne（軽量化後）：採点スクリプトの合格数"
   },
   "source": [
    "apps/studio/.wrangler/efficiency/runs/light-T1-B-r1/score.json",
    "apps/studio/.wrangler/efficiency/runs/light-T1-B-r2/score.json"
   ]
  },
  {
   "id": "T1.light.B.apiCostUsd",
   "value": 1.013,
   "unit": "USD (estimate)",
   "what": {
    "en": "T1, Codex + Laydyne MCP, lighter: API list-price estimate (gpt-6-astra Standard: $10 input, $1 cached, $50 output per 1M); the runs used a ChatGPT sign-in, not API billing",
    "ja": "T1・Codex＋Laydyne（軽量化後）：API定価での推定費用（gpt-6-astra標準：入力$10・キャッシュ$1・出力$50／100万）。実行はChatGPTのサインインでAPI課金ではない"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/light-T1-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/light-T1-B-r2/codex.jsonl",
    "https://developers.openai.com/api/docs/pricing (read 2026-10-06)"
   ]
  },
  {
   "id": "T2.base.A.inputTokens",
   "value": 89478,
   "unit": "tokens",
   "what": {
    "en": "T2, Codex alone: input tokens, mean of 2 runs",
    "ja": "T2・Codexのみ：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T2-A-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T2-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "T2.base.A.inputTokensRange",
   "value": [
    72282,
    106674
   ],
   "unit": "tokens",
   "what": {
    "en": "T2, Codex alone: input tokens, lowest and highest run",
    "ja": "T2・Codexのみ：入力トークンの最小と最大"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T2-A-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T2-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "T2.base.A.uncachedInputTokens",
   "value": 19974,
   "unit": "tokens",
   "what": {
    "en": "T2, Codex alone: uncached input tokens, mean",
    "ja": "T2・Codexのみ：キャッシュされない入力（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T2-A-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T2-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "T2.base.A.outputTokens",
   "value": 1984,
   "unit": "tokens",
   "what": {
    "en": "T2, Codex alone: output tokens, mean",
    "ja": "T2・Codexのみ：出力トークン（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T2-A-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T2-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "T2.base.A.requests",
   "value": 3.5,
   "unit": "model requests",
   "what": {
    "en": "T2, Codex alone: model requests, mean",
    "ja": "T2・Codexのみ：モデルへの要求回数（平均）"
   },
   "source": [
    "apps/studio/.wrangler/efficiency/runs/base-T2-A-r1/rollout.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T2-A-r2/rollout.jsonl"
   ]
  },
  {
   "id": "T2.base.A.seconds",
   "value": 62,
   "unit": "s",
   "what": {
    "en": "T2, Codex alone: wall time, mean",
    "ja": "T2・Codexのみ：所要時間（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T2-A-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T2-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "T2.base.A.passed",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "T2, Codex alone: runs passing the scripted checks",
    "ja": "T2・Codexのみ：採点スクリプトの合格数"
   },
   "source": [
    "apps/studio/.wrangler/efficiency/runs/base-T2-A-r1/score.json",
    "apps/studio/.wrangler/efficiency/runs/base-T2-A-r2/score.json"
   ]
  },
  {
   "id": "T2.base.A.apiCostUsd",
   "value": 0.368,
   "unit": "USD (estimate)",
   "what": {
    "en": "T2, Codex alone: API list-price estimate (gpt-6-astra Standard: $10 input, $1 cached, $50 output per 1M); the runs used a ChatGPT sign-in, not API billing",
    "ja": "T2・Codexのみ：API定価での推定費用（gpt-6-astra標準：入力$10・キャッシュ$1・出力$50／100万）。実行はChatGPTのサインインでAPI課金ではない"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T2-A-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T2-A-r2/codex.jsonl",
    "https://developers.openai.com/api/docs/pricing (read 2026-10-06)"
   ]
  },
  {
   "id": "T2.base.B.inputTokens",
   "value": 453216,
   "unit": "tokens",
   "what": {
    "en": "T2, Codex + Laydyne MCP, before: input tokens, mean of 2 runs",
    "ja": "T2・Codex＋Laydyne（変更前）：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T2-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T2-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "T2.base.B.inputTokensRange",
   "value": [
    407073,
    499358
   ],
   "unit": "tokens",
   "what": {
    "en": "T2, Codex + Laydyne MCP, before: input tokens, lowest and highest run",
    "ja": "T2・Codex＋Laydyne（変更前）：入力トークンの最小と最大"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T2-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T2-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "T2.base.B.uncachedInputTokens",
   "value": 55968,
   "unit": "tokens",
   "what": {
    "en": "T2, Codex + Laydyne MCP, before: uncached input tokens, mean",
    "ja": "T2・Codex＋Laydyne（変更前）：キャッシュされない入力（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T2-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T2-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "T2.base.B.outputTokens",
   "value": 1420,
   "unit": "tokens",
   "what": {
    "en": "T2, Codex + Laydyne MCP, before: output tokens, mean",
    "ja": "T2・Codex＋Laydyne（変更前）：出力トークン（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T2-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T2-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "T2.base.B.requests",
   "value": 11.0,
   "unit": "model requests",
   "what": {
    "en": "T2, Codex + Laydyne MCP, before: model requests, mean",
    "ja": "T2・Codex＋Laydyne（変更前）：モデルへの要求回数（平均）"
   },
   "source": [
    "apps/studio/.wrangler/efficiency/runs/base-T2-B-r1/rollout.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T2-B-r2/rollout.jsonl"
   ]
  },
  {
   "id": "T2.base.B.seconds",
   "value": 77,
   "unit": "s",
   "what": {
    "en": "T2, Codex + Laydyne MCP, before: wall time, mean",
    "ja": "T2・Codex＋Laydyne（変更前）：所要時間（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T2-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T2-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "T2.base.B.passed",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "T2, Codex + Laydyne MCP, before: runs passing the scripted checks",
    "ja": "T2・Codex＋Laydyne（変更前）：採点スクリプトの合格数"
   },
   "source": [
    "apps/studio/.wrangler/efficiency/runs/base-T2-B-r1/score.json",
    "apps/studio/.wrangler/efficiency/runs/base-T2-B-r2/score.json"
   ]
  },
  {
   "id": "T2.base.B.apiCostUsd",
   "value": 1.028,
   "unit": "USD (estimate)",
   "what": {
    "en": "T2, Codex + Laydyne MCP, before: API list-price estimate (gpt-6-astra Standard: $10 input, $1 cached, $50 output per 1M); the runs used a ChatGPT sign-in, not API billing",
    "ja": "T2・Codex＋Laydyne（変更前）：API定価での推定費用（gpt-6-astra標準：入力$10・キャッシュ$1・出力$50／100万）。実行はChatGPTのサインインでAPI課金ではない"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T2-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T2-B-r2/codex.jsonl",
    "https://developers.openai.com/api/docs/pricing (read 2026-10-06)"
   ]
  },
  {
   "id": "T2.light.B.inputTokens",
   "value": 405461,
   "unit": "tokens",
   "what": {
    "en": "T2, Codex + Laydyne MCP, lighter: input tokens, mean of 2 runs",
    "ja": "T2・Codex＋Laydyne（軽量化後）：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/light-T2-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/light-T2-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "T2.light.B.inputTokensRange",
   "value": [
    402677,
    408245
   ],
   "unit": "tokens",
   "what": {
    "en": "T2, Codex + Laydyne MCP, lighter: input tokens, lowest and highest run",
    "ja": "T2・Codex＋Laydyne（軽量化後）：入力トークンの最小と最大"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/light-T2-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/light-T2-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "T2.light.B.uncachedInputTokens",
   "value": 54037,
   "unit": "tokens",
   "what": {
    "en": "T2, Codex + Laydyne MCP, lighter: uncached input tokens, mean",
    "ja": "T2・Codex＋Laydyne（軽量化後）：キャッシュされない入力（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/light-T2-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/light-T2-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "T2.light.B.outputTokens",
   "value": 1153,
   "unit": "tokens",
   "what": {
    "en": "T2, Codex + Laydyne MCP, lighter: output tokens, mean",
    "ja": "T2・Codex＋Laydyne（軽量化後）：出力トークン（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/light-T2-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/light-T2-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "T2.light.B.requests",
   "value": 9.0,
   "unit": "model requests",
   "what": {
    "en": "T2, Codex + Laydyne MCP, lighter: model requests, mean",
    "ja": "T2・Codex＋Laydyne（軽量化後）：モデルへの要求回数（平均）"
   },
   "source": [
    "apps/studio/.wrangler/efficiency/runs/light-T2-B-r1/rollout.jsonl",
    "apps/studio/.wrangler/efficiency/runs/light-T2-B-r2/rollout.jsonl"
   ]
  },
  {
   "id": "T2.light.B.seconds",
   "value": 88,
   "unit": "s",
   "what": {
    "en": "T2, Codex + Laydyne MCP, lighter: wall time, mean",
    "ja": "T2・Codex＋Laydyne（軽量化後）：所要時間（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/light-T2-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/light-T2-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "T2.light.B.passed",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "T2, Codex + Laydyne MCP, lighter: runs passing the scripted checks",
    "ja": "T2・Codex＋Laydyne（軽量化後）：採点スクリプトの合格数"
   },
   "source": [
    "apps/studio/.wrangler/efficiency/runs/light-T2-B-r1/score.json",
    "apps/studio/.wrangler/efficiency/runs/light-T2-B-r2/score.json"
   ]
  },
  {
   "id": "T2.light.B.apiCostUsd",
   "value": 0.949,
   "unit": "USD (estimate)",
   "what": {
    "en": "T2, Codex + Laydyne MCP, lighter: API list-price estimate (gpt-6-astra Standard: $10 input, $1 cached, $50 output per 1M); the runs used a ChatGPT sign-in, not API billing",
    "ja": "T2・Codex＋Laydyne（軽量化後）：API定価での推定費用（gpt-6-astra標準：入力$10・キャッシュ$1・出力$50／100万）。実行はChatGPTのサインインでAPI課金ではない"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/light-T2-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/light-T2-B-r2/codex.jsonl",
    "https://developers.openai.com/api/docs/pricing (read 2026-10-06)"
   ]
  },
  {
   "id": "T3.base.A.inputTokens",
   "value": 107223,
   "unit": "tokens",
   "what": {
    "en": "T3, Codex alone: input tokens, mean of 2 runs",
    "ja": "T3・Codexのみ：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T3-A-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T3-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "T3.base.A.inputTokensRange",
   "value": [
    96658,
    117788
   ],
   "unit": "tokens",
   "what": {
    "en": "T3, Codex alone: input tokens, lowest and highest run",
    "ja": "T3・Codexのみ：入力トークンの最小と最大"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T3-A-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T3-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "T3.base.A.uncachedInputTokens",
   "value": 18647,
   "unit": "tokens",
   "what": {
    "en": "T3, Codex alone: uncached input tokens, mean",
    "ja": "T3・Codexのみ：キャッシュされない入力（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T3-A-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T3-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "T3.base.A.outputTokens",
   "value": 4984,
   "unit": "tokens",
   "what": {
    "en": "T3, Codex alone: output tokens, mean",
    "ja": "T3・Codexのみ：出力トークン（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T3-A-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T3-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "T3.base.A.requests",
   "value": 4.5,
   "unit": "model requests",
   "what": {
    "en": "T3, Codex alone: model requests, mean",
    "ja": "T3・Codexのみ：モデルへの要求回数（平均）"
   },
   "source": [
    "apps/studio/.wrangler/efficiency/runs/base-T3-A-r1/rollout.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T3-A-r2/rollout.jsonl"
   ]
  },
  {
   "id": "T3.base.A.seconds",
   "value": 160,
   "unit": "s",
   "what": {
    "en": "T3, Codex alone: wall time, mean",
    "ja": "T3・Codexのみ：所要時間（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T3-A-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T3-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "T3.base.A.passed",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "T3, Codex alone: runs passing the scripted checks",
    "ja": "T3・Codexのみ：採点スクリプトの合格数"
   },
   "source": [
    "apps/studio/.wrangler/efficiency/runs/base-T3-A-r1/score.json",
    "apps/studio/.wrangler/efficiency/runs/base-T3-A-r2/score.json"
   ]
  },
  {
   "id": "T3.base.A.apiCostUsd",
   "value": 0.524,
   "unit": "USD (estimate)",
   "what": {
    "en": "T3, Codex alone: API list-price estimate (gpt-6-astra Standard: $10 input, $1 cached, $50 output per 1M); the runs used a ChatGPT sign-in, not API billing",
    "ja": "T3・Codexのみ：API定価での推定費用（gpt-6-astra標準：入力$10・キャッシュ$1・出力$50／100万）。実行はChatGPTのサインインでAPI課金ではない"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T3-A-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T3-A-r2/codex.jsonl",
    "https://developers.openai.com/api/docs/pricing (read 2026-10-06)"
   ]
  },
  {
   "id": "T3.base.B.inputTokens",
   "value": 533375,
   "unit": "tokens",
   "what": {
    "en": "T3, Codex + Laydyne MCP, before: input tokens, mean of 2 runs",
    "ja": "T3・Codex＋Laydyne（変更前）：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "T3.base.B.inputTokensRange",
   "value": [
    507096,
    559654
   ],
   "unit": "tokens",
   "what": {
    "en": "T3, Codex + Laydyne MCP, before: input tokens, lowest and highest run",
    "ja": "T3・Codex＋Laydyne（変更前）：入力トークンの最小と最大"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "T3.base.B.uncachedInputTokens",
   "value": 57599,
   "unit": "tokens",
   "what": {
    "en": "T3, Codex + Laydyne MCP, before: uncached input tokens, mean",
    "ja": "T3・Codex＋Laydyne（変更前）：キャッシュされない入力（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "T3.base.B.outputTokens",
   "value": 3764,
   "unit": "tokens",
   "what": {
    "en": "T3, Codex + Laydyne MCP, before: output tokens, mean",
    "ja": "T3・Codex＋Laydyne（変更前）：出力トークン（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "T3.base.B.requests",
   "value": 12.0,
   "unit": "model requests",
   "what": {
    "en": "T3, Codex + Laydyne MCP, before: model requests, mean",
    "ja": "T3・Codex＋Laydyne（変更前）：モデルへの要求回数（平均）"
   },
   "source": [
    "apps/studio/.wrangler/efficiency/runs/base-T3-B-r1/rollout.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T3-B-r2/rollout.jsonl"
   ]
  },
  {
   "id": "T3.base.B.seconds",
   "value": 152,
   "unit": "s",
   "what": {
    "en": "T3, Codex + Laydyne MCP, before: wall time, mean",
    "ja": "T3・Codex＋Laydyne（変更前）：所要時間（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "T3.base.B.passed",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "T3, Codex + Laydyne MCP, before: runs passing the scripted checks",
    "ja": "T3・Codex＋Laydyne（変更前）：採点スクリプトの合格数"
   },
   "source": [
    "apps/studio/.wrangler/efficiency/runs/base-T3-B-r1/score.json",
    "apps/studio/.wrangler/efficiency/runs/base-T3-B-r2/score.json"
   ]
  },
  {
   "id": "T3.base.B.apiCostUsd",
   "value": 1.24,
   "unit": "USD (estimate)",
   "what": {
    "en": "T3, Codex + Laydyne MCP, before: API list-price estimate (gpt-6-astra Standard: $10 input, $1 cached, $50 output per 1M); the runs used a ChatGPT sign-in, not API billing",
    "ja": "T3・Codex＋Laydyne（変更前）：API定価での推定費用（gpt-6-astra標準：入力$10・キャッシュ$1・出力$50／100万）。実行はChatGPTのサインインでAPI課金ではない"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T3-B-r2/codex.jsonl",
    "https://developers.openai.com/api/docs/pricing (read 2026-10-06)"
   ]
  },
  {
   "id": "T3.light.B.inputTokens",
   "value": 834434,
   "unit": "tokens",
   "what": {
    "en": "T3, Codex + Laydyne MCP, lighter: input tokens, mean of 2 runs",
    "ja": "T3・Codex＋Laydyne（軽量化後）：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/light-T3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/light-T3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "T3.light.B.inputTokensRange",
   "value": [
    626050,
    1042819
   ],
   "unit": "tokens",
   "what": {
    "en": "T3, Codex + Laydyne MCP, lighter: input tokens, lowest and highest run",
    "ja": "T3・Codex＋Laydyne（軽量化後）：入力トークンの最小と最大"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/light-T3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/light-T3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "T3.light.B.uncachedInputTokens",
   "value": 65858,
   "unit": "tokens",
   "what": {
    "en": "T3, Codex + Laydyne MCP, lighter: uncached input tokens, mean",
    "ja": "T3・Codex＋Laydyne（軽量化後）：キャッシュされない入力（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/light-T3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/light-T3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "T3.light.B.outputTokens",
   "value": 4096,
   "unit": "tokens",
   "what": {
    "en": "T3, Codex + Laydyne MCP, lighter: output tokens, mean",
    "ja": "T3・Codex＋Laydyne（軽量化後）：出力トークン（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/light-T3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/light-T3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "T3.light.B.requests",
   "value": 16.0,
   "unit": "model requests",
   "what": {
    "en": "T3, Codex + Laydyne MCP, lighter: model requests, mean",
    "ja": "T3・Codex＋Laydyne（軽量化後）：モデルへの要求回数（平均）"
   },
   "source": [
    "apps/studio/.wrangler/efficiency/runs/light-T3-B-r1/rollout.jsonl",
    "apps/studio/.wrangler/efficiency/runs/light-T3-B-r2/rollout.jsonl"
   ]
  },
  {
   "id": "T3.light.B.seconds",
   "value": 206,
   "unit": "s",
   "what": {
    "en": "T3, Codex + Laydyne MCP, lighter: wall time, mean",
    "ja": "T3・Codex＋Laydyne（軽量化後）：所要時間（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/light-T3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/light-T3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "T3.light.B.passed",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "T3, Codex + Laydyne MCP, lighter: runs passing the scripted checks",
    "ja": "T3・Codex＋Laydyne（軽量化後）：採点スクリプトの合格数"
   },
   "source": [
    "apps/studio/.wrangler/efficiency/runs/light-T3-B-r1/score.json",
    "apps/studio/.wrangler/efficiency/runs/light-T3-B-r2/score.json"
   ]
  },
  {
   "id": "T3.light.B.apiCostUsd",
   "value": 1.632,
   "unit": "USD (estimate)",
   "what": {
    "en": "T3, Codex + Laydyne MCP, lighter: API list-price estimate (gpt-6-astra Standard: $10 input, $1 cached, $50 output per 1M); the runs used a ChatGPT sign-in, not API billing",
    "ja": "T3・Codex＋Laydyne（軽量化後）：API定価での推定費用（gpt-6-astra標準：入力$10・キャッシュ$1・出力$50／100万）。実行はChatGPTのサインインでAPI課金ではない"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/light-T3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/light-T3-B-r2/codex.jsonl",
    "https://developers.openai.com/api/docs/pricing (read 2026-10-06)"
   ]
  },
  {
   "id": "T4.base.A.inputTokens",
   "value": 246915,
   "unit": "tokens",
   "what": {
    "en": "T4, Codex alone: input tokens, mean of 2 runs",
    "ja": "T4・Codexのみ：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T4-A-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T4-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "T4.base.A.inputTokensRange",
   "value": [
    231128,
    262702
   ],
   "unit": "tokens",
   "what": {
    "en": "T4, Codex alone: input tokens, lowest and highest run",
    "ja": "T4・Codexのみ：入力トークンの最小と最大"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T4-A-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T4-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "T4.base.A.uncachedInputTokens",
   "value": 32259,
   "unit": "tokens",
   "what": {
    "en": "T4, Codex alone: uncached input tokens, mean",
    "ja": "T4・Codexのみ：キャッシュされない入力（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T4-A-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T4-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "T4.base.A.outputTokens",
   "value": 3252,
   "unit": "tokens",
   "what": {
    "en": "T4, Codex alone: output tokens, mean",
    "ja": "T4・Codexのみ：出力トークン（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T4-A-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T4-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "T4.base.A.requests",
   "value": 7.5,
   "unit": "model requests",
   "what": {
    "en": "T4, Codex alone: model requests, mean",
    "ja": "T4・Codexのみ：モデルへの要求回数（平均）"
   },
   "source": [
    "apps/studio/.wrangler/efficiency/runs/base-T4-A-r1/rollout.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T4-A-r2/rollout.jsonl"
   ]
  },
  {
   "id": "T4.base.A.seconds",
   "value": 104,
   "unit": "s",
   "what": {
    "en": "T4, Codex alone: wall time, mean",
    "ja": "T4・Codexのみ：所要時間（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T4-A-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T4-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "T4.base.A.passed",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "T4, Codex alone: runs passing the scripted checks",
    "ja": "T4・Codexのみ：採点スクリプトの合格数"
   },
   "source": [
    "apps/studio/.wrangler/efficiency/runs/base-T4-A-r1/score.json",
    "apps/studio/.wrangler/efficiency/runs/base-T4-A-r2/score.json"
   ]
  },
  {
   "id": "T4.base.A.apiCostUsd",
   "value": 0.7,
   "unit": "USD (estimate)",
   "what": {
    "en": "T4, Codex alone: API list-price estimate (gpt-6-astra Standard: $10 input, $1 cached, $50 output per 1M); the runs used a ChatGPT sign-in, not API billing",
    "ja": "T4・Codexのみ：API定価での推定費用（gpt-6-astra標準：入力$10・キャッシュ$1・出力$50／100万）。実行はChatGPTのサインインでAPI課金ではない"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T4-A-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T4-A-r2/codex.jsonl",
    "https://developers.openai.com/api/docs/pricing (read 2026-10-06)"
   ]
  },
  {
   "id": "T4.base.B.inputTokens",
   "value": 655818,
   "unit": "tokens",
   "what": {
    "en": "T4, Codex + Laydyne MCP, before: input tokens, mean of 2 runs",
    "ja": "T4・Codex＋Laydyne（変更前）：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "T4.base.B.inputTokensRange",
   "value": [
    448509,
    863126
   ],
   "unit": "tokens",
   "what": {
    "en": "T4, Codex + Laydyne MCP, before: input tokens, lowest and highest run",
    "ja": "T4・Codex＋Laydyne（変更前）：入力トークンの最小と最大"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "T4.base.B.uncachedInputTokens",
   "value": 70922,
   "unit": "tokens",
   "what": {
    "en": "T4, Codex + Laydyne MCP, before: uncached input tokens, mean",
    "ja": "T4・Codex＋Laydyne（変更前）：キャッシュされない入力（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "T4.base.B.outputTokens",
   "value": 2148,
   "unit": "tokens",
   "what": {
    "en": "T4, Codex + Laydyne MCP, before: output tokens, mean",
    "ja": "T4・Codex＋Laydyne（変更前）：出力トークン（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "T4.base.B.requests",
   "value": 11.5,
   "unit": "model requests",
   "what": {
    "en": "T4, Codex + Laydyne MCP, before: model requests, mean",
    "ja": "T4・Codex＋Laydyne（変更前）：モデルへの要求回数（平均）"
   },
   "source": [
    "apps/studio/.wrangler/efficiency/runs/base-T4-B-r1/rollout.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T4-B-r2/rollout.jsonl"
   ]
  },
  {
   "id": "T4.base.B.seconds",
   "value": 94,
   "unit": "s",
   "what": {
    "en": "T4, Codex + Laydyne MCP, before: wall time, mean",
    "ja": "T4・Codex＋Laydyne（変更前）：所要時間（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "T4.base.B.passed",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "T4, Codex + Laydyne MCP, before: runs passing the scripted checks",
    "ja": "T4・Codex＋Laydyne（変更前）：採点スクリプトの合格数"
   },
   "source": [
    "apps/studio/.wrangler/efficiency/runs/base-T4-B-r1/score.json",
    "apps/studio/.wrangler/efficiency/runs/base-T4-B-r2/score.json"
   ]
  },
  {
   "id": "T4.base.B.apiCostUsd",
   "value": 1.402,
   "unit": "USD (estimate)",
   "what": {
    "en": "T4, Codex + Laydyne MCP, before: API list-price estimate (gpt-6-astra Standard: $10 input, $1 cached, $50 output per 1M); the runs used a ChatGPT sign-in, not API billing",
    "ja": "T4・Codex＋Laydyne（変更前）：API定価での推定費用（gpt-6-astra標準：入力$10・キャッシュ$1・出力$50／100万）。実行はChatGPTのサインインでAPI課金ではない"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/base-T4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/base-T4-B-r2/codex.jsonl",
    "https://developers.openai.com/api/docs/pricing (read 2026-10-06)"
   ]
  },
  {
   "id": "T4.light.B.inputTokens",
   "value": 742522,
   "unit": "tokens",
   "what": {
    "en": "T4, Codex + Laydyne MCP, lighter: input tokens, mean of 2 runs",
    "ja": "T4・Codex＋Laydyne（軽量化後）：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/light-T4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/light-T4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "T4.light.B.inputTokensRange",
   "value": [
    674230,
    810815
   ],
   "unit": "tokens",
   "what": {
    "en": "T4, Codex + Laydyne MCP, lighter: input tokens, lowest and highest run",
    "ja": "T4・Codex＋Laydyne（軽量化後）：入力トークンの最小と最大"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/light-T4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/light-T4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "T4.light.B.uncachedInputTokens",
   "value": 80058,
   "unit": "tokens",
   "what": {
    "en": "T4, Codex + Laydyne MCP, lighter: uncached input tokens, mean",
    "ja": "T4・Codex＋Laydyne（軽量化後）：キャッシュされない入力（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/light-T4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/light-T4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "T4.light.B.outputTokens",
   "value": 3270,
   "unit": "tokens",
   "what": {
    "en": "T4, Codex + Laydyne MCP, lighter: output tokens, mean",
    "ja": "T4・Codex＋Laydyne（軽量化後）：出力トークン（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/light-T4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/light-T4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "T4.light.B.requests",
   "value": 12.5,
   "unit": "model requests",
   "what": {
    "en": "T4, Codex + Laydyne MCP, lighter: model requests, mean",
    "ja": "T4・Codex＋Laydyne（軽量化後）：モデルへの要求回数（平均）"
   },
   "source": [
    "apps/studio/.wrangler/efficiency/runs/light-T4-B-r1/rollout.jsonl",
    "apps/studio/.wrangler/efficiency/runs/light-T4-B-r2/rollout.jsonl"
   ]
  },
  {
   "id": "T4.light.B.seconds",
   "value": 152,
   "unit": "s",
   "what": {
    "en": "T4, Codex + Laydyne MCP, lighter: wall time, mean",
    "ja": "T4・Codex＋Laydyne（軽量化後）：所要時間（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/light-T4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/light-T4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "T4.light.B.passed",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "T4, Codex + Laydyne MCP, lighter: runs passing the scripted checks",
    "ja": "T4・Codex＋Laydyne（軽量化後）：採点スクリプトの合格数"
   },
   "source": [
    "apps/studio/.wrangler/efficiency/runs/light-T4-B-r1/score.json",
    "apps/studio/.wrangler/efficiency/runs/light-T4-B-r2/score.json"
   ]
  },
  {
   "id": "T4.light.B.apiCostUsd",
   "value": 1.627,
   "unit": "USD (estimate)",
   "what": {
    "en": "T4, Codex + Laydyne MCP, lighter: API list-price estimate (gpt-6-astra Standard: $10 input, $1 cached, $50 output per 1M); the runs used a ChatGPT sign-in, not API billing",
    "ja": "T4・Codex＋Laydyne（軽量化後）：API定価での推定費用（gpt-6-astra標準：入力$10・キャッシュ$1・出力$50／100万）。実行はChatGPTのサインインでAPI課金ではない"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/aggregate.json",
    "apps/studio/.wrangler/efficiency/runs/light-T4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/light-T4-B-r2/codex.jsonl",
    "https://developers.openai.com/api/docs/pricing (read 2026-10-06)"
   ]
  },
  {
   "id": "surface.before.toolsListTokens",
   "value": 87278,
   "unit": "tokens",
   "what": {
    "en": "before: tools/list, o200k_base tokens",
    "ja": "変更前：tools/listのトークン数（o200k_base）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/surface-tokens.json",
    "apps/studio/.wrangler/efficiency/surface-before.json",
    "apps/studio/.wrangler/efficiency/codex-dump-before.json"
   ]
  },
  {
   "id": "surface.after.toolsListTokens",
   "value": 39632,
   "unit": "tokens",
   "what": {
    "en": "after: tools/list, o200k_base tokens",
    "ja": "軽量化後：tools/listのトークン数（o200k_base）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/surface-tokens.json",
    "apps/studio/.wrangler/efficiency/surface-after.json",
    "apps/studio/.wrangler/efficiency/codex-dump-after.json"
   ]
  },
  {
   "id": "surface.before.instructionsTokens",
   "value": 1884,
   "unit": "tokens",
   "what": {
    "en": "before: server instructions, tokens",
    "ja": "変更前：サーバー指示文のトークン数"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/surface-tokens.json",
    "apps/studio/.wrangler/efficiency/surface-before.json",
    "apps/studio/.wrangler/efficiency/codex-dump-before.json"
   ]
  },
  {
   "id": "surface.after.instructionsTokens",
   "value": 246,
   "unit": "tokens",
   "what": {
    "en": "after: server instructions, tokens",
    "ja": "軽量化後：サーバー指示文のトークン数"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/surface-tokens.json",
    "apps/studio/.wrangler/efficiency/surface-after.json",
    "apps/studio/.wrangler/efficiency/codex-dump-after.json"
   ]
  },
  {
   "id": "surface.before.loadAllPerRequest",
   "value": 89162,
   "unit": "tokens",
   "what": {
    "en": "before: tools/list + instructions: what a client that loads every tool pays per request",
    "ja": "変更前：tools/list＋指示文：全ツールを毎回読むクライアントの1リクエストあたり"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/surface-tokens.json",
    "apps/studio/.wrangler/efficiency/surface-before.json",
    "apps/studio/.wrangler/efficiency/codex-dump-before.json"
   ]
  },
  {
   "id": "surface.after.loadAllPerRequest",
   "value": 39878,
   "unit": "tokens",
   "what": {
    "en": "after: tools/list + instructions: what a client that loads every tool pays per request",
    "ja": "軽量化後：tools/list＋指示文：全ツールを毎回読むクライアントの1リクエストあたり"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/surface-tokens.json",
    "apps/studio/.wrangler/efficiency/surface-after.json",
    "apps/studio/.wrangler/efficiency/codex-dump-after.json"
   ]
  },
  {
   "id": "surface.before.codexDump",
   "value": 161345,
   "unit": "tokens",
   "what": {
    "en": "before: all Laydyne entries of Codex's ALL_TOOLS (instructions + description + TS declaration), tokens",
    "ja": "変更前：CodexのALL_TOOLSにあるLaydyneの全項目（指示文＋説明＋TS宣言）のトークン数"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/surface-tokens.json",
    "apps/studio/.wrangler/efficiency/surface-before.json",
    "apps/studio/.wrangler/efficiency/codex-dump-before.json"
   ]
  },
  {
   "id": "surface.after.codexDump",
   "value": 36806,
   "unit": "tokens",
   "what": {
    "en": "after: all Laydyne entries of Codex's ALL_TOOLS (instructions + description + TS declaration), tokens",
    "ja": "軽量化後：CodexのALL_TOOLSにあるLaydyneの全項目（指示文＋説明＋TS宣言）のトークン数"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/surface-tokens.json",
    "apps/studio/.wrangler/efficiency/surface-after.json",
    "apps/studio/.wrangler/efficiency/codex-dump-after.json"
   ]
  },
  {
   "id": "surface.before.toolsListChars",
   "value": 308540,
   "unit": "characters",
   "what": {
    "en": "before: tools/list, characters",
    "ja": "変更前：tools/listの文字数"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/surface-tokens.json",
    "apps/studio/.wrangler/efficiency/surface-before.json",
    "apps/studio/.wrangler/efficiency/codex-dump-before.json"
   ]
  },
  {
   "id": "surface.after.toolsListChars",
   "value": 147237,
   "unit": "characters",
   "what": {
    "en": "after: tools/list, characters",
    "ja": "軽量化後：tools/listの文字数"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/surface-tokens.json",
    "apps/studio/.wrangler/efficiency/surface-after.json",
    "apps/studio/.wrangler/efficiency/codex-dump-after.json"
   ]
  },
  {
   "id": "surface.before.instructionsChars",
   "value": 9006,
   "unit": "characters",
   "what": {
    "en": "before: server instructions, characters",
    "ja": "変更前：サーバー指示文の文字数"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/surface-tokens.json",
    "apps/studio/.wrangler/efficiency/surface-before.json",
    "apps/studio/.wrangler/efficiency/codex-dump-before.json"
   ]
  },
  {
   "id": "surface.after.instructionsChars",
   "value": 1192,
   "unit": "characters",
   "what": {
    "en": "after: server instructions, characters",
    "ja": "軽量化後：サーバー指示文の文字数"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/surface-tokens.json",
    "apps/studio/.wrangler/efficiency/surface-after.json",
    "apps/studio/.wrangler/efficiency/codex-dump-after.json"
   ]
  },
  {
   "id": "surface.before.instructionsShareOfCodexDump",
   "value": 81,
   "unit": "%",
   "what": {
    "en": "share of the server instructions in Codex's listing of Laydyne's tools before (67 × 9,008 of 747,543 characters)",
    "ja": "変更前、CodexのLaydyneツール一覧に占めるサーバー指示文の割合（747,543文字中67×9,008）"
   },
   "source": [
    "apps/studio/.wrangler/efficiency/codex-dump-before.json"
   ]
  },
  {
   "id": "attribution.base.catalogChars",
   "value": 78997,
   "unit": "characters",
   "what": {
    "en": "base: tool-definition text the agent printed per run (Codex ALL_TOOLS lookups), mean of 8 arm-B runs",
    "ja": "変更前：エージェントが1回の作業で表示したツール説明（ALL_TOOLSの参照）の文字数、Bの8回平均"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/attribution.jsonl"
   ]
  },
  {
   "id": "attribution.base.catalogShare",
   "value": 34.3,
   "unit": "% (estimate)",
   "what": {
    "en": "base: estimated share of input tokens due to tool-definition lookups being re-sent with later requests (characters/4 × later requests)",
    "ja": "変更前：ツール説明の参照が後続リクエストで再送された分の、入力トークンに占める推定割合（文字数/4×後続の要求回数）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/attribution.jsonl"
   ]
  },
  {
   "id": "attribution.base.firstRequestTokens",
   "value": 21049,
   "unit": "tokens",
   "what": {
    "en": "base: Codex's own context at the first request (system prompt, skills list, permissions), mean",
    "ja": "変更前：最初の要求時点のCodex自身の文脈（指示・skill一覧・権限）の平均"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/attribution.jsonl"
   ]
  },
  {
   "id": "attribution.light.catalogChars",
   "value": 65124,
   "unit": "characters",
   "what": {
    "en": "light: tool-definition text the agent printed per run (Codex ALL_TOOLS lookups), mean of 8 arm-B runs",
    "ja": "軽量化後：エージェントが1回の作業で表示したツール説明（ALL_TOOLSの参照）の文字数、Bの8回平均"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/attribution.jsonl"
   ]
  },
  {
   "id": "attribution.light.catalogShare",
   "value": 27.9,
   "unit": "% (estimate)",
   "what": {
    "en": "light: estimated share of input tokens due to tool-definition lookups being re-sent with later requests (characters/4 × later requests)",
    "ja": "軽量化後：ツール説明の参照が後続リクエストで再送された分の、入力トークンに占める推定割合（文字数/4×後続の要求回数）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/attribution.jsonl"
   ]
  },
  {
   "id": "attribution.light.firstRequestTokens",
   "value": 21101,
   "unit": "tokens",
   "what": {
    "en": "light: Codex's own context at the first request (system prompt, skills list, permissions), mean",
    "ja": "軽量化後：最初の要求時点のCodex自身の文脈（指示・skill一覧・権限）の平均"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/attribution.jsonl"
   ]
  },
  {
   "id": "followup.C3.full.inputTokens",
   "value": 1035482,
   "unit": "tokens",
   "what": {
    "en": "clinic S3: process scenario and run, main: full list, long instructions: input tokens, mean of 2 runs",
    "ja": "診療所S3：工程と実行・main：全文の一覧＋長い指示文：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/full-C3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/full-C3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.C3.full.requests",
   "value": 16.5,
   "unit": "model requests",
   "what": {
    "en": "clinic S3: process scenario and run, main: full list, long instructions: model requests, mean",
    "ja": "診療所S3：工程と実行・main：全文の一覧＋長い指示文：モデルへの要求回数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/full-C3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/full-C3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.C3.full.failedCalls",
   "value": 10,
   "unit": "calls",
   "what": {
    "en": "clinic S3: process scenario and run, main: full list, long instructions: failed tool calls, both runs",
    "ja": "診療所S3：工程と実行・main：全文の一覧＋長い指示文：失敗した呼び出し（2回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/full-C3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/full-C3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.C3.full.describeCalls",
   "value": 0,
   "unit": "calls",
   "what": {
    "en": "clinic S3: process scenario and run, main: full list, long instructions: describe_tool calls, both runs",
    "ja": "診療所S3：工程と実行・main：全文の一覧＋長い指示文：describe_toolの呼び出し（2回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/full-C3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/full-C3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.C3.full.outcome",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "clinic S3: process scenario and run, main: full list, long instructions: runs reaching the outcome (scripted)",
    "ja": "診療所S3：工程と実行・main：全文の一覧＋長い指示文：目的を達成した回数（採点スクリプト）"
   },
   "source": [
    "apps/studio/.wrangler/efficiency/runs/full-C3-B-r1/score.json",
    "apps/studio/.wrangler/efficiency/runs/full-C3-B-r2/score.json"
   ]
  },
  {
   "id": "followup.C3.safe.inputTokens",
   "value": 1472178,
   "unit": "tokens",
   "what": {
    "en": "clinic S3: process scenario and run, full list, short instructions: input tokens, mean of 2 runs",
    "ja": "診療所S3：工程と実行・全文の一覧＋短い指示文：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/safe-C3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/safe-C3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.C3.safe.requests",
   "value": 18.0,
   "unit": "model requests",
   "what": {
    "en": "clinic S3: process scenario and run, full list, short instructions: model requests, mean",
    "ja": "診療所S3：工程と実行・全文の一覧＋短い指示文：モデルへの要求回数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/safe-C3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/safe-C3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.C3.safe.failedCalls",
   "value": 1,
   "unit": "calls",
   "what": {
    "en": "clinic S3: process scenario and run, full list, short instructions: failed tool calls, both runs",
    "ja": "診療所S3：工程と実行・全文の一覧＋短い指示文：失敗した呼び出し（2回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/safe-C3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/safe-C3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.C3.safe.describeCalls",
   "value": 0,
   "unit": "calls",
   "what": {
    "en": "clinic S3: process scenario and run, full list, short instructions: describe_tool calls, both runs",
    "ja": "診療所S3：工程と実行・全文の一覧＋短い指示文：describe_toolの呼び出し（2回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/safe-C3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/safe-C3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.C3.safe.outcome",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "clinic S3: process scenario and run, full list, short instructions: runs reaching the outcome (scripted)",
    "ja": "診療所S3：工程と実行・全文の一覧＋短い指示文：目的を達成した回数（採点スクリプト）"
   },
   "source": [
    "apps/studio/.wrangler/efficiency/runs/safe-C3-B-r1/score.json",
    "apps/studio/.wrangler/efficiency/runs/safe-C3-B-r2/score.json"
   ]
  },
  {
   "id": "followup.C3.e2.inputTokens",
   "value": 1746510,
   "unit": "tokens",
   "what": {
    "en": "clinic S3: process scenario and run, lighter MCP (E2): input tokens, mean of 2 runs",
    "ja": "診療所S3：工程と実行・軽量化（E2）：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/e2-C3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/e2-C3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.C3.e2.requests",
   "value": 20.0,
   "unit": "model requests",
   "what": {
    "en": "clinic S3: process scenario and run, lighter MCP (E2): model requests, mean",
    "ja": "診療所S3：工程と実行・軽量化（E2）：モデルへの要求回数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/e2-C3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/e2-C3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.C3.e2.failedCalls",
   "value": 2,
   "unit": "calls",
   "what": {
    "en": "clinic S3: process scenario and run, lighter MCP (E2): failed tool calls, both runs",
    "ja": "診療所S3：工程と実行・軽量化（E2）：失敗した呼び出し（2回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/e2-C3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/e2-C3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.C3.e2.describeCalls",
   "value": 9,
   "unit": "calls",
   "what": {
    "en": "clinic S3: process scenario and run, lighter MCP (E2): describe_tool calls, both runs",
    "ja": "診療所S3：工程と実行・軽量化（E2）：describe_toolの呼び出し（2回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/e2-C3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/e2-C3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.C3.e2.outcome",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "clinic S3: process scenario and run, lighter MCP (E2): runs reaching the outcome (scripted)",
    "ja": "診療所S3：工程と実行・軽量化（E2）：目的を達成した回数（採点スクリプト）"
   },
   "source": [
    "apps/studio/.wrangler/efficiency/runs/e2-C3-B-r1/score.json",
    "apps/studio/.wrangler/efficiency/runs/e2-C3-B-r2/score.json"
   ]
  },
  {
   "id": "followup.C3.e2b.inputTokens",
   "value": 1480176,
   "unit": "tokens",
   "what": {
    "en": "clinic S3: process scenario and run, adjusted lighter MCP (E2b): input tokens, mean of 2 runs",
    "ja": "診療所S3：工程と実行・調整版（E2b）：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/e2b-C3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/e2b-C3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.C3.e2b.requests",
   "value": 18.5,
   "unit": "model requests",
   "what": {
    "en": "clinic S3: process scenario and run, adjusted lighter MCP (E2b): model requests, mean",
    "ja": "診療所S3：工程と実行・調整版（E2b）：モデルへの要求回数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/e2b-C3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/e2b-C3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.C3.e2b.failedCalls",
   "value": 1,
   "unit": "calls",
   "what": {
    "en": "clinic S3: process scenario and run, adjusted lighter MCP (E2b): failed tool calls, both runs",
    "ja": "診療所S3：工程と実行・調整版（E2b）：失敗した呼び出し（2回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/e2b-C3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/e2b-C3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.C3.e2b.describeCalls",
   "value": 8,
   "unit": "calls",
   "what": {
    "en": "clinic S3: process scenario and run, adjusted lighter MCP (E2b): describe_tool calls, both runs",
    "ja": "診療所S3：工程と実行・調整版（E2b）：describe_toolの呼び出し（2回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/e2b-C3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/e2b-C3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.C3.e2b.outcome",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "clinic S3: process scenario and run, adjusted lighter MCP (E2b): runs reaching the outcome (scripted)",
    "ja": "診療所S3：工程と実行・調整版（E2b）：目的を達成した回数（採点スクリプト）"
   },
   "source": [
    "apps/studio/.wrangler/efficiency/runs/e2b-C3-B-r1/score.json",
    "apps/studio/.wrangler/efficiency/runs/e2b-C3-B-r2/score.json"
   ]
  },
  {
   "id": "followup.C4.full.inputTokens",
   "value": 833500,
   "unit": "tokens",
   "what": {
    "en": "clinic S4: option B and compare, main: full list, long instructions: input tokens, mean of 2 runs",
    "ja": "診療所S4：案Bと比較・main：全文の一覧＋長い指示文：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/full-C4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/full-C4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.C4.full.requests",
   "value": 14.0,
   "unit": "model requests",
   "what": {
    "en": "clinic S4: option B and compare, main: full list, long instructions: model requests, mean",
    "ja": "診療所S4：案Bと比較・main：全文の一覧＋長い指示文：モデルへの要求回数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/full-C4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/full-C4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.C4.full.failedCalls",
   "value": 0,
   "unit": "calls",
   "what": {
    "en": "clinic S4: option B and compare, main: full list, long instructions: failed tool calls, both runs",
    "ja": "診療所S4：案Bと比較・main：全文の一覧＋長い指示文：失敗した呼び出し（2回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/full-C4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/full-C4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.C4.full.describeCalls",
   "value": 0,
   "unit": "calls",
   "what": {
    "en": "clinic S4: option B and compare, main: full list, long instructions: describe_tool calls, both runs",
    "ja": "診療所S4：案Bと比較・main：全文の一覧＋長い指示文：describe_toolの呼び出し（2回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/full-C4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/full-C4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.C4.full.outcome",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "clinic S4: option B and compare, main: full list, long instructions: runs reaching the outcome (scripted)",
    "ja": "診療所S4：案Bと比較・main：全文の一覧＋長い指示文：目的を達成した回数（採点スクリプト）"
   },
   "source": [
    "apps/studio/.wrangler/efficiency/runs/full-C4-B-r1/score.json",
    "apps/studio/.wrangler/efficiency/runs/full-C4-B-r2/score.json"
   ]
  },
  {
   "id": "followup.C4.safe.inputTokens",
   "value": 889155,
   "unit": "tokens",
   "what": {
    "en": "clinic S4: option B and compare, full list, short instructions: input tokens, mean of 2 runs",
    "ja": "診療所S4：案Bと比較・全文の一覧＋短い指示文：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/safe-C4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/safe-C4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.C4.safe.requests",
   "value": 14.0,
   "unit": "model requests",
   "what": {
    "en": "clinic S4: option B and compare, full list, short instructions: model requests, mean",
    "ja": "診療所S4：案Bと比較・全文の一覧＋短い指示文：モデルへの要求回数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/safe-C4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/safe-C4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.C4.safe.failedCalls",
   "value": 0,
   "unit": "calls",
   "what": {
    "en": "clinic S4: option B and compare, full list, short instructions: failed tool calls, both runs",
    "ja": "診療所S4：案Bと比較・全文の一覧＋短い指示文：失敗した呼び出し（2回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/safe-C4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/safe-C4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.C4.safe.describeCalls",
   "value": 0,
   "unit": "calls",
   "what": {
    "en": "clinic S4: option B and compare, full list, short instructions: describe_tool calls, both runs",
    "ja": "診療所S4：案Bと比較・全文の一覧＋短い指示文：describe_toolの呼び出し（2回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/safe-C4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/safe-C4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.C4.safe.outcome",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "clinic S4: option B and compare, full list, short instructions: runs reaching the outcome (scripted)",
    "ja": "診療所S4：案Bと比較・全文の一覧＋短い指示文：目的を達成した回数（採点スクリプト）"
   },
   "source": [
    "apps/studio/.wrangler/efficiency/runs/safe-C4-B-r1/score.json",
    "apps/studio/.wrangler/efficiency/runs/safe-C4-B-r2/score.json"
   ]
  },
  {
   "id": "followup.C4.e2.inputTokens",
   "value": 1214427,
   "unit": "tokens",
   "what": {
    "en": "clinic S4: option B and compare, lighter MCP (E2): input tokens, mean of 2 runs",
    "ja": "診療所S4：案Bと比較・軽量化（E2）：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/e2-C4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/e2-C4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.C4.e2.requests",
   "value": 16.0,
   "unit": "model requests",
   "what": {
    "en": "clinic S4: option B and compare, lighter MCP (E2): model requests, mean",
    "ja": "診療所S4：案Bと比較・軽量化（E2）：モデルへの要求回数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/e2-C4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/e2-C4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.C4.e2.failedCalls",
   "value": 1,
   "unit": "calls",
   "what": {
    "en": "clinic S4: option B and compare, lighter MCP (E2): failed tool calls, both runs",
    "ja": "診療所S4：案Bと比較・軽量化（E2）：失敗した呼び出し（2回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/e2-C4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/e2-C4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.C4.e2.describeCalls",
   "value": 8,
   "unit": "calls",
   "what": {
    "en": "clinic S4: option B and compare, lighter MCP (E2): describe_tool calls, both runs",
    "ja": "診療所S4：案Bと比較・軽量化（E2）：describe_toolの呼び出し（2回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/e2-C4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/e2-C4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.C4.e2.outcome",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "clinic S4: option B and compare, lighter MCP (E2): runs reaching the outcome (scripted)",
    "ja": "診療所S4：案Bと比較・軽量化（E2）：目的を達成した回数（採点スクリプト）"
   },
   "source": [
    "apps/studio/.wrangler/efficiency/runs/e2-C4-B-r1/score.json",
    "apps/studio/.wrangler/efficiency/runs/e2-C4-B-r2/score.json"
   ]
  },
  {
   "id": "followup.C4.e2b.inputTokens",
   "value": 1353778,
   "unit": "tokens",
   "what": {
    "en": "clinic S4: option B and compare, adjusted lighter MCP (E2b): input tokens, mean of 2 runs",
    "ja": "診療所S4：案Bと比較・調整版（E2b）：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/e2b-C4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/e2b-C4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.C4.e2b.requests",
   "value": 17.5,
   "unit": "model requests",
   "what": {
    "en": "clinic S4: option B and compare, adjusted lighter MCP (E2b): model requests, mean",
    "ja": "診療所S4：案Bと比較・調整版（E2b）：モデルへの要求回数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/e2b-C4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/e2b-C4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.C4.e2b.failedCalls",
   "value": 2,
   "unit": "calls",
   "what": {
    "en": "clinic S4: option B and compare, adjusted lighter MCP (E2b): failed tool calls, both runs",
    "ja": "診療所S4：案Bと比較・調整版（E2b）：失敗した呼び出し（2回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/e2b-C4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/e2b-C4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.C4.e2b.describeCalls",
   "value": 11,
   "unit": "calls",
   "what": {
    "en": "clinic S4: option B and compare, adjusted lighter MCP (E2b): describe_tool calls, both runs",
    "ja": "診療所S4：案Bと比較・調整版（E2b）：describe_toolの呼び出し（2回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/e2b-C4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/e2b-C4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.C4.e2b.outcome",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "clinic S4: option B and compare, adjusted lighter MCP (E2b): runs reaching the outcome (scripted)",
    "ja": "診療所S4：案Bと比較・調整版（E2b）：目的を達成した回数（採点スクリプト）"
   },
   "source": [
    "apps/studio/.wrangler/efficiency/runs/e2b-C4-B-r1/score.json",
    "apps/studio/.wrangler/efficiency/runs/e2b-C4-B-r2/score.json"
   ]
  },
  {
   "id": "followup.MF.full.inputTokens",
   "value": 760124,
   "unit": "tokens",
   "what": {
    "en": "2-storey office with stair and lift, main: full list, long instructions: input tokens, mean of 2 runs",
    "ja": "2階建て事務所（階段と昇降機）・main：全文の一覧＋長い指示文：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/full-MF-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/full-MF-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.MF.full.requests",
   "value": 15.0,
   "unit": "model requests",
   "what": {
    "en": "2-storey office with stair and lift, main: full list, long instructions: model requests, mean",
    "ja": "2階建て事務所（階段と昇降機）・main：全文の一覧＋長い指示文：モデルへの要求回数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/full-MF-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/full-MF-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.MF.full.failedCalls",
   "value": 4,
   "unit": "calls",
   "what": {
    "en": "2-storey office with stair and lift, main: full list, long instructions: failed tool calls, both runs",
    "ja": "2階建て事務所（階段と昇降機）・main：全文の一覧＋長い指示文：失敗した呼び出し（2回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/full-MF-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/full-MF-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.MF.full.describeCalls",
   "value": 0,
   "unit": "calls",
   "what": {
    "en": "2-storey office with stair and lift, main: full list, long instructions: describe_tool calls, both runs",
    "ja": "2階建て事務所（階段と昇降機）・main：全文の一覧＋長い指示文：describe_toolの呼び出し（2回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/full-MF-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/full-MF-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.MF.full.outcome",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "2-storey office with stair and lift, main: full list, long instructions: runs reaching the outcome (scripted)",
    "ja": "2階建て事務所（階段と昇降機）・main：全文の一覧＋長い指示文：目的を達成した回数（採点スクリプト）"
   },
   "source": [
    "apps/studio/.wrangler/efficiency/runs/full-MF-B-r1/score.json",
    "apps/studio/.wrangler/efficiency/runs/full-MF-B-r2/score.json"
   ]
  },
  {
   "id": "followup.MF.safe.inputTokens",
   "value": 889900,
   "unit": "tokens",
   "what": {
    "en": "2-storey office with stair and lift, full list, short instructions: input tokens, mean of 2 runs",
    "ja": "2階建て事務所（階段と昇降機）・全文の一覧＋短い指示文：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/safe-MF-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/safe-MF-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.MF.safe.requests",
   "value": 18.5,
   "unit": "model requests",
   "what": {
    "en": "2-storey office with stair and lift, full list, short instructions: model requests, mean",
    "ja": "2階建て事務所（階段と昇降機）・全文の一覧＋短い指示文：モデルへの要求回数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/safe-MF-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/safe-MF-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.MF.safe.failedCalls",
   "value": 8,
   "unit": "calls",
   "what": {
    "en": "2-storey office with stair and lift, full list, short instructions: failed tool calls, both runs",
    "ja": "2階建て事務所（階段と昇降機）・全文の一覧＋短い指示文：失敗した呼び出し（2回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/safe-MF-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/safe-MF-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.MF.safe.describeCalls",
   "value": 0,
   "unit": "calls",
   "what": {
    "en": "2-storey office with stair and lift, full list, short instructions: describe_tool calls, both runs",
    "ja": "2階建て事務所（階段と昇降機）・全文の一覧＋短い指示文：describe_toolの呼び出し（2回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/safe-MF-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/safe-MF-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.MF.safe.outcome",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "2-storey office with stair and lift, full list, short instructions: runs reaching the outcome (scripted)",
    "ja": "2階建て事務所（階段と昇降機）・全文の一覧＋短い指示文：目的を達成した回数（採点スクリプト）"
   },
   "source": [
    "apps/studio/.wrangler/efficiency/runs/safe-MF-B-r1/score.json",
    "apps/studio/.wrangler/efficiency/runs/safe-MF-B-r2/score.json"
   ]
  },
  {
   "id": "followup.MF.e2.inputTokens",
   "value": 809918,
   "unit": "tokens",
   "what": {
    "en": "2-storey office with stair and lift, lighter MCP (E2): input tokens, mean of 2 runs",
    "ja": "2階建て事務所（階段と昇降機）・軽量化（E2）：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/e2-MF-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/e2-MF-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.MF.e2.requests",
   "value": 16.0,
   "unit": "model requests",
   "what": {
    "en": "2-storey office with stair and lift, lighter MCP (E2): model requests, mean",
    "ja": "2階建て事務所（階段と昇降機）・軽量化（E2）：モデルへの要求回数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/e2-MF-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/e2-MF-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.MF.e2.failedCalls",
   "value": 3,
   "unit": "calls",
   "what": {
    "en": "2-storey office with stair and lift, lighter MCP (E2): failed tool calls, both runs",
    "ja": "2階建て事務所（階段と昇降機）・軽量化（E2）：失敗した呼び出し（2回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/e2-MF-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/e2-MF-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.MF.e2.describeCalls",
   "value": 4,
   "unit": "calls",
   "what": {
    "en": "2-storey office with stair and lift, lighter MCP (E2): describe_tool calls, both runs",
    "ja": "2階建て事務所（階段と昇降機）・軽量化（E2）：describe_toolの呼び出し（2回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/e2-MF-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/e2-MF-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.MF.e2.outcome",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "2-storey office with stair and lift, lighter MCP (E2): runs reaching the outcome (scripted)",
    "ja": "2階建て事務所（階段と昇降機）・軽量化（E2）：目的を達成した回数（採点スクリプト）"
   },
   "source": [
    "apps/studio/.wrangler/efficiency/runs/e2-MF-B-r1/score.json",
    "apps/studio/.wrangler/efficiency/runs/e2-MF-B-r2/score.json"
   ]
  },
  {
   "id": "followup.MF.e2b.inputTokens",
   "value": 942110,
   "unit": "tokens",
   "what": {
    "en": "2-storey office with stair and lift, adjusted lighter MCP (E2b): input tokens, mean of 2 runs",
    "ja": "2階建て事務所（階段と昇降機）・調整版（E2b）：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/e2b-MF-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/e2b-MF-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.MF.e2b.requests",
   "value": 17.5,
   "unit": "model requests",
   "what": {
    "en": "2-storey office with stair and lift, adjusted lighter MCP (E2b): model requests, mean",
    "ja": "2階建て事務所（階段と昇降機）・調整版（E2b）：モデルへの要求回数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/e2b-MF-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/e2b-MF-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.MF.e2b.failedCalls",
   "value": 6,
   "unit": "calls",
   "what": {
    "en": "2-storey office with stair and lift, adjusted lighter MCP (E2b): failed tool calls, both runs",
    "ja": "2階建て事務所（階段と昇降機）・調整版（E2b）：失敗した呼び出し（2回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/e2b-MF-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/e2b-MF-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.MF.e2b.describeCalls",
   "value": 2,
   "unit": "calls",
   "what": {
    "en": "2-storey office with stair and lift, adjusted lighter MCP (E2b): describe_tool calls, both runs",
    "ja": "2階建て事務所（階段と昇降機）・調整版（E2b）：describe_toolの呼び出し（2回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json",
    "apps/studio/.wrangler/efficiency/runs/e2b-MF-B-r1/codex.jsonl",
    "apps/studio/.wrangler/efficiency/runs/e2b-MF-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "followup.MF.e2b.outcome",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "2-storey office with stair and lift, adjusted lighter MCP (E2b): runs reaching the outcome (scripted)",
    "ja": "2階建て事務所（階段と昇降機）・調整版（E2b）：目的を達成した回数（採点スクリプト）"
   },
   "source": [
    "apps/studio/.wrangler/efficiency/runs/e2b-MF-B-r1/score.json",
    "apps/studio/.wrangler/efficiency/runs/e2b-MF-B-r2/score.json"
   ]
  },
  {
   "id": "followup.total.full.inputTokens",
   "value": 2629106,
   "unit": "tokens",
   "what": {
    "en": "Sum over C3, C4 and MF of the mean input tokens, main: full list, long instructions",
    "ja": "C3・C4・MFの平均入力トークンの合計：main：全文の一覧＋長い指示文"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json"
   ]
  },
  {
   "id": "followup.total.safe.inputTokens",
   "value": 3251232,
   "unit": "tokens",
   "what": {
    "en": "Sum over C3, C4 and MF of the mean input tokens, full list, short instructions",
    "ja": "C3・C4・MFの平均入力トークンの合計：全文の一覧＋短い指示文"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json"
   ]
  },
  {
   "id": "followup.total.e2.inputTokens",
   "value": 3770855,
   "unit": "tokens",
   "what": {
    "en": "Sum over C3, C4 and MF of the mean input tokens, lighter MCP (E2)",
    "ja": "C3・C4・MFの平均入力トークンの合計：軽量化（E2）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json"
   ]
  },
  {
   "id": "followup.total.e2b.inputTokens",
   "value": 3776065,
   "unit": "tokens",
   "what": {
    "en": "Sum over C3, C4 and MF of the mean input tokens, adjusted lighter MCP (E2b)",
    "ja": "C3・C4・MFの平均入力トークンの合計：調整版（E2b）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/followup-flows.json"
   ]
  },
  {
   "id": "surface.e2b.loadAllPerRequest",
   "value": 52522,
   "unit": "tokens",
   "what": {
    "en": "adjusted lighter MCP (E2b): tools/list + instructions per request for a client that loads every tool (estimate)",
    "ja": "調整版（E2b）：全ツールを毎回読むクライアントの1要求あたり（推定）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/surface-tokens.json"
   ]
  },
  {
   "id": "e4.MF.mfbase.inputTokens",
   "value": 986278,
   "unit": "tokens",
   "what": {
    "en": "MF, 2-storey office, E4 baseline: input tokens, mean of 2 runs",
    "ja": "MF・2階建て事務所、E4の変更前：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/mfbase-MF-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/mfbase-MF-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.MF.mfbase.inputTokensRange",
   "value": [
    861628,
    1110929
   ],
   "unit": "tokens",
   "what": {
    "en": "MF, 2-storey office, E4 baseline: input tokens, lowest and highest run",
    "ja": "MF・2階建て事務所、E4の変更前：入力トークンの最小と最大"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/mfbase-MF-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/mfbase-MF-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.MF.mfbase.cachedInputTokens",
   "value": 899904,
   "unit": "tokens",
   "what": {
    "en": "MF, 2-storey office, E4 baseline: cached input tokens, mean",
    "ja": "MF・2階建て事務所、E4の変更前：うちキャッシュ（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/mfbase-MF-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/mfbase-MF-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.MF.mfbase.uncachedInputTokens",
   "value": 86374,
   "unit": "tokens",
   "what": {
    "en": "MF, 2-storey office, E4 baseline: uncached input tokens, mean",
    "ja": "MF・2階建て事務所、E4の変更前：キャッシュされない入力（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/mfbase-MF-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/mfbase-MF-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.MF.mfbase.outputTokens",
   "value": 3722,
   "unit": "tokens",
   "what": {
    "en": "MF, 2-storey office, E4 baseline: output tokens, mean",
    "ja": "MF・2階建て事務所、E4の変更前：出力トークン（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/mfbase-MF-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/mfbase-MF-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.MF.mfbase.requests",
   "value": 19.5,
   "unit": "model requests",
   "what": {
    "en": "MF, 2-storey office, E4 baseline: model requests, mean",
    "ja": "MF・2階建て事務所、E4の変更前：モデルへの要求回数（平均）"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/mfbase-MF-B-r1/rollout.jsonl",
    "apps/studio/.wrangler/e4/runs/mfbase-MF-B-r2/rollout.jsonl"
   ]
  },
  {
   "id": "e4.MF.mfbase.requestsRange",
   "value": [
    17,
    22
   ],
   "unit": "model requests",
   "what": {
    "en": "MF, 2-storey office, E4 baseline: model requests, lowest and highest run",
    "ja": "MF・2階建て事務所、E4の変更前：要求回数の最小と最大"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/mfbase-MF-B-r1/rollout.jsonl",
    "apps/studio/.wrangler/e4/runs/mfbase-MF-B-r2/rollout.jsonl"
   ]
  },
  {
   "id": "e4.MF.mfbase.seconds",
   "value": 279,
   "unit": "s",
   "what": {
    "en": "MF, 2-storey office, E4 baseline: wall time, mean",
    "ja": "MF・2階建て事務所、E4の変更前：所要時間（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/mfbase-MF-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/mfbase-MF-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.MF.mfbase.passed",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "MF, 2-storey office, E4 baseline: runs passing the scripted checks",
    "ja": "MF・2階建て事務所、E4の変更前：採点スクリプトの合格数"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/mfbase-MF-B-r1/score.json",
    "apps/studio/.wrangler/e4/runs/mfbase-MF-B-r2/score.json"
   ]
  },
  {
   "id": "e4.MF.mfbase.apiCostUsd",
   "value": 1.95,
   "unit": "USD (estimate)",
   "what": {
    "en": "MF, 2-storey office, E4 baseline: API list-price estimate (gpt-6-astra Standard: $10 input, $1 cached, $50 output per 1M); the runs used a ChatGPT sign-in, not API billing",
    "ja": "MF・2階建て事務所、E4の変更前：API定価での推定費用（gpt-6-astra標準：入力$10・キャッシュ$1・出力$50／100万）。実行はChatGPTのサインインでAPI課金ではない"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/mfbase-MF-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/mfbase-MF-B-r2/codex.jsonl",
    "https://developers.openai.com/api/docs/pricing (read 2026-10-06)"
   ]
  },
  {
   "id": "e4.MF.mfbase.resultChars",
   "value": 136546,
   "unit": "characters",
   "what": {
    "en": "MF, 2-storey office, E4 baseline: characters of tool output the model saw per run (exec outputs), mean",
    "ja": "MF・2階建て事務所、E4の変更前：モデルが見た道具の出力の文字数（1回あたり、平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.MF.mfbase.doubledChars",
   "value": 9969,
   "unit": "characters",
   "what": {
    "en": "MF, 2-storey office, E4 baseline: characters of results printed twice (JSON text and structuredContent), mean",
    "ja": "MF・2階建て事務所、E4の変更前：二重に表示された結果の文字数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.MF.mfbase.failedCalls",
   "value": 3,
   "unit": "calls",
   "what": {
    "en": "MF, 2-storey office, E4 baseline: failed tool calls, all runs",
    "ja": "MF・2階建て事務所、E4の変更前：失敗した呼び出し（全回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/mfbase-MF-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/mfbase-MF-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.MF.mfbase.busy",
   "value": 0,
   "unit": "outputs",
   "what": {
    "en": "MF, 2-storey office, E4 baseline: outputs answered busy, all runs",
    "ja": "MF・2階建て事務所、E4の変更前：busyの出力（全回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.MF.mfe4.inputTokens",
   "value": 576901,
   "unit": "tokens",
   "what": {
    "en": "MF, 2-storey office, E4 final: input tokens, mean of 2 runs",
    "ja": "MF・2階建て事務所、E4の最終：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/mfe4-MF-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/mfe4-MF-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.MF.mfe4.inputTokensRange",
   "value": [
    506393,
    647409
   ],
   "unit": "tokens",
   "what": {
    "en": "MF, 2-storey office, E4 final: input tokens, lowest and highest run",
    "ja": "MF・2階建て事務所、E4の最終：入力トークンの最小と最大"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/mfe4-MF-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/mfe4-MF-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.MF.mfe4.cachedInputTokens",
   "value": 532544,
   "unit": "tokens",
   "what": {
    "en": "MF, 2-storey office, E4 final: cached input tokens, mean",
    "ja": "MF・2階建て事務所、E4の最終：うちキャッシュ（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/mfe4-MF-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/mfe4-MF-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.MF.mfe4.uncachedInputTokens",
   "value": 44357,
   "unit": "tokens",
   "what": {
    "en": "MF, 2-storey office, E4 final: uncached input tokens, mean",
    "ja": "MF・2階建て事務所、E4の最終：キャッシュされない入力（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/mfe4-MF-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/mfe4-MF-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.MF.mfe4.outputTokens",
   "value": 3386,
   "unit": "tokens",
   "what": {
    "en": "MF, 2-storey office, E4 final: output tokens, mean",
    "ja": "MF・2階建て事務所、E4の最終：出力トークン（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/mfe4-MF-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/mfe4-MF-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.MF.mfe4.requests",
   "value": 14,
   "unit": "model requests",
   "what": {
    "en": "MF, 2-storey office, E4 final: model requests, mean",
    "ja": "MF・2階建て事務所、E4の最終：モデルへの要求回数（平均）"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/mfe4-MF-B-r1/rollout.jsonl",
    "apps/studio/.wrangler/e4/runs/mfe4-MF-B-r2/rollout.jsonl"
   ]
  },
  {
   "id": "e4.MF.mfe4.requestsRange",
   "value": [
    12,
    16
   ],
   "unit": "model requests",
   "what": {
    "en": "MF, 2-storey office, E4 final: model requests, lowest and highest run",
    "ja": "MF・2階建て事務所、E4の最終：要求回数の最小と最大"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/mfe4-MF-B-r1/rollout.jsonl",
    "apps/studio/.wrangler/e4/runs/mfe4-MF-B-r2/rollout.jsonl"
   ]
  },
  {
   "id": "e4.MF.mfe4.seconds",
   "value": 151,
   "unit": "s",
   "what": {
    "en": "MF, 2-storey office, E4 final: wall time, mean",
    "ja": "MF・2階建て事務所、E4の最終：所要時間（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/mfe4-MF-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/mfe4-MF-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.MF.mfe4.passed",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "MF, 2-storey office, E4 final: runs passing the scripted checks",
    "ja": "MF・2階建て事務所、E4の最終：採点スクリプトの合格数"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/mfe4-MF-B-r1/score.json",
    "apps/studio/.wrangler/e4/runs/mfe4-MF-B-r2/score.json"
   ]
  },
  {
   "id": "e4.MF.mfe4.apiCostUsd",
   "value": 1.145,
   "unit": "USD (estimate)",
   "what": {
    "en": "MF, 2-storey office, E4 final: API list-price estimate (gpt-6-astra Standard: $10 input, $1 cached, $50 output per 1M); the runs used a ChatGPT sign-in, not API billing",
    "ja": "MF・2階建て事務所、E4の最終：API定価での推定費用（gpt-6-astra標準：入力$10・キャッシュ$1・出力$50／100万）。実行はChatGPTのサインインでAPI課金ではない"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/mfe4-MF-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/mfe4-MF-B-r2/codex.jsonl",
    "https://developers.openai.com/api/docs/pricing (read 2026-10-06)"
   ]
  },
  {
   "id": "e4.MF.mfe4.resultChars",
   "value": 97056,
   "unit": "characters",
   "what": {
    "en": "MF, 2-storey office, E4 final: characters of tool output the model saw per run (exec outputs), mean",
    "ja": "MF・2階建て事務所、E4の最終：モデルが見た道具の出力の文字数（1回あたり、平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.MF.mfe4.doubledChars",
   "value": 0,
   "unit": "characters",
   "what": {
    "en": "MF, 2-storey office, E4 final: characters of results printed twice (JSON text and structuredContent), mean",
    "ja": "MF・2階建て事務所、E4の最終：二重に表示された結果の文字数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.MF.mfe4.failedCalls",
   "value": 2,
   "unit": "calls",
   "what": {
    "en": "MF, 2-storey office, E4 final: failed tool calls, all runs",
    "ja": "MF・2階建て事務所、E4の最終：失敗した呼び出し（全回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/mfe4-MF-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/mfe4-MF-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.MF.mfe4.busy",
   "value": 0,
   "unit": "outputs",
   "what": {
    "en": "MF, 2-storey office, E4 final: outputs answered busy, all runs",
    "ja": "MF・2階建て事務所、E4の最終：busyの出力（全回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.MF.mfe4b.inputTokens",
   "value": 574033,
   "unit": "tokens",
   "what": {
    "en": "MF, mfe4b: input tokens, mean of 2 runs",
    "ja": "MF・mfe4b：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/mfe4b-MF-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/mfe4b-MF-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.MF.mfe4b.inputTokensRange",
   "value": [
    542757,
    605309
   ],
   "unit": "tokens",
   "what": {
    "en": "MF, mfe4b: input tokens, lowest and highest run",
    "ja": "MF・mfe4b：入力トークンの最小と最大"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/mfe4b-MF-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/mfe4b-MF-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.MF.mfe4b.cachedInputTokens",
   "value": 529088,
   "unit": "tokens",
   "what": {
    "en": "MF, mfe4b: cached input tokens, mean",
    "ja": "MF・mfe4b：うちキャッシュ（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/mfe4b-MF-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/mfe4b-MF-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.MF.mfe4b.uncachedInputTokens",
   "value": 44945,
   "unit": "tokens",
   "what": {
    "en": "MF, mfe4b: uncached input tokens, mean",
    "ja": "MF・mfe4b：キャッシュされない入力（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/mfe4b-MF-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/mfe4b-MF-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.MF.mfe4b.outputTokens",
   "value": 3126,
   "unit": "tokens",
   "what": {
    "en": "MF, mfe4b: output tokens, mean",
    "ja": "MF・mfe4b：出力トークン（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/mfe4b-MF-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/mfe4b-MF-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.MF.mfe4b.requests",
   "value": 13.5,
   "unit": "model requests",
   "what": {
    "en": "MF, mfe4b: model requests, mean",
    "ja": "MF・mfe4b：モデルへの要求回数（平均）"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/mfe4b-MF-B-r1/rollout.jsonl",
    "apps/studio/.wrangler/e4/runs/mfe4b-MF-B-r2/rollout.jsonl"
   ]
  },
  {
   "id": "e4.MF.mfe4b.requestsRange",
   "value": [
    13,
    14
   ],
   "unit": "model requests",
   "what": {
    "en": "MF, mfe4b: model requests, lowest and highest run",
    "ja": "MF・mfe4b：要求回数の最小と最大"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/mfe4b-MF-B-r1/rollout.jsonl",
    "apps/studio/.wrangler/e4/runs/mfe4b-MF-B-r2/rollout.jsonl"
   ]
  },
  {
   "id": "e4.MF.mfe4b.seconds",
   "value": 142,
   "unit": "s",
   "what": {
    "en": "MF, mfe4b: wall time, mean",
    "ja": "MF・mfe4b：所要時間（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/mfe4b-MF-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/mfe4b-MF-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.MF.mfe4b.passed",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "MF, mfe4b: runs passing the scripted checks",
    "ja": "MF・mfe4b：採点スクリプトの合格数"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/mfe4b-MF-B-r1/score.json",
    "apps/studio/.wrangler/e4/runs/mfe4b-MF-B-r2/score.json"
   ]
  },
  {
   "id": "e4.MF.mfe4b.apiCostUsd",
   "value": 1.135,
   "unit": "USD (estimate)",
   "what": {
    "en": "MF, mfe4b: API list-price estimate (gpt-6-astra Standard: $10 input, $1 cached, $50 output per 1M); the runs used a ChatGPT sign-in, not API billing",
    "ja": "MF・mfe4b：API定価での推定費用（gpt-6-astra標準：入力$10・キャッシュ$1・出力$50／100万）。実行はChatGPTのサインインでAPI課金ではない"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/mfe4b-MF-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/mfe4b-MF-B-r2/codex.jsonl",
    "https://developers.openai.com/api/docs/pricing (read 2026-10-06)"
   ]
  },
  {
   "id": "e4.MF.mfe4b.resultChars",
   "value": 104980,
   "unit": "characters",
   "what": {
    "en": "MF, mfe4b: characters of tool output the model saw per run (exec outputs), mean",
    "ja": "MF・mfe4b：モデルが見た道具の出力の文字数（1回あたり、平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.MF.mfe4b.doubledChars",
   "value": 0,
   "unit": "characters",
   "what": {
    "en": "MF, mfe4b: characters of results printed twice (JSON text and structuredContent), mean",
    "ja": "MF・mfe4b：二重に表示された結果の文字数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.MF.mfe4b.failedCalls",
   "value": 1,
   "unit": "calls",
   "what": {
    "en": "MF, mfe4b: failed tool calls, all runs",
    "ja": "MF・mfe4b：失敗した呼び出し（全回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/mfe4b-MF-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/mfe4b-MF-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.MF.mfe4b.busy",
   "value": 0,
   "unit": "outputs",
   "what": {
    "en": "MF, mfe4b: outputs answered busy, all runs",
    "ja": "MF・mfe4b：busyの出力（全回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4a.inputTokens",
   "value": 342065,
   "unit": "tokens",
   "what": {
    "en": "T1, Codex + Laydyne, E4 first change set: input tokens, mean of 2 runs",
    "ja": "T1・Codex＋Laydyne、E4の1回目の変更：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4a-T1-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4a-T1-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4a.inputTokensRange",
   "value": [
    301192,
    382938
   ],
   "unit": "tokens",
   "what": {
    "en": "T1, Codex + Laydyne, E4 first change set: input tokens, lowest and highest run",
    "ja": "T1・Codex＋Laydyne、E4の1回目の変更：入力トークンの最小と最大"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4a-T1-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4a-T1-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4a.cachedInputTokens",
   "value": 303872,
   "unit": "tokens",
   "what": {
    "en": "T1, Codex + Laydyne, E4 first change set: cached input tokens, mean",
    "ja": "T1・Codex＋Laydyne、E4の1回目の変更：うちキャッシュ（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4a-T1-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4a-T1-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4a.uncachedInputTokens",
   "value": 38193,
   "unit": "tokens",
   "what": {
    "en": "T1, Codex + Laydyne, E4 first change set: uncached input tokens, mean",
    "ja": "T1・Codex＋Laydyne、E4の1回目の変更：キャッシュされない入力（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4a-T1-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4a-T1-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4a.outputTokens",
   "value": 1758,
   "unit": "tokens",
   "what": {
    "en": "T1, Codex + Laydyne, E4 first change set: output tokens, mean",
    "ja": "T1・Codex＋Laydyne、E4の1回目の変更：出力トークン（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4a-T1-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4a-T1-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4a.requests",
   "value": 9,
   "unit": "model requests",
   "what": {
    "en": "T1, Codex + Laydyne, E4 first change set: model requests, mean",
    "ja": "T1・Codex＋Laydyne、E4の1回目の変更：モデルへの要求回数（平均）"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4a-T1-B-r1/rollout.jsonl",
    "apps/studio/.wrangler/e4/runs/e4a-T1-B-r2/rollout.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4a.requestsRange",
   "value": [
    8,
    10
   ],
   "unit": "model requests",
   "what": {
    "en": "T1, Codex + Laydyne, E4 first change set: model requests, lowest and highest run",
    "ja": "T1・Codex＋Laydyne、E4の1回目の変更：要求回数の最小と最大"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4a-T1-B-r1/rollout.jsonl",
    "apps/studio/.wrangler/e4/runs/e4a-T1-B-r2/rollout.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4a.seconds",
   "value": 136,
   "unit": "s",
   "what": {
    "en": "T1, Codex + Laydyne, E4 first change set: wall time, mean",
    "ja": "T1・Codex＋Laydyne、E4の1回目の変更：所要時間（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4a-T1-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4a-T1-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4a.passed",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "T1, Codex + Laydyne, E4 first change set: runs passing the scripted checks",
    "ja": "T1・Codex＋Laydyne、E4の1回目の変更：採点スクリプトの合格数"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4a-T1-B-r1/score.json",
    "apps/studio/.wrangler/e4/runs/e4a-T1-B-r2/score.json"
   ]
  },
  {
   "id": "e4.T1.e4a.apiCostUsd",
   "value": 0.773,
   "unit": "USD (estimate)",
   "what": {
    "en": "T1, Codex + Laydyne, E4 first change set: API list-price estimate (gpt-6-astra Standard: $10 input, $1 cached, $50 output per 1M); the runs used a ChatGPT sign-in, not API billing",
    "ja": "T1・Codex＋Laydyne、E4の1回目の変更：API定価での推定費用（gpt-6-astra標準：入力$10・キャッシュ$1・出力$50／100万）。実行はChatGPTのサインインでAPI課金ではない"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4a-T1-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4a-T1-B-r2/codex.jsonl",
    "https://developers.openai.com/api/docs/pricing (read 2026-10-06)"
   ]
  },
  {
   "id": "e4.T1.e4a.resultChars",
   "value": 89550,
   "unit": "characters",
   "what": {
    "en": "T1, Codex + Laydyne, E4 first change set: characters of tool output the model saw per run (exec outputs), mean",
    "ja": "T1・Codex＋Laydyne、E4の1回目の変更：モデルが見た道具の出力の文字数（1回あたり、平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4a.doubledChars",
   "value": 0,
   "unit": "characters",
   "what": {
    "en": "T1, Codex + Laydyne, E4 first change set: characters of results printed twice (JSON text and structuredContent), mean",
    "ja": "T1・Codex＋Laydyne、E4の1回目の変更：二重に表示された結果の文字数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4a.failedCalls",
   "value": 0,
   "unit": "calls",
   "what": {
    "en": "T1, Codex + Laydyne, E4 first change set: failed tool calls, all runs",
    "ja": "T1・Codex＋Laydyne、E4の1回目の変更：失敗した呼び出し（全回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4a-T1-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4a-T1-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4a.busy",
   "value": 0,
   "unit": "outputs",
   "what": {
    "en": "T1, Codex + Laydyne, E4 first change set: outputs answered busy, all runs",
    "ja": "T1・Codex＋Laydyne、E4の1回目の変更：busyの出力（全回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4alone.inputTokens",
   "value": 135326,
   "unit": "tokens",
   "what": {
    "en": "T1, e4alone: input tokens, mean of 2 runs",
    "ja": "T1・e4alone：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4alone-T1-A-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4alone-T1-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4alone.inputTokensRange",
   "value": [
    127779,
    142874
   ],
   "unit": "tokens",
   "what": {
    "en": "T1, e4alone: input tokens, lowest and highest run",
    "ja": "T1・e4alone：入力トークンの最小と最大"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4alone-T1-A-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4alone-T1-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4alone.cachedInputTokens",
   "value": 110912,
   "unit": "tokens",
   "what": {
    "en": "T1, e4alone: cached input tokens, mean",
    "ja": "T1・e4alone：うちキャッシュ（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4alone-T1-A-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4alone-T1-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4alone.uncachedInputTokens",
   "value": 24414,
   "unit": "tokens",
   "what": {
    "en": "T1, e4alone: uncached input tokens, mean",
    "ja": "T1・e4alone：キャッシュされない入力（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4alone-T1-A-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4alone-T1-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4alone.outputTokens",
   "value": 4896,
   "unit": "tokens",
   "what": {
    "en": "T1, e4alone: output tokens, mean",
    "ja": "T1・e4alone：出力トークン（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4alone-T1-A-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4alone-T1-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4alone.requests",
   "value": 5.5,
   "unit": "model requests",
   "what": {
    "en": "T1, e4alone: model requests, mean",
    "ja": "T1・e4alone：モデルへの要求回数（平均）"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4alone-T1-A-r1/rollout.jsonl",
    "apps/studio/.wrangler/e4/runs/e4alone-T1-A-r2/rollout.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4alone.requestsRange",
   "value": [
    5,
    6
   ],
   "unit": "model requests",
   "what": {
    "en": "T1, e4alone: model requests, lowest and highest run",
    "ja": "T1・e4alone：要求回数の最小と最大"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4alone-T1-A-r1/rollout.jsonl",
    "apps/studio/.wrangler/e4/runs/e4alone-T1-A-r2/rollout.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4alone.seconds",
   "value": 136,
   "unit": "s",
   "what": {
    "en": "T1, e4alone: wall time, mean",
    "ja": "T1・e4alone：所要時間（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4alone-T1-A-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4alone-T1-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4alone.passed",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "T1, e4alone: runs passing the scripted checks",
    "ja": "T1・e4alone：採点スクリプトの合格数"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4alone-T1-A-r1/score.json",
    "apps/studio/.wrangler/e4/runs/e4alone-T1-A-r2/score.json"
   ]
  },
  {
   "id": "e4.T1.e4alone.apiCostUsd",
   "value": 0.6,
   "unit": "USD (estimate)",
   "what": {
    "en": "T1, e4alone: API list-price estimate (gpt-6-astra Standard: $10 input, $1 cached, $50 output per 1M); the runs used a ChatGPT sign-in, not API billing",
    "ja": "T1・e4alone：API定価での推定費用（gpt-6-astra標準：入力$10・キャッシュ$1・出力$50／100万）。実行はChatGPTのサインインでAPI課金ではない"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4alone-T1-A-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4alone-T1-A-r2/codex.jsonl",
    "https://developers.openai.com/api/docs/pricing (read 2026-10-06)"
   ]
  },
  {
   "id": "e4.T1.e4alone.resultChars",
   "value": 10223,
   "unit": "characters",
   "what": {
    "en": "T1, e4alone: characters of tool output the model saw per run (exec outputs), mean",
    "ja": "T1・e4alone：モデルが見た道具の出力の文字数（1回あたり、平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4alone.doubledChars",
   "value": 0,
   "unit": "characters",
   "what": {
    "en": "T1, e4alone: characters of results printed twice (JSON text and structuredContent), mean",
    "ja": "T1・e4alone：二重に表示された結果の文字数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4alone.failedCalls",
   "value": 0,
   "unit": "calls",
   "what": {
    "en": "T1, e4alone: failed tool calls, all runs",
    "ja": "T1・e4alone：失敗した呼び出し（全回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4alone-T1-A-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4alone-T1-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4alone.busy",
   "value": 0,
   "unit": "outputs",
   "what": {
    "en": "T1, e4alone: outputs answered busy, all runs",
    "ja": "T1・e4alone：busyの出力（全回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4b.inputTokens",
   "value": 244564,
   "unit": "tokens",
   "what": {
    "en": "T1, Codex + Laydyne, E4 final: input tokens, mean of 2 runs",
    "ja": "T1・Codex＋Laydyne、E4の最終：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4b-T1-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4b-T1-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4b.inputTokensRange",
   "value": [
    242718,
    246410
   ],
   "unit": "tokens",
   "what": {
    "en": "T1, Codex + Laydyne, E4 final: input tokens, lowest and highest run",
    "ja": "T1・Codex＋Laydyne、E4の最終：入力トークンの最小と最大"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4b-T1-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4b-T1-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4b.cachedInputTokens",
   "value": 212928,
   "unit": "tokens",
   "what": {
    "en": "T1, Codex + Laydyne, E4 final: cached input tokens, mean",
    "ja": "T1・Codex＋Laydyne、E4の最終：うちキャッシュ（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4b-T1-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4b-T1-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4b.uncachedInputTokens",
   "value": 31636,
   "unit": "tokens",
   "what": {
    "en": "T1, Codex + Laydyne, E4 final: uncached input tokens, mean",
    "ja": "T1・Codex＋Laydyne、E4の最終：キャッシュされない入力（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4b-T1-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4b-T1-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4b.outputTokens",
   "value": 1469,
   "unit": "tokens",
   "what": {
    "en": "T1, Codex + Laydyne, E4 final: output tokens, mean",
    "ja": "T1・Codex＋Laydyne、E4の最終：出力トークン（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4b-T1-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4b-T1-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4b.requests",
   "value": 7,
   "unit": "model requests",
   "what": {
    "en": "T1, Codex + Laydyne, E4 final: model requests, mean",
    "ja": "T1・Codex＋Laydyne、E4の最終：モデルへの要求回数（平均）"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4b-T1-B-r1/rollout.jsonl",
    "apps/studio/.wrangler/e4/runs/e4b-T1-B-r2/rollout.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4b.requestsRange",
   "value": [
    7,
    7
   ],
   "unit": "model requests",
   "what": {
    "en": "T1, Codex + Laydyne, E4 final: model requests, lowest and highest run",
    "ja": "T1・Codex＋Laydyne、E4の最終：要求回数の最小と最大"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4b-T1-B-r1/rollout.jsonl",
    "apps/studio/.wrangler/e4/runs/e4b-T1-B-r2/rollout.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4b.seconds",
   "value": 88,
   "unit": "s",
   "what": {
    "en": "T1, Codex + Laydyne, E4 final: wall time, mean",
    "ja": "T1・Codex＋Laydyne、E4の最終：所要時間（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4b-T1-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4b-T1-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4b.passed",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "T1, Codex + Laydyne, E4 final: runs passing the scripted checks",
    "ja": "T1・Codex＋Laydyne、E4の最終：採点スクリプトの合格数"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4b-T1-B-r1/score.json",
    "apps/studio/.wrangler/e4/runs/e4b-T1-B-r2/score.json"
   ]
  },
  {
   "id": "e4.T1.e4b.apiCostUsd",
   "value": 0.603,
   "unit": "USD (estimate)",
   "what": {
    "en": "T1, Codex + Laydyne, E4 final: API list-price estimate (gpt-6-astra Standard: $10 input, $1 cached, $50 output per 1M); the runs used a ChatGPT sign-in, not API billing",
    "ja": "T1・Codex＋Laydyne、E4の最終：API定価での推定費用（gpt-6-astra標準：入力$10・キャッシュ$1・出力$50／100万）。実行はChatGPTのサインインでAPI課金ではない"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4b-T1-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4b-T1-B-r2/codex.jsonl",
    "https://developers.openai.com/api/docs/pricing (read 2026-10-06)"
   ]
  },
  {
   "id": "e4.T1.e4b.resultChars",
   "value": 72563,
   "unit": "characters",
   "what": {
    "en": "T1, Codex + Laydyne, E4 final: characters of tool output the model saw per run (exec outputs), mean",
    "ja": "T1・Codex＋Laydyne、E4の最終：モデルが見た道具の出力の文字数（1回あたり、平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4b.doubledChars",
   "value": 0,
   "unit": "characters",
   "what": {
    "en": "T1, Codex + Laydyne, E4 final: characters of results printed twice (JSON text and structuredContent), mean",
    "ja": "T1・Codex＋Laydyne、E4の最終：二重に表示された結果の文字数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4b.failedCalls",
   "value": 0,
   "unit": "calls",
   "what": {
    "en": "T1, Codex + Laydyne, E4 final: failed tool calls, all runs",
    "ja": "T1・Codex＋Laydyne、E4の最終：失敗した呼び出し（全回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4b-T1-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4b-T1-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4b.busy",
   "value": 0,
   "unit": "outputs",
   "what": {
    "en": "T1, Codex + Laydyne, E4 final: outputs answered busy, all runs",
    "ja": "T1・Codex＋Laydyne、E4の最終：busyの出力（全回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4base.inputTokens",
   "value": 520728,
   "unit": "tokens",
   "what": {
    "en": "T1, Codex + Laydyne, E4 baseline (main bf7bfec7): input tokens, mean of 2 runs",
    "ja": "T1・Codex＋Laydyne、E4の変更前（main bf7bfec7）：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4base-T1-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4base-T1-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4base.inputTokensRange",
   "value": [
    514293,
    527163
   ],
   "unit": "tokens",
   "what": {
    "en": "T1, Codex + Laydyne, E4 baseline (main bf7bfec7): input tokens, lowest and highest run",
    "ja": "T1・Codex＋Laydyne、E4の変更前（main bf7bfec7）：入力トークンの最小と最大"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4base-T1-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4base-T1-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4base.cachedInputTokens",
   "value": 468800,
   "unit": "tokens",
   "what": {
    "en": "T1, Codex + Laydyne, E4 baseline (main bf7bfec7): cached input tokens, mean",
    "ja": "T1・Codex＋Laydyne、E4の変更前（main bf7bfec7）：うちキャッシュ（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4base-T1-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4base-T1-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4base.uncachedInputTokens",
   "value": 51928,
   "unit": "tokens",
   "what": {
    "en": "T1, Codex + Laydyne, E4 baseline (main bf7bfec7): uncached input tokens, mean",
    "ja": "T1・Codex＋Laydyne、E4の変更前（main bf7bfec7）：キャッシュされない入力（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4base-T1-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4base-T1-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4base.outputTokens",
   "value": 2120,
   "unit": "tokens",
   "what": {
    "en": "T1, Codex + Laydyne, E4 baseline (main bf7bfec7): output tokens, mean",
    "ja": "T1・Codex＋Laydyne、E4の変更前（main bf7bfec7）：出力トークン（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4base-T1-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4base-T1-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4base.requests",
   "value": 12,
   "unit": "model requests",
   "what": {
    "en": "T1, Codex + Laydyne, E4 baseline (main bf7bfec7): model requests, mean",
    "ja": "T1・Codex＋Laydyne、E4の変更前（main bf7bfec7）：モデルへの要求回数（平均）"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4base-T1-B-r1/rollout.jsonl",
    "apps/studio/.wrangler/e4/runs/e4base-T1-B-r2/rollout.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4base.requestsRange",
   "value": [
    12,
    12
   ],
   "unit": "model requests",
   "what": {
    "en": "T1, Codex + Laydyne, E4 baseline (main bf7bfec7): model requests, lowest and highest run",
    "ja": "T1・Codex＋Laydyne、E4の変更前（main bf7bfec7）：要求回数の最小と最大"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4base-T1-B-r1/rollout.jsonl",
    "apps/studio/.wrangler/e4/runs/e4base-T1-B-r2/rollout.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4base.seconds",
   "value": 158,
   "unit": "s",
   "what": {
    "en": "T1, Codex + Laydyne, E4 baseline (main bf7bfec7): wall time, mean",
    "ja": "T1・Codex＋Laydyne、E4の変更前（main bf7bfec7）：所要時間（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4base-T1-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4base-T1-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4base.passed",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "T1, Codex + Laydyne, E4 baseline (main bf7bfec7): runs passing the scripted checks",
    "ja": "T1・Codex＋Laydyne、E4の変更前（main bf7bfec7）：採点スクリプトの合格数"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4base-T1-B-r1/score.json",
    "apps/studio/.wrangler/e4/runs/e4base-T1-B-r2/score.json"
   ]
  },
  {
   "id": "e4.T1.e4base.apiCostUsd",
   "value": 1.094,
   "unit": "USD (estimate)",
   "what": {
    "en": "T1, Codex + Laydyne, E4 baseline (main bf7bfec7): API list-price estimate (gpt-6-astra Standard: $10 input, $1 cached, $50 output per 1M); the runs used a ChatGPT sign-in, not API billing",
    "ja": "T1・Codex＋Laydyne、E4の変更前（main bf7bfec7）：API定価での推定費用（gpt-6-astra標準：入力$10・キャッシュ$1・出力$50／100万）。実行はChatGPTのサインインでAPI課金ではない"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4base-T1-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4base-T1-B-r2/codex.jsonl",
    "https://developers.openai.com/api/docs/pricing (read 2026-10-06)"
   ]
  },
  {
   "id": "e4.T1.e4base.resultChars",
   "value": 129122,
   "unit": "characters",
   "what": {
    "en": "T1, Codex + Laydyne, E4 baseline (main bf7bfec7): characters of tool output the model saw per run (exec outputs), mean",
    "ja": "T1・Codex＋Laydyne、E4の変更前（main bf7bfec7）：モデルが見た道具の出力の文字数（1回あたり、平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4base.doubledChars",
   "value": 27408,
   "unit": "characters",
   "what": {
    "en": "T1, Codex + Laydyne, E4 baseline (main bf7bfec7): characters of results printed twice (JSON text and structuredContent), mean",
    "ja": "T1・Codex＋Laydyne、E4の変更前（main bf7bfec7）：二重に表示された結果の文字数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4base.failedCalls",
   "value": 1,
   "unit": "calls",
   "what": {
    "en": "T1, Codex + Laydyne, E4 baseline (main bf7bfec7): failed tool calls, all runs",
    "ja": "T1・Codex＋Laydyne、E4の変更前（main bf7bfec7）：失敗した呼び出し（全回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4base-T1-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4base-T1-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T1.e4base.busy",
   "value": 0,
   "unit": "outputs",
   "what": {
    "en": "T1, Codex + Laydyne, E4 baseline (main bf7bfec7): outputs answered busy, all runs",
    "ja": "T1・Codex＋Laydyne、E4の変更前（main bf7bfec7）：busyの出力（全回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4a.inputTokens",
   "value": 473171,
   "unit": "tokens",
   "what": {
    "en": "T2, Codex + Laydyne, E4 first change set: input tokens, mean of 2 runs",
    "ja": "T2・Codex＋Laydyne、E4の1回目の変更：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4a-T2-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4a-T2-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4a.inputTokensRange",
   "value": [
    422390,
    523952
   ],
   "unit": "tokens",
   "what": {
    "en": "T2, Codex + Laydyne, E4 first change set: input tokens, lowest and highest run",
    "ja": "T2・Codex＋Laydyne、E4の1回目の変更：入力トークンの最小と最大"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4a-T2-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4a-T2-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4a.cachedInputTokens",
   "value": 440640,
   "unit": "tokens",
   "what": {
    "en": "T2, Codex + Laydyne, E4 first change set: cached input tokens, mean",
    "ja": "T2・Codex＋Laydyne、E4の1回目の変更：うちキャッシュ（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4a-T2-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4a-T2-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4a.uncachedInputTokens",
   "value": 32531,
   "unit": "tokens",
   "what": {
    "en": "T2, Codex + Laydyne, E4 first change set: uncached input tokens, mean",
    "ja": "T2・Codex＋Laydyne、E4の1回目の変更：キャッシュされない入力（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4a-T2-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4a-T2-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4a.outputTokens",
   "value": 1214,
   "unit": "tokens",
   "what": {
    "en": "T2, Codex + Laydyne, E4 first change set: output tokens, mean",
    "ja": "T2・Codex＋Laydyne、E4の1回目の変更：出力トークン（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4a-T2-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4a-T2-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4a.requests",
   "value": 12.5,
   "unit": "model requests",
   "what": {
    "en": "T2, Codex + Laydyne, E4 first change set: model requests, mean",
    "ja": "T2・Codex＋Laydyne、E4の1回目の変更：モデルへの要求回数（平均）"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4a-T2-B-r1/rollout.jsonl",
    "apps/studio/.wrangler/e4/runs/e4a-T2-B-r2/rollout.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4a.requestsRange",
   "value": [
    12,
    13
   ],
   "unit": "model requests",
   "what": {
    "en": "T2, Codex + Laydyne, E4 first change set: model requests, lowest and highest run",
    "ja": "T2・Codex＋Laydyne、E4の1回目の変更：要求回数の最小と最大"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4a-T2-B-r1/rollout.jsonl",
    "apps/studio/.wrangler/e4/runs/e4a-T2-B-r2/rollout.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4a.seconds",
   "value": 108,
   "unit": "s",
   "what": {
    "en": "T2, Codex + Laydyne, E4 first change set: wall time, mean",
    "ja": "T2・Codex＋Laydyne、E4の1回目の変更：所要時間（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4a-T2-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4a-T2-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4a.passed",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "T2, Codex + Laydyne, E4 first change set: runs passing the scripted checks",
    "ja": "T2・Codex＋Laydyne、E4の1回目の変更：採点スクリプトの合格数"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4a-T2-B-r1/score.json",
    "apps/studio/.wrangler/e4/runs/e4a-T2-B-r2/score.json"
   ]
  },
  {
   "id": "e4.T2.e4a.apiCostUsd",
   "value": 0.827,
   "unit": "USD (estimate)",
   "what": {
    "en": "T2, Codex + Laydyne, E4 first change set: API list-price estimate (gpt-6-astra Standard: $10 input, $1 cached, $50 output per 1M); the runs used a ChatGPT sign-in, not API billing",
    "ja": "T2・Codex＋Laydyne、E4の1回目の変更：API定価での推定費用（gpt-6-astra標準：入力$10・キャッシュ$1・出力$50／100万）。実行はChatGPTのサインインでAPI課金ではない"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4a-T2-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4a-T2-B-r2/codex.jsonl",
    "https://developers.openai.com/api/docs/pricing (read 2026-10-06)"
   ]
  },
  {
   "id": "e4.T2.e4a.resultChars",
   "value": 84418,
   "unit": "characters",
   "what": {
    "en": "T2, Codex + Laydyne, E4 first change set: characters of tool output the model saw per run (exec outputs), mean",
    "ja": "T2・Codex＋Laydyne、E4の1回目の変更：モデルが見た道具の出力の文字数（1回あたり、平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4a.doubledChars",
   "value": 0,
   "unit": "characters",
   "what": {
    "en": "T2, Codex + Laydyne, E4 first change set: characters of results printed twice (JSON text and structuredContent), mean",
    "ja": "T2・Codex＋Laydyne、E4の1回目の変更：二重に表示された結果の文字数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4a.failedCalls",
   "value": 0,
   "unit": "calls",
   "what": {
    "en": "T2, Codex + Laydyne, E4 first change set: failed tool calls, all runs",
    "ja": "T2・Codex＋Laydyne、E4の1回目の変更：失敗した呼び出し（全回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4a-T2-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4a-T2-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4a.busy",
   "value": 0,
   "unit": "outputs",
   "what": {
    "en": "T2, Codex + Laydyne, E4 first change set: outputs answered busy, all runs",
    "ja": "T2・Codex＋Laydyne、E4の1回目の変更：busyの出力（全回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4alone.inputTokens",
   "value": 117899,
   "unit": "tokens",
   "what": {
    "en": "T2, e4alone: input tokens, mean of 2 runs",
    "ja": "T2・e4alone：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4alone-T2-A-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4alone-T2-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4alone.inputTokensRange",
   "value": [
    104282,
    131516
   ],
   "unit": "tokens",
   "what": {
    "en": "T2, e4alone: input tokens, lowest and highest run",
    "ja": "T2・e4alone：入力トークンの最小と最大"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4alone-T2-A-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4alone-T2-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4alone.cachedInputTokens",
   "value": 95424,
   "unit": "tokens",
   "what": {
    "en": "T2, e4alone: cached input tokens, mean",
    "ja": "T2・e4alone：うちキャッシュ（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4alone-T2-A-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4alone-T2-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4alone.uncachedInputTokens",
   "value": 22475,
   "unit": "tokens",
   "what": {
    "en": "T2, e4alone: uncached input tokens, mean",
    "ja": "T2・e4alone：キャッシュされない入力（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4alone-T2-A-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4alone-T2-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4alone.outputTokens",
   "value": 1718,
   "unit": "tokens",
   "what": {
    "en": "T2, e4alone: output tokens, mean",
    "ja": "T2・e4alone：出力トークン（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4alone-T2-A-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4alone-T2-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4alone.requests",
   "value": 4.5,
   "unit": "model requests",
   "what": {
    "en": "T2, e4alone: model requests, mean",
    "ja": "T2・e4alone：モデルへの要求回数（平均）"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4alone-T2-A-r1/rollout.jsonl",
    "apps/studio/.wrangler/e4/runs/e4alone-T2-A-r2/rollout.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4alone.requestsRange",
   "value": [
    4,
    5
   ],
   "unit": "model requests",
   "what": {
    "en": "T2, e4alone: model requests, lowest and highest run",
    "ja": "T2・e4alone：要求回数の最小と最大"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4alone-T2-A-r1/rollout.jsonl",
    "apps/studio/.wrangler/e4/runs/e4alone-T2-A-r2/rollout.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4alone.seconds",
   "value": 66,
   "unit": "s",
   "what": {
    "en": "T2, e4alone: wall time, mean",
    "ja": "T2・e4alone：所要時間（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4alone-T2-A-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4alone-T2-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4alone.passed",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "T2, e4alone: runs passing the scripted checks",
    "ja": "T2・e4alone：採点スクリプトの合格数"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4alone-T2-A-r1/score.json",
    "apps/studio/.wrangler/e4/runs/e4alone-T2-A-r2/score.json"
   ]
  },
  {
   "id": "e4.T2.e4alone.apiCostUsd",
   "value": 0.406,
   "unit": "USD (estimate)",
   "what": {
    "en": "T2, e4alone: API list-price estimate (gpt-6-astra Standard: $10 input, $1 cached, $50 output per 1M); the runs used a ChatGPT sign-in, not API billing",
    "ja": "T2・e4alone：API定価での推定費用（gpt-6-astra標準：入力$10・キャッシュ$1・出力$50／100万）。実行はChatGPTのサインインでAPI課金ではない"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4alone-T2-A-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4alone-T2-A-r2/codex.jsonl",
    "https://developers.openai.com/api/docs/pricing (read 2026-10-06)"
   ]
  },
  {
   "id": "e4.T2.e4alone.resultChars",
   "value": 23536,
   "unit": "characters",
   "what": {
    "en": "T2, e4alone: characters of tool output the model saw per run (exec outputs), mean",
    "ja": "T2・e4alone：モデルが見た道具の出力の文字数（1回あたり、平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4alone.doubledChars",
   "value": 0,
   "unit": "characters",
   "what": {
    "en": "T2, e4alone: characters of results printed twice (JSON text and structuredContent), mean",
    "ja": "T2・e4alone：二重に表示された結果の文字数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4alone.failedCalls",
   "value": 0,
   "unit": "calls",
   "what": {
    "en": "T2, e4alone: failed tool calls, all runs",
    "ja": "T2・e4alone：失敗した呼び出し（全回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4alone-T2-A-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4alone-T2-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4alone.busy",
   "value": 0,
   "unit": "outputs",
   "what": {
    "en": "T2, e4alone: outputs answered busy, all runs",
    "ja": "T2・e4alone：busyの出力（全回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4b.inputTokens",
   "value": 391216,
   "unit": "tokens",
   "what": {
    "en": "T2, Codex + Laydyne, E4 final: input tokens, mean of 2 runs",
    "ja": "T2・Codex＋Laydyne、E4の最終：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4b-T2-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4b-T2-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4b.inputTokensRange",
   "value": [
    344092,
    438341
   ],
   "unit": "tokens",
   "what": {
    "en": "T2, Codex + Laydyne, E4 final: input tokens, lowest and highest run",
    "ja": "T2・Codex＋Laydyne、E4の最終：入力トークンの最小と最大"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4b-T2-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4b-T2-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4b.cachedInputTokens",
   "value": 358144,
   "unit": "tokens",
   "what": {
    "en": "T2, Codex + Laydyne, E4 final: cached input tokens, mean",
    "ja": "T2・Codex＋Laydyne、E4の最終：うちキャッシュ（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4b-T2-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4b-T2-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4b.uncachedInputTokens",
   "value": 33072,
   "unit": "tokens",
   "what": {
    "en": "T2, Codex + Laydyne, E4 final: uncached input tokens, mean",
    "ja": "T2・Codex＋Laydyne、E4の最終：キャッシュされない入力（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4b-T2-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4b-T2-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4b.outputTokens",
   "value": 1190,
   "unit": "tokens",
   "what": {
    "en": "T2, Codex + Laydyne, E4 final: output tokens, mean",
    "ja": "T2・Codex＋Laydyne、E4の最終：出力トークン（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4b-T2-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4b-T2-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4b.requests",
   "value": 11,
   "unit": "model requests",
   "what": {
    "en": "T2, Codex + Laydyne, E4 final: model requests, mean",
    "ja": "T2・Codex＋Laydyne、E4の最終：モデルへの要求回数（平均）"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4b-T2-B-r1/rollout.jsonl",
    "apps/studio/.wrangler/e4/runs/e4b-T2-B-r2/rollout.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4b.requestsRange",
   "value": [
    10,
    12
   ],
   "unit": "model requests",
   "what": {
    "en": "T2, Codex + Laydyne, E4 final: model requests, lowest and highest run",
    "ja": "T2・Codex＋Laydyne、E4の最終：要求回数の最小と最大"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4b-T2-B-r1/rollout.jsonl",
    "apps/studio/.wrangler/e4/runs/e4b-T2-B-r2/rollout.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4b.seconds",
   "value": 70,
   "unit": "s",
   "what": {
    "en": "T2, Codex + Laydyne, E4 final: wall time, mean",
    "ja": "T2・Codex＋Laydyne、E4の最終：所要時間（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4b-T2-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4b-T2-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4b.passed",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "T2, Codex + Laydyne, E4 final: runs passing the scripted checks",
    "ja": "T2・Codex＋Laydyne、E4の最終：採点スクリプトの合格数"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4b-T2-B-r1/score.json",
    "apps/studio/.wrangler/e4/runs/e4b-T2-B-r2/score.json"
   ]
  },
  {
   "id": "e4.T2.e4b.apiCostUsd",
   "value": 0.748,
   "unit": "USD (estimate)",
   "what": {
    "en": "T2, Codex + Laydyne, E4 final: API list-price estimate (gpt-6-astra Standard: $10 input, $1 cached, $50 output per 1M); the runs used a ChatGPT sign-in, not API billing",
    "ja": "T2・Codex＋Laydyne、E4の最終：API定価での推定費用（gpt-6-astra標準：入力$10・キャッシュ$1・出力$50／100万）。実行はChatGPTのサインインでAPI課金ではない"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4b-T2-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4b-T2-B-r2/codex.jsonl",
    "https://developers.openai.com/api/docs/pricing (read 2026-10-06)"
   ]
  },
  {
   "id": "e4.T2.e4b.resultChars",
   "value": 79402,
   "unit": "characters",
   "what": {
    "en": "T2, Codex + Laydyne, E4 final: characters of tool output the model saw per run (exec outputs), mean",
    "ja": "T2・Codex＋Laydyne、E4の最終：モデルが見た道具の出力の文字数（1回あたり、平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4b.doubledChars",
   "value": 0,
   "unit": "characters",
   "what": {
    "en": "T2, Codex + Laydyne, E4 final: characters of results printed twice (JSON text and structuredContent), mean",
    "ja": "T2・Codex＋Laydyne、E4の最終：二重に表示された結果の文字数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4b.failedCalls",
   "value": 0,
   "unit": "calls",
   "what": {
    "en": "T2, Codex + Laydyne, E4 final: failed tool calls, all runs",
    "ja": "T2・Codex＋Laydyne、E4の最終：失敗した呼び出し（全回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4b-T2-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4b-T2-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4b.busy",
   "value": 0,
   "unit": "outputs",
   "what": {
    "en": "T2, Codex + Laydyne, E4 final: outputs answered busy, all runs",
    "ja": "T2・Codex＋Laydyne、E4の最終：busyの出力（全回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4base.inputTokens",
   "value": 613148,
   "unit": "tokens",
   "what": {
    "en": "T2, Codex + Laydyne, E4 baseline (main bf7bfec7): input tokens, mean of 2 runs",
    "ja": "T2・Codex＋Laydyne、E4の変更前（main bf7bfec7）：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4base-T2-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4base-T2-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4base.inputTokensRange",
   "value": [
    566146,
    660150
   ],
   "unit": "tokens",
   "what": {
    "en": "T2, Codex + Laydyne, E4 baseline (main bf7bfec7): input tokens, lowest and highest run",
    "ja": "T2・Codex＋Laydyne、E4の変更前（main bf7bfec7）：入力トークンの最小と最大"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4base-T2-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4base-T2-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4base.cachedInputTokens",
   "value": 538880,
   "unit": "tokens",
   "what": {
    "en": "T2, Codex + Laydyne, E4 baseline (main bf7bfec7): cached input tokens, mean",
    "ja": "T2・Codex＋Laydyne、E4の変更前（main bf7bfec7）：うちキャッシュ（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4base-T2-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4base-T2-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4base.uncachedInputTokens",
   "value": 74268,
   "unit": "tokens",
   "what": {
    "en": "T2, Codex + Laydyne, E4 baseline (main bf7bfec7): uncached input tokens, mean",
    "ja": "T2・Codex＋Laydyne、E4の変更前（main bf7bfec7）：キャッシュされない入力（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4base-T2-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4base-T2-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4base.outputTokens",
   "value": 1325,
   "unit": "tokens",
   "what": {
    "en": "T2, Codex + Laydyne, E4 baseline (main bf7bfec7): output tokens, mean",
    "ja": "T2・Codex＋Laydyne、E4の変更前（main bf7bfec7）：出力トークン（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4base-T2-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4base-T2-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4base.requests",
   "value": 13,
   "unit": "model requests",
   "what": {
    "en": "T2, Codex + Laydyne, E4 baseline (main bf7bfec7): model requests, mean",
    "ja": "T2・Codex＋Laydyne、E4の変更前（main bf7bfec7）：モデルへの要求回数（平均）"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4base-T2-B-r1/rollout.jsonl",
    "apps/studio/.wrangler/e4/runs/e4base-T2-B-r2/rollout.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4base.requestsRange",
   "value": [
    12,
    14
   ],
   "unit": "model requests",
   "what": {
    "en": "T2, Codex + Laydyne, E4 baseline (main bf7bfec7): model requests, lowest and highest run",
    "ja": "T2・Codex＋Laydyne、E4の変更前（main bf7bfec7）：要求回数の最小と最大"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4base-T2-B-r1/rollout.jsonl",
    "apps/studio/.wrangler/e4/runs/e4base-T2-B-r2/rollout.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4base.seconds",
   "value": 107,
   "unit": "s",
   "what": {
    "en": "T2, Codex + Laydyne, E4 baseline (main bf7bfec7): wall time, mean",
    "ja": "T2・Codex＋Laydyne、E4の変更前（main bf7bfec7）：所要時間（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4base-T2-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4base-T2-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4base.passed",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "T2, Codex + Laydyne, E4 baseline (main bf7bfec7): runs passing the scripted checks",
    "ja": "T2・Codex＋Laydyne、E4の変更前（main bf7bfec7）：採点スクリプトの合格数"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4base-T2-B-r1/score.json",
    "apps/studio/.wrangler/e4/runs/e4base-T2-B-r2/score.json"
   ]
  },
  {
   "id": "e4.T2.e4base.apiCostUsd",
   "value": 1.348,
   "unit": "USD (estimate)",
   "what": {
    "en": "T2, Codex + Laydyne, E4 baseline (main bf7bfec7): API list-price estimate (gpt-6-astra Standard: $10 input, $1 cached, $50 output per 1M); the runs used a ChatGPT sign-in, not API billing",
    "ja": "T2・Codex＋Laydyne、E4の変更前（main bf7bfec7）：API定価での推定費用（gpt-6-astra標準：入力$10・キャッシュ$1・出力$50／100万）。実行はChatGPTのサインインでAPI課金ではない"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4base-T2-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4base-T2-B-r2/codex.jsonl",
    "https://developers.openai.com/api/docs/pricing (read 2026-10-06)"
   ]
  },
  {
   "id": "e4.T2.e4base.resultChars",
   "value": 155649,
   "unit": "characters",
   "what": {
    "en": "T2, Codex + Laydyne, E4 baseline (main bf7bfec7): characters of tool output the model saw per run (exec outputs), mean",
    "ja": "T2・Codex＋Laydyne、E4の変更前（main bf7bfec7）：モデルが見た道具の出力の文字数（1回あたり、平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4base.doubledChars",
   "value": 22760,
   "unit": "characters",
   "what": {
    "en": "T2, Codex + Laydyne, E4 baseline (main bf7bfec7): characters of results printed twice (JSON text and structuredContent), mean",
    "ja": "T2・Codex＋Laydyne、E4の変更前（main bf7bfec7）：二重に表示された結果の文字数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4base.failedCalls",
   "value": 1,
   "unit": "calls",
   "what": {
    "en": "T2, Codex + Laydyne, E4 baseline (main bf7bfec7): failed tool calls, all runs",
    "ja": "T2・Codex＋Laydyne、E4の変更前（main bf7bfec7）：失敗した呼び出し（全回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4base-T2-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4base-T2-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T2.e4base.busy",
   "value": 0,
   "unit": "outputs",
   "what": {
    "en": "T2, Codex + Laydyne, E4 baseline (main bf7bfec7): outputs answered busy, all runs",
    "ja": "T2・Codex＋Laydyne、E4の変更前（main bf7bfec7）：busyの出力（全回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4a.inputTokens",
   "value": 398784,
   "unit": "tokens",
   "what": {
    "en": "T3, Codex + Laydyne, E4 first change set: input tokens, mean of 2 runs",
    "ja": "T3・Codex＋Laydyne、E4の1回目の変更：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4a-T3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4a-T3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4a.inputTokensRange",
   "value": [
    378022,
    419545
   ],
   "unit": "tokens",
   "what": {
    "en": "T3, Codex + Laydyne, E4 first change set: input tokens, lowest and highest run",
    "ja": "T3・Codex＋Laydyne、E4の1回目の変更：入力トークンの最小と最大"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4a-T3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4a-T3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4a.cachedInputTokens",
   "value": 363392,
   "unit": "tokens",
   "what": {
    "en": "T3, Codex + Laydyne, E4 first change set: cached input tokens, mean",
    "ja": "T3・Codex＋Laydyne、E4の1回目の変更：うちキャッシュ（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4a-T3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4a-T3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4a.uncachedInputTokens",
   "value": 35392,
   "unit": "tokens",
   "what": {
    "en": "T3, Codex + Laydyne, E4 first change set: uncached input tokens, mean",
    "ja": "T3・Codex＋Laydyne、E4の1回目の変更：キャッシュされない入力（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4a-T3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4a-T3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4a.outputTokens",
   "value": 2952,
   "unit": "tokens",
   "what": {
    "en": "T3, Codex + Laydyne, E4 first change set: output tokens, mean",
    "ja": "T3・Codex＋Laydyne、E4の1回目の変更：出力トークン（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4a-T3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4a-T3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4a.requests",
   "value": 10.5,
   "unit": "model requests",
   "what": {
    "en": "T3, Codex + Laydyne, E4 first change set: model requests, mean",
    "ja": "T3・Codex＋Laydyne、E4の1回目の変更：モデルへの要求回数（平均）"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4a-T3-B-r1/rollout.jsonl",
    "apps/studio/.wrangler/e4/runs/e4a-T3-B-r2/rollout.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4a.requestsRange",
   "value": [
    10,
    11
   ],
   "unit": "model requests",
   "what": {
    "en": "T3, Codex + Laydyne, E4 first change set: model requests, lowest and highest run",
    "ja": "T3・Codex＋Laydyne、E4の1回目の変更：要求回数の最小と最大"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4a-T3-B-r1/rollout.jsonl",
    "apps/studio/.wrangler/e4/runs/e4a-T3-B-r2/rollout.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4a.seconds",
   "value": 200,
   "unit": "s",
   "what": {
    "en": "T3, Codex + Laydyne, E4 first change set: wall time, mean",
    "ja": "T3・Codex＋Laydyne、E4の1回目の変更：所要時間（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4a-T3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4a-T3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4a.passed",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "T3, Codex + Laydyne, E4 first change set: runs passing the scripted checks",
    "ja": "T3・Codex＋Laydyne、E4の1回目の変更：採点スクリプトの合格数"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4a-T3-B-r1/score.json",
    "apps/studio/.wrangler/e4/runs/e4a-T3-B-r2/score.json"
   ]
  },
  {
   "id": "e4.T3.e4a.apiCostUsd",
   "value": 0.865,
   "unit": "USD (estimate)",
   "what": {
    "en": "T3, Codex + Laydyne, E4 first change set: API list-price estimate (gpt-6-astra Standard: $10 input, $1 cached, $50 output per 1M); the runs used a ChatGPT sign-in, not API billing",
    "ja": "T3・Codex＋Laydyne、E4の1回目の変更：API定価での推定費用（gpt-6-astra標準：入力$10・キャッシュ$1・出力$50／100万）。実行はChatGPTのサインインでAPI課金ではない"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4a-T3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4a-T3-B-r2/codex.jsonl",
    "https://developers.openai.com/api/docs/pricing (read 2026-10-06)"
   ]
  },
  {
   "id": "e4.T3.e4a.resultChars",
   "value": 75488,
   "unit": "characters",
   "what": {
    "en": "T3, Codex + Laydyne, E4 first change set: characters of tool output the model saw per run (exec outputs), mean",
    "ja": "T3・Codex＋Laydyne、E4の1回目の変更：モデルが見た道具の出力の文字数（1回あたり、平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4a.doubledChars",
   "value": 0,
   "unit": "characters",
   "what": {
    "en": "T3, Codex + Laydyne, E4 first change set: characters of results printed twice (JSON text and structuredContent), mean",
    "ja": "T3・Codex＋Laydyne、E4の1回目の変更：二重に表示された結果の文字数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4a.failedCalls",
   "value": 0,
   "unit": "calls",
   "what": {
    "en": "T3, Codex + Laydyne, E4 first change set: failed tool calls, all runs",
    "ja": "T3・Codex＋Laydyne、E4の1回目の変更：失敗した呼び出し（全回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4a-T3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4a-T3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4a.busy",
   "value": 0,
   "unit": "outputs",
   "what": {
    "en": "T3, Codex + Laydyne, E4 first change set: outputs answered busy, all runs",
    "ja": "T3・Codex＋Laydyne、E4の1回目の変更：busyの出力（全回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4alone.inputTokens",
   "value": 121202,
   "unit": "tokens",
   "what": {
    "en": "T3, e4alone: input tokens, mean of 2 runs",
    "ja": "T3・e4alone：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4alone-T3-A-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4alone-T3-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4alone.inputTokensRange",
   "value": [
    119295,
    123110
   ],
   "unit": "tokens",
   "what": {
    "en": "T3, e4alone: input tokens, lowest and highest run",
    "ja": "T3・e4alone：入力トークンの最小と最大"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4alone-T3-A-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4alone-T3-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4alone.cachedInputTokens",
   "value": 95296,
   "unit": "tokens",
   "what": {
    "en": "T3, e4alone: cached input tokens, mean",
    "ja": "T3・e4alone：うちキャッシュ（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4alone-T3-A-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4alone-T3-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4alone.uncachedInputTokens",
   "value": 25906,
   "unit": "tokens",
   "what": {
    "en": "T3, e4alone: uncached input tokens, mean",
    "ja": "T3・e4alone：キャッシュされない入力（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4alone-T3-A-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4alone-T3-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4alone.outputTokens",
   "value": 5172,
   "unit": "tokens",
   "what": {
    "en": "T3, e4alone: output tokens, mean",
    "ja": "T3・e4alone：出力トークン（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4alone-T3-A-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4alone-T3-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4alone.requests",
   "value": 5,
   "unit": "model requests",
   "what": {
    "en": "T3, e4alone: model requests, mean",
    "ja": "T3・e4alone：モデルへの要求回数（平均）"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4alone-T3-A-r1/rollout.jsonl",
    "apps/studio/.wrangler/e4/runs/e4alone-T3-A-r2/rollout.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4alone.requestsRange",
   "value": [
    5,
    5
   ],
   "unit": "model requests",
   "what": {
    "en": "T3, e4alone: model requests, lowest and highest run",
    "ja": "T3・e4alone：要求回数の最小と最大"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4alone-T3-A-r1/rollout.jsonl",
    "apps/studio/.wrangler/e4/runs/e4alone-T3-A-r2/rollout.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4alone.seconds",
   "value": 182,
   "unit": "s",
   "what": {
    "en": "T3, e4alone: wall time, mean",
    "ja": "T3・e4alone：所要時間（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4alone-T3-A-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4alone-T3-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4alone.passed",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "T3, e4alone: runs passing the scripted checks",
    "ja": "T3・e4alone：採点スクリプトの合格数"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4alone-T3-A-r1/score.json",
    "apps/studio/.wrangler/e4/runs/e4alone-T3-A-r2/score.json"
   ]
  },
  {
   "id": "e4.T3.e4alone.apiCostUsd",
   "value": 0.613,
   "unit": "USD (estimate)",
   "what": {
    "en": "T3, e4alone: API list-price estimate (gpt-6-astra Standard: $10 input, $1 cached, $50 output per 1M); the runs used a ChatGPT sign-in, not API billing",
    "ja": "T3・e4alone：API定価での推定費用（gpt-6-astra標準：入力$10・キャッシュ$1・出力$50／100万）。実行はChatGPTのサインインでAPI課金ではない"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4alone-T3-A-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4alone-T3-A-r2/codex.jsonl",
    "https://developers.openai.com/api/docs/pricing (read 2026-10-06)"
   ]
  },
  {
   "id": "e4.T3.e4alone.resultChars",
   "value": 1975,
   "unit": "characters",
   "what": {
    "en": "T3, e4alone: characters of tool output the model saw per run (exec outputs), mean",
    "ja": "T3・e4alone：モデルが見た道具の出力の文字数（1回あたり、平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4alone.doubledChars",
   "value": 0,
   "unit": "characters",
   "what": {
    "en": "T3, e4alone: characters of results printed twice (JSON text and structuredContent), mean",
    "ja": "T3・e4alone：二重に表示された結果の文字数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4alone.failedCalls",
   "value": 0,
   "unit": "calls",
   "what": {
    "en": "T3, e4alone: failed tool calls, all runs",
    "ja": "T3・e4alone：失敗した呼び出し（全回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4alone-T3-A-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4alone-T3-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4alone.busy",
   "value": 0,
   "unit": "outputs",
   "what": {
    "en": "T3, e4alone: outputs answered busy, all runs",
    "ja": "T3・e4alone：busyの出力（全回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4b.inputTokens",
   "value": 334425,
   "unit": "tokens",
   "what": {
    "en": "T3, Codex + Laydyne, E4 final: input tokens, mean of 2 runs",
    "ja": "T3・Codex＋Laydyne、E4の最終：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4b-T3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4b-T3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4b.inputTokensRange",
   "value": [
    332668,
    336182
   ],
   "unit": "tokens",
   "what": {
    "en": "T3, Codex + Laydyne, E4 final: input tokens, lowest and highest run",
    "ja": "T3・Codex＋Laydyne、E4の最終：入力トークンの最小と最大"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4b-T3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4b-T3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4b.cachedInputTokens",
   "value": 299648,
   "unit": "tokens",
   "what": {
    "en": "T3, Codex + Laydyne, E4 final: cached input tokens, mean",
    "ja": "T3・Codex＋Laydyne、E4の最終：うちキャッシュ（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4b-T3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4b-T3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4b.uncachedInputTokens",
   "value": 34777,
   "unit": "tokens",
   "what": {
    "en": "T3, Codex + Laydyne, E4 final: uncached input tokens, mean",
    "ja": "T3・Codex＋Laydyne、E4の最終：キャッシュされない入力（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4b-T3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4b-T3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4b.outputTokens",
   "value": 2569,
   "unit": "tokens",
   "what": {
    "en": "T3, Codex + Laydyne, E4 final: output tokens, mean",
    "ja": "T3・Codex＋Laydyne、E4の最終：出力トークン（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4b-T3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4b-T3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4b.requests",
   "value": 9,
   "unit": "model requests",
   "what": {
    "en": "T3, Codex + Laydyne, E4 final: model requests, mean",
    "ja": "T3・Codex＋Laydyne、E4の最終：モデルへの要求回数（平均）"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4b-T3-B-r1/rollout.jsonl",
    "apps/studio/.wrangler/e4/runs/e4b-T3-B-r2/rollout.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4b.requestsRange",
   "value": [
    9,
    9
   ],
   "unit": "model requests",
   "what": {
    "en": "T3, Codex + Laydyne, E4 final: model requests, lowest and highest run",
    "ja": "T3・Codex＋Laydyne、E4の最終：要求回数の最小と最大"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4b-T3-B-r1/rollout.jsonl",
    "apps/studio/.wrangler/e4/runs/e4b-T3-B-r2/rollout.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4b.seconds",
   "value": 132,
   "unit": "s",
   "what": {
    "en": "T3, Codex + Laydyne, E4 final: wall time, mean",
    "ja": "T3・Codex＋Laydyne、E4の最終：所要時間（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4b-T3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4b-T3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4b.passed",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "T3, Codex + Laydyne, E4 final: runs passing the scripted checks",
    "ja": "T3・Codex＋Laydyne、E4の最終：採点スクリプトの合格数"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4b-T3-B-r1/score.json",
    "apps/studio/.wrangler/e4/runs/e4b-T3-B-r2/score.json"
   ]
  },
  {
   "id": "e4.T3.e4b.apiCostUsd",
   "value": 0.776,
   "unit": "USD (estimate)",
   "what": {
    "en": "T3, Codex + Laydyne, E4 final: API list-price estimate (gpt-6-astra Standard: $10 input, $1 cached, $50 output per 1M); the runs used a ChatGPT sign-in, not API billing",
    "ja": "T3・Codex＋Laydyne、E4の最終：API定価での推定費用（gpt-6-astra標準：入力$10・キャッシュ$1・出力$50／100万）。実行はChatGPTのサインインでAPI課金ではない"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4b-T3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4b-T3-B-r2/codex.jsonl",
    "https://developers.openai.com/api/docs/pricing (read 2026-10-06)"
   ]
  },
  {
   "id": "e4.T3.e4b.resultChars",
   "value": 74440,
   "unit": "characters",
   "what": {
    "en": "T3, Codex + Laydyne, E4 final: characters of tool output the model saw per run (exec outputs), mean",
    "ja": "T3・Codex＋Laydyne、E4の最終：モデルが見た道具の出力の文字数（1回あたり、平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4b.doubledChars",
   "value": 0,
   "unit": "characters",
   "what": {
    "en": "T3, Codex + Laydyne, E4 final: characters of results printed twice (JSON text and structuredContent), mean",
    "ja": "T3・Codex＋Laydyne、E4の最終：二重に表示された結果の文字数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4b.failedCalls",
   "value": 0,
   "unit": "calls",
   "what": {
    "en": "T3, Codex + Laydyne, E4 final: failed tool calls, all runs",
    "ja": "T3・Codex＋Laydyne、E4の最終：失敗した呼び出し（全回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4b-T3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4b-T3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4b.busy",
   "value": 0,
   "unit": "outputs",
   "what": {
    "en": "T3, Codex + Laydyne, E4 final: outputs answered busy, all runs",
    "ja": "T3・Codex＋Laydyne、E4の最終：busyの出力（全回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4base.inputTokens",
   "value": 581063,
   "unit": "tokens",
   "what": {
    "en": "T3, Codex + Laydyne, E4 baseline (main bf7bfec7): input tokens, mean of 2 runs",
    "ja": "T3・Codex＋Laydyne、E4の変更前（main bf7bfec7）：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4base-T3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4base-T3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4base.inputTokensRange",
   "value": [
    455869,
    706257
   ],
   "unit": "tokens",
   "what": {
    "en": "T3, Codex + Laydyne, E4 baseline (main bf7bfec7): input tokens, lowest and highest run",
    "ja": "T3・Codex＋Laydyne、E4の変更前（main bf7bfec7）：入力トークンの最小と最大"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4base-T3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4base-T3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4base.cachedInputTokens",
   "value": 510592,
   "unit": "tokens",
   "what": {
    "en": "T3, Codex + Laydyne, E4 baseline (main bf7bfec7): cached input tokens, mean",
    "ja": "T3・Codex＋Laydyne、E4の変更前（main bf7bfec7）：うちキャッシュ（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4base-T3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4base-T3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4base.uncachedInputTokens",
   "value": 70471,
   "unit": "tokens",
   "what": {
    "en": "T3, Codex + Laydyne, E4 baseline (main bf7bfec7): uncached input tokens, mean",
    "ja": "T3・Codex＋Laydyne、E4の変更前（main bf7bfec7）：キャッシュされない入力（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4base-T3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4base-T3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4base.outputTokens",
   "value": 3263,
   "unit": "tokens",
   "what": {
    "en": "T3, Codex + Laydyne, E4 baseline (main bf7bfec7): output tokens, mean",
    "ja": "T3・Codex＋Laydyne、E4の変更前（main bf7bfec7）：出力トークン（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4base-T3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4base-T3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4base.requests",
   "value": 11.5,
   "unit": "model requests",
   "what": {
    "en": "T3, Codex + Laydyne, E4 baseline (main bf7bfec7): model requests, mean",
    "ja": "T3・Codex＋Laydyne、E4の変更前（main bf7bfec7）：モデルへの要求回数（平均）"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4base-T3-B-r1/rollout.jsonl",
    "apps/studio/.wrangler/e4/runs/e4base-T3-B-r2/rollout.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4base.requestsRange",
   "value": [
    10,
    13
   ],
   "unit": "model requests",
   "what": {
    "en": "T3, Codex + Laydyne, E4 baseline (main bf7bfec7): model requests, lowest and highest run",
    "ja": "T3・Codex＋Laydyne、E4の変更前（main bf7bfec7）：要求回数の最小と最大"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4base-T3-B-r1/rollout.jsonl",
    "apps/studio/.wrangler/e4/runs/e4base-T3-B-r2/rollout.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4base.seconds",
   "value": 194,
   "unit": "s",
   "what": {
    "en": "T3, Codex + Laydyne, E4 baseline (main bf7bfec7): wall time, mean",
    "ja": "T3・Codex＋Laydyne、E4の変更前（main bf7bfec7）：所要時間（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4base-T3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4base-T3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4base.passed",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "T3, Codex + Laydyne, E4 baseline (main bf7bfec7): runs passing the scripted checks",
    "ja": "T3・Codex＋Laydyne、E4の変更前（main bf7bfec7）：採点スクリプトの合格数"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4base-T3-B-r1/score.json",
    "apps/studio/.wrangler/e4/runs/e4base-T3-B-r2/score.json"
   ]
  },
  {
   "id": "e4.T3.e4base.apiCostUsd",
   "value": 1.379,
   "unit": "USD (estimate)",
   "what": {
    "en": "T3, Codex + Laydyne, E4 baseline (main bf7bfec7): API list-price estimate (gpt-6-astra Standard: $10 input, $1 cached, $50 output per 1M); the runs used a ChatGPT sign-in, not API billing",
    "ja": "T3・Codex＋Laydyne、E4の変更前（main bf7bfec7）：API定価での推定費用（gpt-6-astra標準：入力$10・キャッシュ$1・出力$50／100万）。実行はChatGPTのサインインでAPI課金ではない"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4base-T3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4base-T3-B-r2/codex.jsonl",
    "https://developers.openai.com/api/docs/pricing (read 2026-10-06)"
   ]
  },
  {
   "id": "e4.T3.e4base.resultChars",
   "value": 148014,
   "unit": "characters",
   "what": {
    "en": "T3, Codex + Laydyne, E4 baseline (main bf7bfec7): characters of tool output the model saw per run (exec outputs), mean",
    "ja": "T3・Codex＋Laydyne、E4の変更前（main bf7bfec7）：モデルが見た道具の出力の文字数（1回あたり、平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4base.doubledChars",
   "value": 9964,
   "unit": "characters",
   "what": {
    "en": "T3, Codex + Laydyne, E4 baseline (main bf7bfec7): characters of results printed twice (JSON text and structuredContent), mean",
    "ja": "T3・Codex＋Laydyne、E4の変更前（main bf7bfec7）：二重に表示された結果の文字数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4base.failedCalls",
   "value": 0,
   "unit": "calls",
   "what": {
    "en": "T3, Codex + Laydyne, E4 baseline (main bf7bfec7): failed tool calls, all runs",
    "ja": "T3・Codex＋Laydyne、E4の変更前（main bf7bfec7）：失敗した呼び出し（全回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4base-T3-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4base-T3-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T3.e4base.busy",
   "value": 0,
   "unit": "outputs",
   "what": {
    "en": "T3, Codex + Laydyne, E4 baseline (main bf7bfec7): outputs answered busy, all runs",
    "ja": "T3・Codex＋Laydyne、E4の変更前（main bf7bfec7）：busyの出力（全回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4a.inputTokens",
   "value": 577098,
   "unit": "tokens",
   "what": {
    "en": "T4, Codex + Laydyne, E4 first change set: input tokens, mean of 2 runs",
    "ja": "T4・Codex＋Laydyne、E4の1回目の変更：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4a-T4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4a-T4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4a.inputTokensRange",
   "value": [
    519072,
    635124
   ],
   "unit": "tokens",
   "what": {
    "en": "T4, Codex + Laydyne, E4 first change set: input tokens, lowest and highest run",
    "ja": "T4・Codex＋Laydyne、E4の1回目の変更：入力トークンの最小と最大"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4a-T4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4a-T4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4a.cachedInputTokens",
   "value": 527744,
   "unit": "tokens",
   "what": {
    "en": "T4, Codex + Laydyne, E4 first change set: cached input tokens, mean",
    "ja": "T4・Codex＋Laydyne、E4の1回目の変更：うちキャッシュ（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4a-T4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4a-T4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4a.uncachedInputTokens",
   "value": 49354,
   "unit": "tokens",
   "what": {
    "en": "T4, Codex + Laydyne, E4 first change set: uncached input tokens, mean",
    "ja": "T4・Codex＋Laydyne、E4の1回目の変更：キャッシュされない入力（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4a-T4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4a-T4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4a.outputTokens",
   "value": 2317,
   "unit": "tokens",
   "what": {
    "en": "T4, Codex + Laydyne, E4 first change set: output tokens, mean",
    "ja": "T4・Codex＋Laydyne、E4の1回目の変更：出力トークン（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4a-T4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4a-T4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4a.requests",
   "value": 13.5,
   "unit": "model requests",
   "what": {
    "en": "T4, Codex + Laydyne, E4 first change set: model requests, mean",
    "ja": "T4・Codex＋Laydyne、E4の1回目の変更：モデルへの要求回数（平均）"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4a-T4-B-r1/rollout.jsonl",
    "apps/studio/.wrangler/e4/runs/e4a-T4-B-r2/rollout.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4a.requestsRange",
   "value": [
    13,
    14
   ],
   "unit": "model requests",
   "what": {
    "en": "T4, Codex + Laydyne, E4 first change set: model requests, lowest and highest run",
    "ja": "T4・Codex＋Laydyne、E4の1回目の変更：要求回数の最小と最大"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4a-T4-B-r1/rollout.jsonl",
    "apps/studio/.wrangler/e4/runs/e4a-T4-B-r2/rollout.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4a.seconds",
   "value": 164,
   "unit": "s",
   "what": {
    "en": "T4, Codex + Laydyne, E4 first change set: wall time, mean",
    "ja": "T4・Codex＋Laydyne、E4の1回目の変更：所要時間（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4a-T4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4a-T4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4a.passed",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "T4, Codex + Laydyne, E4 first change set: runs passing the scripted checks",
    "ja": "T4・Codex＋Laydyne、E4の1回目の変更：採点スクリプトの合格数"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4a-T4-B-r1/score.json",
    "apps/studio/.wrangler/e4/runs/e4a-T4-B-r2/score.json"
   ]
  },
  {
   "id": "e4.T4.e4a.apiCostUsd",
   "value": 1.137,
   "unit": "USD (estimate)",
   "what": {
    "en": "T4, Codex + Laydyne, E4 first change set: API list-price estimate (gpt-6-astra Standard: $10 input, $1 cached, $50 output per 1M); the runs used a ChatGPT sign-in, not API billing",
    "ja": "T4・Codex＋Laydyne、E4の1回目の変更：API定価での推定費用（gpt-6-astra標準：入力$10・キャッシュ$1・出力$50／100万）。実行はChatGPTのサインインでAPI課金ではない"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4a-T4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4a-T4-B-r2/codex.jsonl",
    "https://developers.openai.com/api/docs/pricing (read 2026-10-06)"
   ]
  },
  {
   "id": "e4.T4.e4a.resultChars",
   "value": 86966,
   "unit": "characters",
   "what": {
    "en": "T4, Codex + Laydyne, E4 first change set: characters of tool output the model saw per run (exec outputs), mean",
    "ja": "T4・Codex＋Laydyne、E4の1回目の変更：モデルが見た道具の出力の文字数（1回あたり、平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4a.doubledChars",
   "value": 0,
   "unit": "characters",
   "what": {
    "en": "T4, Codex + Laydyne, E4 first change set: characters of results printed twice (JSON text and structuredContent), mean",
    "ja": "T4・Codex＋Laydyne、E4の1回目の変更：二重に表示された結果の文字数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4a.failedCalls",
   "value": 0,
   "unit": "calls",
   "what": {
    "en": "T4, Codex + Laydyne, E4 first change set: failed tool calls, all runs",
    "ja": "T4・Codex＋Laydyne、E4の1回目の変更：失敗した呼び出し（全回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4a-T4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4a-T4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4a.busy",
   "value": 0,
   "unit": "outputs",
   "what": {
    "en": "T4, Codex + Laydyne, E4 first change set: outputs answered busy, all runs",
    "ja": "T4・Codex＋Laydyne、E4の1回目の変更：busyの出力（全回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4alone.inputTokens",
   "value": 257262,
   "unit": "tokens",
   "what": {
    "en": "T4, e4alone: input tokens, mean of 2 runs",
    "ja": "T4・e4alone：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4alone-T4-A-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4alone-T4-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4alone.inputTokensRange",
   "value": [
    254735,
    259788
   ],
   "unit": "tokens",
   "what": {
    "en": "T4, e4alone: input tokens, lowest and highest run",
    "ja": "T4・e4alone：入力トークンの最小と最大"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4alone-T4-A-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4alone-T4-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4alone.cachedInputTokens",
   "value": 227008,
   "unit": "tokens",
   "what": {
    "en": "T4, e4alone: cached input tokens, mean",
    "ja": "T4・e4alone：うちキャッシュ（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4alone-T4-A-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4alone-T4-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4alone.uncachedInputTokens",
   "value": 30254,
   "unit": "tokens",
   "what": {
    "en": "T4, e4alone: uncached input tokens, mean",
    "ja": "T4・e4alone：キャッシュされない入力（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4alone-T4-A-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4alone-T4-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4alone.outputTokens",
   "value": 3486,
   "unit": "tokens",
   "what": {
    "en": "T4, e4alone: output tokens, mean",
    "ja": "T4・e4alone：出力トークン（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4alone-T4-A-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4alone-T4-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4alone.requests",
   "value": 7.5,
   "unit": "model requests",
   "what": {
    "en": "T4, e4alone: model requests, mean",
    "ja": "T4・e4alone：モデルへの要求回数（平均）"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4alone-T4-A-r1/rollout.jsonl",
    "apps/studio/.wrangler/e4/runs/e4alone-T4-A-r2/rollout.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4alone.requestsRange",
   "value": [
    7,
    8
   ],
   "unit": "model requests",
   "what": {
    "en": "T4, e4alone: model requests, lowest and highest run",
    "ja": "T4・e4alone：要求回数の最小と最大"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4alone-T4-A-r1/rollout.jsonl",
    "apps/studio/.wrangler/e4/runs/e4alone-T4-A-r2/rollout.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4alone.seconds",
   "value": 282,
   "unit": "s",
   "what": {
    "en": "T4, e4alone: wall time, mean",
    "ja": "T4・e4alone：所要時間（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4alone-T4-A-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4alone-T4-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4alone.passed",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "T4, e4alone: runs passing the scripted checks",
    "ja": "T4・e4alone：採点スクリプトの合格数"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4alone-T4-A-r1/score.json",
    "apps/studio/.wrangler/e4/runs/e4alone-T4-A-r2/score.json"
   ]
  },
  {
   "id": "e4.T4.e4alone.apiCostUsd",
   "value": 0.704,
   "unit": "USD (estimate)",
   "what": {
    "en": "T4, e4alone: API list-price estimate (gpt-6-astra Standard: $10 input, $1 cached, $50 output per 1M); the runs used a ChatGPT sign-in, not API billing",
    "ja": "T4・e4alone：API定価での推定費用（gpt-6-astra標準：入力$10・キャッシュ$1・出力$50／100万）。実行はChatGPTのサインインでAPI課金ではない"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4alone-T4-A-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4alone-T4-A-r2/codex.jsonl",
    "https://developers.openai.com/api/docs/pricing (read 2026-10-06)"
   ]
  },
  {
   "id": "e4.T4.e4alone.resultChars",
   "value": 44025,
   "unit": "characters",
   "what": {
    "en": "T4, e4alone: characters of tool output the model saw per run (exec outputs), mean",
    "ja": "T4・e4alone：モデルが見た道具の出力の文字数（1回あたり、平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4alone.doubledChars",
   "value": 0,
   "unit": "characters",
   "what": {
    "en": "T4, e4alone: characters of results printed twice (JSON text and structuredContent), mean",
    "ja": "T4・e4alone：二重に表示された結果の文字数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4alone.failedCalls",
   "value": 0,
   "unit": "calls",
   "what": {
    "en": "T4, e4alone: failed tool calls, all runs",
    "ja": "T4・e4alone：失敗した呼び出し（全回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4alone-T4-A-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4alone-T4-A-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4alone.busy",
   "value": 0,
   "unit": "outputs",
   "what": {
    "en": "T4, e4alone: outputs answered busy, all runs",
    "ja": "T4・e4alone：busyの出力（全回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4b.inputTokens",
   "value": 375160,
   "unit": "tokens",
   "what": {
    "en": "T4, Codex + Laydyne, E4 final: input tokens, mean of 2 runs",
    "ja": "T4・Codex＋Laydyne、E4の最終：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4b-T4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4b-T4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4b.inputTokensRange",
   "value": [
    238334,
    511987
   ],
   "unit": "tokens",
   "what": {
    "en": "T4, Codex + Laydyne, E4 final: input tokens, lowest and highest run",
    "ja": "T4・Codex＋Laydyne、E4の最終：入力トークンの最小と最大"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4b-T4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4b-T4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4b.cachedInputTokens",
   "value": 338752,
   "unit": "tokens",
   "what": {
    "en": "T4, Codex + Laydyne, E4 final: cached input tokens, mean",
    "ja": "T4・Codex＋Laydyne、E4の最終：うちキャッシュ（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4b-T4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4b-T4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4b.uncachedInputTokens",
   "value": 36408,
   "unit": "tokens",
   "what": {
    "en": "T4, Codex + Laydyne, E4 final: uncached input tokens, mean",
    "ja": "T4・Codex＋Laydyne、E4の最終：キャッシュされない入力（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4b-T4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4b-T4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4b.outputTokens",
   "value": 2002,
   "unit": "tokens",
   "what": {
    "en": "T4, Codex + Laydyne, E4 final: output tokens, mean",
    "ja": "T4・Codex＋Laydyne、E4の最終：出力トークン（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4b-T4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4b-T4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4b.requests",
   "value": 9.5,
   "unit": "model requests",
   "what": {
    "en": "T4, Codex + Laydyne, E4 final: model requests, mean",
    "ja": "T4・Codex＋Laydyne、E4の最終：モデルへの要求回数（平均）"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4b-T4-B-r1/rollout.jsonl",
    "apps/studio/.wrangler/e4/runs/e4b-T4-B-r2/rollout.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4b.requestsRange",
   "value": [
    7,
    12
   ],
   "unit": "model requests",
   "what": {
    "en": "T4, Codex + Laydyne, E4 final: model requests, lowest and highest run",
    "ja": "T4・Codex＋Laydyne、E4の最終：要求回数の最小と最大"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4b-T4-B-r1/rollout.jsonl",
    "apps/studio/.wrangler/e4/runs/e4b-T4-B-r2/rollout.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4b.seconds",
   "value": 100,
   "unit": "s",
   "what": {
    "en": "T4, Codex + Laydyne, E4 final: wall time, mean",
    "ja": "T4・Codex＋Laydyne、E4の最終：所要時間（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4b-T4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4b-T4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4b.passed",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "T4, Codex + Laydyne, E4 final: runs passing the scripted checks",
    "ja": "T4・Codex＋Laydyne、E4の最終：採点スクリプトの合格数"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4b-T4-B-r1/score.json",
    "apps/studio/.wrangler/e4/runs/e4b-T4-B-r2/score.json"
   ]
  },
  {
   "id": "e4.T4.e4b.apiCostUsd",
   "value": 0.803,
   "unit": "USD (estimate)",
   "what": {
    "en": "T4, Codex + Laydyne, E4 final: API list-price estimate (gpt-6-astra Standard: $10 input, $1 cached, $50 output per 1M); the runs used a ChatGPT sign-in, not API billing",
    "ja": "T4・Codex＋Laydyne、E4の最終：API定価での推定費用（gpt-6-astra標準：入力$10・キャッシュ$1・出力$50／100万）。実行はChatGPTのサインインでAPI課金ではない"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4b-T4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4b-T4-B-r2/codex.jsonl",
    "https://developers.openai.com/api/docs/pricing (read 2026-10-06)"
   ]
  },
  {
   "id": "e4.T4.e4b.resultChars",
   "value": 79323,
   "unit": "characters",
   "what": {
    "en": "T4, Codex + Laydyne, E4 final: characters of tool output the model saw per run (exec outputs), mean",
    "ja": "T4・Codex＋Laydyne、E4の最終：モデルが見た道具の出力の文字数（1回あたり、平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4b.doubledChars",
   "value": 0,
   "unit": "characters",
   "what": {
    "en": "T4, Codex + Laydyne, E4 final: characters of results printed twice (JSON text and structuredContent), mean",
    "ja": "T4・Codex＋Laydyne、E4の最終：二重に表示された結果の文字数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4b.failedCalls",
   "value": 0,
   "unit": "calls",
   "what": {
    "en": "T4, Codex + Laydyne, E4 final: failed tool calls, all runs",
    "ja": "T4・Codex＋Laydyne、E4の最終：失敗した呼び出し（全回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4b-T4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4b-T4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4b.busy",
   "value": 0,
   "unit": "outputs",
   "what": {
    "en": "T4, Codex + Laydyne, E4 final: outputs answered busy, all runs",
    "ja": "T4・Codex＋Laydyne、E4の最終：busyの出力（全回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4base.inputTokens",
   "value": 506980,
   "unit": "tokens",
   "what": {
    "en": "T4, Codex + Laydyne, E4 baseline (main bf7bfec7): input tokens, mean of 2 runs",
    "ja": "T4・Codex＋Laydyne、E4の変更前（main bf7bfec7）：入力トークン（2回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4base-T4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4base-T4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4base.inputTokensRange",
   "value": [
    483249,
    530710
   ],
   "unit": "tokens",
   "what": {
    "en": "T4, Codex + Laydyne, E4 baseline (main bf7bfec7): input tokens, lowest and highest run",
    "ja": "T4・Codex＋Laydyne、E4の変更前（main bf7bfec7）：入力トークンの最小と最大"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4base-T4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4base-T4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4base.cachedInputTokens",
   "value": 442240,
   "unit": "tokens",
   "what": {
    "en": "T4, Codex + Laydyne, E4 baseline (main bf7bfec7): cached input tokens, mean",
    "ja": "T4・Codex＋Laydyne、E4の変更前（main bf7bfec7）：うちキャッシュ（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4base-T4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4base-T4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4base.uncachedInputTokens",
   "value": 64740,
   "unit": "tokens",
   "what": {
    "en": "T4, Codex + Laydyne, E4 baseline (main bf7bfec7): uncached input tokens, mean",
    "ja": "T4・Codex＋Laydyne、E4の変更前（main bf7bfec7）：キャッシュされない入力（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4base-T4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4base-T4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4base.outputTokens",
   "value": 2102,
   "unit": "tokens",
   "what": {
    "en": "T4, Codex + Laydyne, E4 baseline (main bf7bfec7): output tokens, mean",
    "ja": "T4・Codex＋Laydyne、E4の変更前（main bf7bfec7）：出力トークン（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4base-T4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4base-T4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4base.requests",
   "value": 11,
   "unit": "model requests",
   "what": {
    "en": "T4, Codex + Laydyne, E4 baseline (main bf7bfec7): model requests, mean",
    "ja": "T4・Codex＋Laydyne、E4の変更前（main bf7bfec7）：モデルへの要求回数（平均）"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4base-T4-B-r1/rollout.jsonl",
    "apps/studio/.wrangler/e4/runs/e4base-T4-B-r2/rollout.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4base.requestsRange",
   "value": [
    10,
    12
   ],
   "unit": "model requests",
   "what": {
    "en": "T4, Codex + Laydyne, E4 baseline (main bf7bfec7): model requests, lowest and highest run",
    "ja": "T4・Codex＋Laydyne、E4の変更前（main bf7bfec7）：要求回数の最小と最大"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4base-T4-B-r1/rollout.jsonl",
    "apps/studio/.wrangler/e4/runs/e4base-T4-B-r2/rollout.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4base.seconds",
   "value": 137,
   "unit": "s",
   "what": {
    "en": "T4, Codex + Laydyne, E4 baseline (main bf7bfec7): wall time, mean",
    "ja": "T4・Codex＋Laydyne、E4の変更前（main bf7bfec7）：所要時間（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4base-T4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4base-T4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4base.passed",
   "value": "2/2",
   "unit": "runs",
   "what": {
    "en": "T4, Codex + Laydyne, E4 baseline (main bf7bfec7): runs passing the scripted checks",
    "ja": "T4・Codex＋Laydyne、E4の変更前（main bf7bfec7）：採点スクリプトの合格数"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4base-T4-B-r1/score.json",
    "apps/studio/.wrangler/e4/runs/e4base-T4-B-r2/score.json"
   ]
  },
  {
   "id": "e4.T4.e4base.apiCostUsd",
   "value": 1.195,
   "unit": "USD (estimate)",
   "what": {
    "en": "T4, Codex + Laydyne, E4 baseline (main bf7bfec7): API list-price estimate (gpt-6-astra Standard: $10 input, $1 cached, $50 output per 1M); the runs used a ChatGPT sign-in, not API billing",
    "ja": "T4・Codex＋Laydyne、E4の変更前（main bf7bfec7）：API定価での推定費用（gpt-6-astra標準：入力$10・キャッシュ$1・出力$50／100万）。実行はChatGPTのサインインでAPI課金ではない"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4base-T4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4base-T4-B-r2/codex.jsonl",
    "https://developers.openai.com/api/docs/pricing (read 2026-10-06)"
   ]
  },
  {
   "id": "e4.T4.e4base.resultChars",
   "value": 129040,
   "unit": "characters",
   "what": {
    "en": "T4, Codex + Laydyne, E4 baseline (main bf7bfec7): characters of tool output the model saw per run (exec outputs), mean",
    "ja": "T4・Codex＋Laydyne、E4の変更前（main bf7bfec7）：モデルが見た道具の出力の文字数（1回あたり、平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4base.doubledChars",
   "value": 10469,
   "unit": "characters",
   "what": {
    "en": "T4, Codex + Laydyne, E4 baseline (main bf7bfec7): characters of results printed twice (JSON text and structuredContent), mean",
    "ja": "T4・Codex＋Laydyne、E4の変更前（main bf7bfec7）：二重に表示された結果の文字数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4base.failedCalls",
   "value": 0,
   "unit": "calls",
   "what": {
    "en": "T4, Codex + Laydyne, E4 baseline (main bf7bfec7): failed tool calls, all runs",
    "ja": "T4・Codex＋Laydyne、E4の変更前（main bf7bfec7）：失敗した呼び出し（全回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-compare.json",
    "apps/studio/.wrangler/e4/runs/e4base-T4-B-r1/codex.jsonl",
    "apps/studio/.wrangler/e4/runs/e4base-T4-B-r2/codex.jsonl"
   ]
  },
  {
   "id": "e4.T4.e4base.busy",
   "value": 0,
   "unit": "outputs",
   "what": {
    "en": "T4, Codex + Laydyne, E4 baseline (main bf7bfec7): outputs answered busy, all runs",
    "ja": "T4・Codex＋Laydyne、E4の変更前（main bf7bfec7）：busyの出力（全回の合計）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4a.inputTokensPerRun",
   "value": 447779,
   "unit": "tokens",
   "what": {
    "en": "T1–T4, Codex + Laydyne, E4 first change set: input tokens per run, mean of 8 runs",
    "ja": "T1〜T4・Codex＋Laydyne、E4の1回目の変更：1回あたりの入力トークン（8回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4a.requestsPerRun",
   "value": 11.38,
   "unit": "model requests",
   "what": {
    "en": "T1–T4, Codex + Laydyne, E4 first change set: model requests per run, mean",
    "ja": "T1〜T4・Codex＋Laydyne、E4の1回目の変更：1回あたりの要求回数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4a.resultCharsPerRun",
   "value": 84105,
   "unit": "characters",
   "what": {
    "en": "T1–T4, Codex + Laydyne, E4 first change set: tool output characters the model saw per run, mean",
    "ja": "T1〜T4・Codex＋Laydyne、E4の1回目の変更：1回あたりモデルが見た道具の出力の文字数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4a.doubledCharsPerRun",
   "value": 0,
   "unit": "characters",
   "what": {
    "en": "T1–T4, Codex + Laydyne, E4 first change set: characters printed twice per run, mean",
    "ja": "T1〜T4・Codex＋Laydyne、E4の1回目の変更：1回あたり二重に表示された文字数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4a.catalogShare",
   "value": 29.8,
   "unit": "% (estimate)",
   "what": {
    "en": "T1–T4, Codex + Laydyne, E4 first change set: estimated share of input tokens from tool-list lookups re-sent with later requests (characters/4 × later requests)",
    "ja": "T1〜T4・Codex＋Laydyne、E4の1回目の変更：道具一覧の参照の出力が後続の要求で再送された分の推定割合（文字数/4×後続の要求回数）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4a.catalogRequestsPerRun",
   "value": 2.5,
   "unit": "model requests",
   "what": {
    "en": "T1–T4, Codex + Laydyne, E4 first change set: requests whose script did tool-list lookups, per run",
    "ja": "T1〜T4・Codex＋Laydyne、E4の1回目の変更：道具一覧の参照をした要求の回数（1回あたり）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4a.readShare",
   "value": 4.6,
   "unit": "% (estimate)",
   "what": {
    "en": "T1–T4, Codex + Laydyne, E4 first change set: estimated share of input tokens from reads (get_space and the like) re-sent with later requests (characters/4 × later requests)",
    "ja": "T1〜T4・Codex＋Laydyne、E4の1回目の変更：読み取りの出力が後続の要求で再送された分の推定割合（文字数/4×後続の要求回数）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4a.readRequestsPerRun",
   "value": 1.38,
   "unit": "model requests",
   "what": {
    "en": "T1–T4, Codex + Laydyne, E4 first change set: requests whose script did reads (get_space and the like), per run",
    "ja": "T1〜T4・Codex＋Laydyne、E4の1回目の変更：読み取りをした要求の回数（1回あたり）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4a.fileShare",
   "value": 3.8,
   "unit": "% (estimate)",
   "what": {
    "en": "T1–T4, Codex + Laydyne, E4 first change set: estimated share of input tokens from files (SVG, PDF, export) re-sent with later requests (characters/4 × later requests)",
    "ja": "T1〜T4・Codex＋Laydyne、E4の1回目の変更：ファイルの出力が後続の要求で再送された分の推定割合（文字数/4×後続の要求回数）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4a.fileRequestsPerRun",
   "value": 2.62,
   "unit": "model requests",
   "what": {
    "en": "T1–T4, Codex + Laydyne, E4 first change set: requests whose script did files (SVG, PDF, export), per run",
    "ja": "T1〜T4・Codex＋Laydyne、E4の1回目の変更：ファイルをした要求の回数（1回あたり）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4a.ownContextShare",
   "value": 53.5,
   "unit": "% (estimate)",
   "what": {
    "en": "T1–T4, Codex + Laydyne, E4 first change set: Codex's own context (first request) × requests, share of input tokens",
    "ja": "T1〜T4・Codex＋Laydyne、E4の1回目の変更：Codex自身の文脈（最初の要求）×要求回数が入力に占める割合"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4a.firstRequestTokens",
   "value": 21082,
   "unit": "tokens",
   "what": {
    "en": "T1–T4, Codex + Laydyne, E4 first change set: Codex's own context at the first request, mean",
    "ja": "T1〜T4・Codex＋Laydyne、E4の1回目の変更：最初の要求時点のCodex自身の文脈（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4alone.inputTokensPerRun",
   "value": 157922,
   "unit": "tokens",
   "what": {
    "en": "T1–T4, e4alone: input tokens per run, mean of 8 runs",
    "ja": "T1〜T4・e4alone：1回あたりの入力トークン（8回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4alone.requestsPerRun",
   "value": 5.62,
   "unit": "model requests",
   "what": {
    "en": "T1–T4, e4alone: model requests per run, mean",
    "ja": "T1〜T4・e4alone：1回あたりの要求回数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4alone.resultCharsPerRun",
   "value": 19940,
   "unit": "characters",
   "what": {
    "en": "T1–T4, e4alone: tool output characters the model saw per run, mean",
    "ja": "T1〜T4・e4alone：1回あたりモデルが見た道具の出力の文字数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4alone.doubledCharsPerRun",
   "value": 0,
   "unit": "characters",
   "what": {
    "en": "T1–T4, e4alone: characters printed twice per run, mean",
    "ja": "T1〜T4・e4alone：1回あたり二重に表示された文字数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4alone.catalogShare",
   "value": 0.0,
   "unit": "% (estimate)",
   "what": {
    "en": "T1–T4, e4alone: estimated share of input tokens from tool-list lookups re-sent with later requests (characters/4 × later requests)",
    "ja": "T1〜T4・e4alone：道具一覧の参照の出力が後続の要求で再送された分の推定割合（文字数/4×後続の要求回数）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4alone.catalogRequestsPerRun",
   "value": 0.0,
   "unit": "model requests",
   "what": {
    "en": "T1–T4, e4alone: requests whose script did tool-list lookups, per run",
    "ja": "T1〜T4・e4alone：道具一覧の参照をした要求の回数（1回あたり）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4alone.readShare",
   "value": 0.0,
   "unit": "% (estimate)",
   "what": {
    "en": "T1–T4, e4alone: estimated share of input tokens from reads (get_space and the like) re-sent with later requests (characters/4 × later requests)",
    "ja": "T1〜T4・e4alone：読み取りの出力が後続の要求で再送された分の推定割合（文字数/4×後続の要求回数）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4alone.readRequestsPerRun",
   "value": 0.0,
   "unit": "model requests",
   "what": {
    "en": "T1–T4, e4alone: requests whose script did reads (get_space and the like), per run",
    "ja": "T1〜T4・e4alone：読み取りをした要求の回数（1回あたり）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4alone.fileShare",
   "value": 0.0,
   "unit": "% (estimate)",
   "what": {
    "en": "T1–T4, e4alone: estimated share of input tokens from files (SVG, PDF, export) re-sent with later requests (characters/4 × later requests)",
    "ja": "T1〜T4・e4alone：ファイルの出力が後続の要求で再送された分の推定割合（文字数/4×後続の要求回数）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4alone.fileRequestsPerRun",
   "value": 0.0,
   "unit": "model requests",
   "what": {
    "en": "T1–T4, e4alone: requests whose script did files (SVG, PDF, export), per run",
    "ja": "T1〜T4・e4alone：ファイルをした要求の回数（1回あたり）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4alone.ownContextShare",
   "value": 75.5,
   "unit": "% (estimate)",
   "what": {
    "en": "T1–T4, e4alone: Codex's own context (first request) × requests, share of input tokens",
    "ja": "T1〜T4・e4alone：Codex自身の文脈（最初の要求）×要求回数が入力に占める割合"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4alone.firstRequestTokens",
   "value": 21198,
   "unit": "tokens",
   "what": {
    "en": "T1–T4, e4alone: Codex's own context at the first request, mean",
    "ja": "T1〜T4・e4alone：最初の要求時点のCodex自身の文脈（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4b.inputTokensPerRun",
   "value": 336342,
   "unit": "tokens",
   "what": {
    "en": "T1–T4, Codex + Laydyne, E4 final: input tokens per run, mean of 8 runs",
    "ja": "T1〜T4・Codex＋Laydyne、E4の最終：1回あたりの入力トークン（8回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4b.requestsPerRun",
   "value": 9.12,
   "unit": "model requests",
   "what": {
    "en": "T1–T4, Codex + Laydyne, E4 final: model requests per run, mean",
    "ja": "T1〜T4・Codex＋Laydyne、E4の最終：1回あたりの要求回数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4b.resultCharsPerRun",
   "value": 76432,
   "unit": "characters",
   "what": {
    "en": "T1–T4, Codex + Laydyne, E4 final: tool output characters the model saw per run, mean",
    "ja": "T1〜T4・Codex＋Laydyne、E4の最終：1回あたりモデルが見た道具の出力の文字数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4b.doubledCharsPerRun",
   "value": 0,
   "unit": "characters",
   "what": {
    "en": "T1–T4, Codex + Laydyne, E4 final: characters printed twice per run, mean",
    "ja": "T1〜T4・Codex＋Laydyne、E4の最終：1回あたり二重に表示された文字数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4b.catalogShare",
   "value": 27.4,
   "unit": "% (estimate)",
   "what": {
    "en": "T1–T4, Codex + Laydyne, E4 final: estimated share of input tokens from tool-list lookups re-sent with later requests (characters/4 × later requests)",
    "ja": "T1〜T4・Codex＋Laydyne、E4の最終：道具一覧の参照の出力が後続の要求で再送された分の推定割合（文字数/4×後続の要求回数）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4b.catalogRequestsPerRun",
   "value": 1.88,
   "unit": "model requests",
   "what": {
    "en": "T1–T4, Codex + Laydyne, E4 final: requests whose script did tool-list lookups, per run",
    "ja": "T1〜T4・Codex＋Laydyne、E4の最終：道具一覧の参照をした要求の回数（1回あたり）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4b.readShare",
   "value": 5.2,
   "unit": "% (estimate)",
   "what": {
    "en": "T1–T4, Codex + Laydyne, E4 final: estimated share of input tokens from reads (get_space and the like) re-sent with later requests (characters/4 × later requests)",
    "ja": "T1〜T4・Codex＋Laydyne、E4の最終：読み取りの出力が後続の要求で再送された分の推定割合（文字数/4×後続の要求回数）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4b.readRequestsPerRun",
   "value": 1.25,
   "unit": "model requests",
   "what": {
    "en": "T1–T4, Codex + Laydyne, E4 final: requests whose script did reads (get_space and the like), per run",
    "ja": "T1〜T4・Codex＋Laydyne、E4の最終：読み取りをした要求の回数（1回あたり）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4b.fileShare",
   "value": 3.3,
   "unit": "% (estimate)",
   "what": {
    "en": "T1–T4, Codex + Laydyne, E4 final: estimated share of input tokens from files (SVG, PDF, export) re-sent with later requests (characters/4 × later requests)",
    "ja": "T1〜T4・Codex＋Laydyne、E4の最終：ファイルの出力が後続の要求で再送された分の推定割合（文字数/4×後続の要求回数）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4b.fileRequestsPerRun",
   "value": 1.88,
   "unit": "model requests",
   "what": {
    "en": "T1–T4, Codex + Laydyne, E4 final: requests whose script did files (SVG, PDF, export), per run",
    "ja": "T1〜T4・Codex＋Laydyne、E4の最終：ファイルをした要求の回数（1回あたり）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4b.ownContextShare",
   "value": 57.2,
   "unit": "% (estimate)",
   "what": {
    "en": "T1–T4, Codex + Laydyne, E4 final: Codex's own context (first request) × requests, share of input tokens",
    "ja": "T1〜T4・Codex＋Laydyne、E4の最終：Codex自身の文脈（最初の要求）×要求回数が入力に占める割合"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4b.firstRequestTokens",
   "value": 21107,
   "unit": "tokens",
   "what": {
    "en": "T1–T4, Codex + Laydyne, E4 final: Codex's own context at the first request, mean",
    "ja": "T1〜T4・Codex＋Laydyne、E4の最終：最初の要求時点のCodex自身の文脈（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4base.inputTokensPerRun",
   "value": 555480,
   "unit": "tokens",
   "what": {
    "en": "T1–T4, Codex + Laydyne, E4 baseline (main bf7bfec7): input tokens per run, mean of 8 runs",
    "ja": "T1〜T4・Codex＋Laydyne、E4の変更前（main bf7bfec7）：1回あたりの入力トークン（8回の平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4base.requestsPerRun",
   "value": 11.88,
   "unit": "model requests",
   "what": {
    "en": "T1–T4, Codex + Laydyne, E4 baseline (main bf7bfec7): model requests per run, mean",
    "ja": "T1〜T4・Codex＋Laydyne、E4の変更前（main bf7bfec7）：1回あたりの要求回数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4base.resultCharsPerRun",
   "value": 140456,
   "unit": "characters",
   "what": {
    "en": "T1–T4, Codex + Laydyne, E4 baseline (main bf7bfec7): tool output characters the model saw per run, mean",
    "ja": "T1〜T4・Codex＋Laydyne、E4の変更前（main bf7bfec7）：1回あたりモデルが見た道具の出力の文字数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4base.doubledCharsPerRun",
   "value": 17650,
   "unit": "characters",
   "what": {
    "en": "T1–T4, Codex + Laydyne, E4 baseline (main bf7bfec7): characters printed twice per run, mean",
    "ja": "T1〜T4・Codex＋Laydyne、E4の変更前（main bf7bfec7）：1回あたり二重に表示された文字数（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4base.catalogShare",
   "value": 27.4,
   "unit": "% (estimate)",
   "what": {
    "en": "T1–T4, Codex + Laydyne, E4 baseline (main bf7bfec7): estimated share of input tokens from tool-list lookups re-sent with later requests (characters/4 × later requests)",
    "ja": "T1〜T4・Codex＋Laydyne、E4の変更前（main bf7bfec7）：道具一覧の参照の出力が後続の要求で再送された分の推定割合（文字数/4×後続の要求回数）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4base.catalogRequestsPerRun",
   "value": 2.5,
   "unit": "model requests",
   "what": {
    "en": "T1–T4, Codex + Laydyne, E4 baseline (main bf7bfec7): requests whose script did tool-list lookups, per run",
    "ja": "T1〜T4・Codex＋Laydyne、E4の変更前（main bf7bfec7）：道具一覧の参照をした要求の回数（1回あたり）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4base.readShare",
   "value": 12.7,
   "unit": "% (estimate)",
   "what": {
    "en": "T1–T4, Codex + Laydyne, E4 baseline (main bf7bfec7): estimated share of input tokens from reads (get_space and the like) re-sent with later requests (characters/4 × later requests)",
    "ja": "T1〜T4・Codex＋Laydyne、E4の変更前（main bf7bfec7）：読み取りの出力が後続の要求で再送された分の推定割合（文字数/4×後続の要求回数）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4base.readRequestsPerRun",
   "value": 2.25,
   "unit": "model requests",
   "what": {
    "en": "T1–T4, Codex + Laydyne, E4 baseline (main bf7bfec7): requests whose script did reads (get_space and the like), per run",
    "ja": "T1〜T4・Codex＋Laydyne、E4の変更前（main bf7bfec7）：読み取りをした要求の回数（1回あたり）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4base.fileShare",
   "value": 5.7,
   "unit": "% (estimate)",
   "what": {
    "en": "T1–T4, Codex + Laydyne, E4 baseline (main bf7bfec7): estimated share of input tokens from files (SVG, PDF, export) re-sent with later requests (characters/4 × later requests)",
    "ja": "T1〜T4・Codex＋Laydyne、E4の変更前（main bf7bfec7）：ファイルの出力が後続の要求で再送された分の推定割合（文字数/4×後続の要求回数）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4base.fileRequestsPerRun",
   "value": 2.88,
   "unit": "model requests",
   "what": {
    "en": "T1–T4, Codex + Laydyne, E4 baseline (main bf7bfec7): requests whose script did files (SVG, PDF, export), per run",
    "ja": "T1〜T4・Codex＋Laydyne、E4の変更前（main bf7bfec7）：ファイルをした要求の回数（1回あたり）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4base.ownContextShare",
   "value": 45.2,
   "unit": "% (estimate)",
   "what": {
    "en": "T1–T4, Codex + Laydyne, E4 baseline (main bf7bfec7): Codex's own context (first request) × requests, share of input tokens",
    "ja": "T1〜T4・Codex＋Laydyne、E4の変更前（main bf7bfec7）：Codex自身の文脈（最初の要求）×要求回数が入力に占める割合"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.all4.e4base.firstRequestTokens",
   "value": 21149,
   "unit": "tokens",
   "what": {
    "en": "T1–T4, Codex + Laydyne, E4 baseline (main bf7bfec7): Codex's own context at the first request, mean",
    "ja": "T1〜T4・Codex＋Laydyne、E4の変更前（main bf7bfec7）：最初の要求時点のCodex自身の文脈（平均）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-breakdown.jsonl"
   ]
  },
  {
   "id": "e4.surface.before.instructionsChars",
   "value": 9354,
   "unit": "characters",
   "what": {
    "en": "MCP server instructions (initialize instructions + link instructions), before E4 (bf7bfec7)",
    "ja": "MCPサーバーの指示文（initializeの指示文＋接続の指示文）の文字数、E4の変更前（bf7bfec7）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-surface.json"
   ]
  },
  {
   "id": "e4.surface.before.toolsListChars",
   "value": 315068,
   "unit": "characters",
   "what": {
    "en": "tools/list as JSON, before E4",
    "ja": "tools/list（JSON）の文字数、E4の変更前"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-surface.json"
   ]
  },
  {
   "id": "e4.surface.after.instructionsChars",
   "value": 11468,
   "unit": "characters",
   "what": {
    "en": "MCP server instructions (initialize instructions + link instructions), after E4 (b707f463)",
    "ja": "MCPサーバーの指示文（initializeの指示文＋接続の指示文）の文字数、E4の後（b707f463）"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-surface.json"
   ]
  },
  {
   "id": "e4.surface.after.toolsListChars",
   "value": 315841,
   "unit": "characters",
   "what": {
    "en": "tools/list as JSON, after E4",
    "ja": "tools/list（JSON）の文字数、E4の後"
   },
   "source": [
    "docs/verification/mcp-efficiency/data/e4-surface.json"
   ]
  },
  {
   "id": "e4.summaryChars.before",
   "value": 9001,
   "unit": "characters",
   "what": {
    "en": "get_space detail=summary of the blank start (10 × 8 m), JSON of the result, before E4",
    "ja": "空の開始状態（10 × 8 m）のget_space要約の結果のJSONの文字数（E4の変更前）"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4base-T1-B-r1/rollout.jsonl"
   ]
  },
  {
   "id": "e4.summaryChars.after",
   "value": 2596,
   "unit": "characters",
   "what": {
    "en": "get_space detail=summary of the blank start (10 × 8 m), JSON of the result, after E4",
    "ja": "空の開始状態（10 × 8 m）のget_space要約の結果のJSONの文字数（E4の後）"
   },
   "source": [
    "apps/studio/.wrangler/e4/runs/e4b-T1-B-r1/rollout.jsonl"
   ]
  }
 ]
}