{
  "type": "controlled-study-observation-ledger",
  "schemaVersion": 1,
  "ledgerVersion": "v1",
  "observations": [
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T13:44:22.523Z",
      "runId": "phase-2-baseline-run-01",
      "planHash": "949e7642f85ac2f598e006a0c47ccdcf7c0a4c44f8f2697a0e5c7993bfc0add4",
      "task": {
        "taskId": "consumer-02-documentation",
        "repositoryId": "consumer-02",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 1,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 79451,
        "responseBytes": 170,
        "stderrBytes": 0,
        "stdoutHash": "b1e2c75ee947d484ea44f70f741a6dc46547f2a3c40c5c61bf48f4fbbd29e9b1",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 208214,
        "outputTokens": 1490,
        "tokenMethod": "provider",
        "toolCalls": 6
      },
      "contextBytes": 1326,
      "evidenceIds": [],
      "round": "baseline-codex-2026-08-31",
      "measurements": {
        "cachedInputTokens": 172288,
        "reasoningOutputTokens": 420,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "b7a422dd86725010e4d67a56a6210f5dcf31879ab2d4446af0fa0db5835addaf",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T13:45:44.672Z",
      "runId": "phase-2-baseline-run-01",
      "planHash": "949e7642f85ac2f598e006a0c47ccdcf7c0a4c44f8f2697a0e5c7993bfc0add4",
      "task": {
        "taskId": "consumer-01-architecture",
        "repositoryId": "consumer-01",
        "category": "architecture",
        "scenarioId": "registry-assisted",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "registry-assisted",
        "agentId": "ecosystem-doc-bridge-corpus-scanner",
        "agentVersion": "v1.0.0",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 82142,
        "responseBytes": 171,
        "stderrBytes": 0,
        "stdoutHash": "c46c42c9bc15a6512459949c7edc2d8d13625ea93a92366529298e7afc859e1e",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 396917,
        "outputTokens": 3161,
        "tokenMethod": "provider",
        "toolCalls": 12,
        "errorCode": "token-budget"
      },
      "contextBytes": 1156,
      "evidenceIds": [],
      "round": "baseline-codex-2026-08-31",
      "measurements": {
        "cachedInputTokens": 350464,
        "reasoningOutputTokens": 726,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "edae4de48b24e666d96543c98f6c0c5cb687888ca63b1484bc5101918bedf5a0",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:01:18.815Z",
      "runId": "phase-2-reduced-sample-run-01",
      "planHash": "2b60c2f3ff9d042c77748e5cd4d1c615b6db349b0f23f37b2e4f2255202adf3a",
      "task": {
        "taskId": "consumer-01-discovery",
        "repositoryId": "consumer-01",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 80483,
        "responseBytes": 170,
        "stderrBytes": 0,
        "stdoutHash": "325d29523a5e2078a72f706410e0d0bf5bb25caa920710b60f052a4839193ae5",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 170029,
        "outputTokens": 1227,
        "tokenMethod": "provider",
        "toolCalls": 5
      },
      "contextBytes": 1125,
      "evidenceIds": [],
      "round": "reduced-sample-2026-08-31",
      "measurements": {
        "cachedInputTokens": 135680,
        "reasoningOutputTokens": 363,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "d46095646b4fb4a9c9510b7b0c3e85445ccbca053c1d3d368cf4f718ef91ad69",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:02:52.058Z",
      "runId": "phase-2-reduced-sample-run-01",
      "planHash": "2b60c2f3ff9d042c77748e5cd4d1c615b6db349b0f23f37b2e4f2255202adf3a",
      "task": {
        "taskId": "consumer-01-architecture",
        "repositoryId": "consumer-01",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 93227,
        "responseBytes": 171,
        "stderrBytes": 0,
        "stdoutHash": "1437cb5e2c0b43fc3e0d34989c95d1b872134804e347e1677ec6a4b5f025485e",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 347712,
        "outputTokens": 4107,
        "tokenMethod": "provider",
        "toolCalls": 8
      },
      "contextBytes": 1154,
      "evidenceIds": [],
      "round": "reduced-sample-2026-08-31",
      "measurements": {
        "cachedInputTokens": 290560,
        "reasoningOutputTokens": 1511,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "fd0893648792314122fdd85a3754319fddece7d995ced55c6843943abd1759ce",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:03:53.586Z",
      "runId": "phase-2-reduced-sample-run-01",
      "planHash": "2b60c2f3ff9d042c77748e5cd4d1c615b6db349b0f23f37b2e4f2255202adf3a",
      "task": {
        "taskId": "consumer-01-documentation",
        "repositoryId": "consumer-01",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 61525,
        "responseBytes": 170,
        "stderrBytes": 0,
        "stdoutHash": "10b7f7c128c581e7670c0de0e34b94dccb54f83a5bbefcb88ccb101e71d7c725",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 166526,
        "outputTokens": 1200,
        "tokenMethod": "provider",
        "toolCalls": 5
      },
      "contextBytes": 1326,
      "evidenceIds": [],
      "round": "reduced-sample-2026-08-31",
      "measurements": {
        "cachedInputTokens": 137728,
        "reasoningOutputTokens": 236,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "ec7aa24359756efea51ba555719fd23f02fc05772143c76f7bc6e753cf974e12",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:05:51.751Z",
      "runId": "phase-2-reduced-sample-run-01",
      "planHash": "2b60c2f3ff9d042c77748e5cd4d1c615b6db349b0f23f37b2e4f2255202adf3a",
      "task": {
        "taskId": "consumer-01-implementation",
        "repositoryId": "consumer-01",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 118162,
        "responseBytes": 173,
        "stderrBytes": 0,
        "stdoutHash": "94f5db1dfd4da84dc59b94f18888cd33a94d4b0af7f9773d0393ae67f6c352ff",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 1071772,
        "outputTokens": 4141,
        "tokenMethod": "provider",
        "toolCalls": 20,
        "errorCode": "token-budget"
      },
      "contextBytes": 1364,
      "evidenceIds": [],
      "round": "reduced-sample-2026-08-31",
      "measurements": {
        "cachedInputTokens": 990976,
        "reasoningOutputTokens": 1277,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "098183c52bec70a8a41e7d3194723ef2e878f04974ec5dce0a3e70d2a31fedcf",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:06:18.043Z",
      "runId": "phase-2-reduced-sample-run-01",
      "planHash": "2b60c2f3ff9d042c77748e5cd4d1c615b6db349b0f23f37b2e4f2255202adf3a",
      "task": {
        "taskId": "consumer-02-discovery",
        "repositoryId": "consumer-02",
        "category": "discovery",
        "scenarioId": "registry-assisted",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "registry-assisted",
        "agentId": "ecosystem-doc-bridge-corpus-scanner",
        "agentVersion": "v1.0.0",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 26290,
        "responseBytes": 167,
        "stderrBytes": 0,
        "stdoutHash": "dfe7a78c921090b026d4befe7b0f14dfc68deb90583b447aefacc02cf15c89b9",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 62904,
        "outputTokens": 370,
        "tokenMethod": "provider",
        "toolCalls": 2
      },
      "contextBytes": 1127,
      "evidenceIds": [],
      "round": "reduced-sample-2026-08-31",
      "measurements": {
        "cachedInputTokens": 50432,
        "reasoningOutputTokens": 150,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "e73665bdad372d4eff9707a7a9400282dbe296a88244ed8d5aea2f2ed3f6acd5",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:06:56.891Z",
      "runId": "phase-2-reduced-sample-run-01",
      "planHash": "2b60c2f3ff9d042c77748e5cd4d1c615b6db349b0f23f37b2e4f2255202adf3a",
      "task": {
        "taskId": "consumer-02-architecture",
        "repositoryId": "consumer-02",
        "category": "architecture",
        "scenarioId": "registry-assisted",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "registry-assisted",
        "agentId": "ecosystem-doc-bridge-corpus-scanner",
        "agentVersion": "v1.0.0",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 38846,
        "responseBytes": 170,
        "stderrBytes": 0,
        "stdoutHash": "c6a715a613640b06f5e18c969d00090f8c5b5bf8a380c7ddb4aa5cec8726a541",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 157624,
        "outputTokens": 1483,
        "tokenMethod": "provider",
        "toolCalls": 5
      },
      "contextBytes": 1156,
      "evidenceIds": [],
      "round": "reduced-sample-2026-08-31",
      "measurements": {
        "cachedInputTokens": 126464,
        "reasoningOutputTokens": 452,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "28ffd4e3cb89ee803ea216d55fd13c43d84828ecb71716537f447f89975af8db",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:08:10.449Z",
      "runId": "phase-2-reduced-sample-run-01",
      "planHash": "2b60c2f3ff9d042c77748e5cd4d1c615b6db349b0f23f37b2e4f2255202adf3a",
      "task": {
        "taskId": "consumer-02-documentation",
        "repositoryId": "consumer-02",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 73556,
        "responseBytes": 170,
        "stderrBytes": 0,
        "stdoutHash": "9725b5794ecc527fdda0f13a3aa79089f640c59e3c97441ea24a1f5c5ac3186b",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 258988,
        "outputTokens": 1451,
        "tokenMethod": "provider",
        "toolCalls": 7
      },
      "contextBytes": 1317,
      "evidenceIds": [],
      "round": "reduced-sample-2026-08-31",
      "measurements": {
        "cachedInputTokens": 218112,
        "reasoningOutputTokens": 271,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "52f3b9a97118ccf9981b517750923721c0b238a02f70a498d5be31b6f8fafab1",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:09:09.770Z",
      "runId": "phase-2-reduced-sample-run-01",
      "planHash": "2b60c2f3ff9d042c77748e5cd4d1c615b6db349b0f23f37b2e4f2255202adf3a",
      "task": {
        "taskId": "consumer-02-implementation",
        "repositoryId": "consumer-02",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 59319,
        "responseBytes": 171,
        "stderrBytes": 0,
        "stdoutHash": "ddd25c2a22857e08ffac5d9f6c8cf46c24e686e4d78f6dd270bd537ebf862374",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 509846,
        "outputTokens": 2153,
        "tokenMethod": "provider",
        "toolCalls": 12,
        "errorCode": "token-budget"
      },
      "contextBytes": 1355,
      "evidenceIds": [],
      "round": "reduced-sample-2026-08-31",
      "measurements": {
        "cachedInputTokens": 452352,
        "reasoningOutputTokens": 726,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "b770a9d72b1554c93daa8e26aa92001a3f6255ddd23a4229e58525fd89fcc72b",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:09:47.024Z",
      "runId": "phase-2-reduced-sample-run-01",
      "planHash": "2b60c2f3ff9d042c77748e5cd4d1c615b6db349b0f23f37b2e4f2255202adf3a",
      "task": {
        "taskId": "consumer-03-discovery",
        "repositoryId": "consumer-03",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 37252,
        "responseBytes": 167,
        "stderrBytes": 0,
        "stdoutHash": "76b24fd2b65417d7fc5eb2c95d60e694cf78e3076f176541a5ddb6e2734d4fb4",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 82252,
        "outputTokens": 628,
        "tokenMethod": "provider",
        "toolCalls": 2
      },
      "contextBytes": 1134,
      "evidenceIds": [],
      "round": "reduced-sample-2026-08-31",
      "measurements": {
        "cachedInputTokens": 60672,
        "reasoningOutputTokens": 360,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "d6cba36f85744c652d419059131a3a19fb4bb396eef53e9cc55717a4c5e8da92",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:10:43.472Z",
      "runId": "phase-2-reduced-sample-run-01",
      "planHash": "2b60c2f3ff9d042c77748e5cd4d1c615b6db349b0f23f37b2e4f2255202adf3a",
      "task": {
        "taskId": "consumer-03-architecture",
        "repositoryId": "consumer-03",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 56446,
        "responseBytes": 170,
        "stderrBytes": 0,
        "stdoutHash": "1075f651c4bb6ec47e5459ab71c4f22b9f46ceb0127a2d9e6b5660c574472e38",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 264992,
        "outputTokens": 2286,
        "tokenMethod": "provider",
        "toolCalls": 8
      },
      "contextBytes": 1163,
      "evidenceIds": [],
      "round": "reduced-sample-2026-08-31",
      "measurements": {
        "cachedInputTokens": 230144,
        "reasoningOutputTokens": 903,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "128828cf96ba714bad2456671ba6db4b2076b7758c53134c93a3cdcf08ebfbbd",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:11:39.724Z",
      "runId": "phase-2-reduced-sample-run-01",
      "planHash": "2b60c2f3ff9d042c77748e5cd4d1c615b6db349b0f23f37b2e4f2255202adf3a",
      "task": {
        "taskId": "consumer-03-documentation",
        "repositoryId": "consumer-03",
        "category": "documentation",
        "scenarioId": "registry-assisted",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "registry-assisted",
        "agentId": "ecosystem-doc-bridge-corpus-scanner",
        "agentVersion": "v1.0.0",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 56249,
        "responseBytes": 169,
        "stderrBytes": 0,
        "stdoutHash": "e0d53fcd02156ef54b6643c921de463e2965e37e767f93cf381b3c5e954877de",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 108066,
        "outputTokens": 983,
        "tokenMethod": "provider",
        "toolCalls": 3
      },
      "contextBytes": 1319,
      "evidenceIds": [],
      "round": "reduced-sample-2026-08-31",
      "measurements": {
        "cachedInputTokens": 103424,
        "reasoningOutputTokens": 162,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "f92633261d95f3c436dadb988b5cfa5e49ecc9d3bf88d124fc6217a12e0ed810",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:12:33.706Z",
      "runId": "phase-2-reduced-sample-run-01",
      "planHash": "2b60c2f3ff9d042c77748e5cd4d1c615b6db349b0f23f37b2e4f2255202adf3a",
      "task": {
        "taskId": "consumer-03-implementation",
        "repositoryId": "consumer-03",
        "category": "implementation",
        "scenarioId": "registry-assisted",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "registry-assisted",
        "agentId": "ecosystem-doc-bridge-corpus-scanner",
        "agentVersion": "v1.0.0",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 53979,
        "responseBytes": 171,
        "stderrBytes": 0,
        "stdoutHash": "bb360360f239ecb3cbe842cf01df4e90a46213b412f8f80f0ffdf008004eb54b",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 321044,
        "outputTokens": 2084,
        "tokenMethod": "provider",
        "toolCalls": 9
      },
      "contextBytes": 1357,
      "evidenceIds": [],
      "round": "reduced-sample-2026-08-31",
      "measurements": {
        "cachedInputTokens": 275968,
        "reasoningOutputTokens": 1034,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "47ac60248dd911cae64b74cc575857b8273a7269051e155ea49d85e5459e3bb9",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:13:52.187Z",
      "runId": "phase-2-reduced-sample-run-01",
      "planHash": "2b60c2f3ff9d042c77748e5cd4d1c615b6db349b0f23f37b2e4f2255202adf3a",
      "task": {
        "taskId": "consumer-04-discovery",
        "repositoryId": "consumer-04",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 78478,
        "responseBytes": 170,
        "stderrBytes": 0,
        "stdoutHash": "8577f4f7127018e596c74b5651c9baaeb805243aa305194c025ea04cebca793e",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 159807,
        "outputTokens": 1153,
        "tokenMethod": "provider",
        "toolCalls": 5
      },
      "contextBytes": 1125,
      "evidenceIds": [],
      "round": "reduced-sample-2026-08-31",
      "measurements": {
        "cachedInputTokens": 130560,
        "reasoningOutputTokens": 463,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "569a10d4be218864fe77f8305b6e38f5848223b3d53cbb74822a8a100e300e2f",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:15:34.603Z",
      "runId": "phase-2-reduced-sample-run-01",
      "planHash": "2b60c2f3ff9d042c77748e5cd4d1c615b6db349b0f23f37b2e4f2255202adf3a",
      "task": {
        "taskId": "consumer-04-architecture",
        "repositoryId": "consumer-04",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "invalid-output",
        "exitCode": 0,
        "signal": null,
        "durationMs": 102408,
        "responseBytes": 821,
        "stderrBytes": 0,
        "stdoutHash": "b858a7e2e2a6629f3a40fb7c4f60b3a6a05d7563eeffa8e8d2e2c62fdcf4283b",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "errorCode": "invalid-metrics"
      },
      "contextBytes": 1154,
      "evidenceIds": [],
      "round": "reduced-sample-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "d4f64deb677a02da60b2111d4a4508fc599c7425196fe6b759d998162772ba3e",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:17:38.619Z",
      "runId": "phase-2-reduced-sample-run-01",
      "planHash": "2b60c2f3ff9d042c77748e5cd4d1c615b6db349b0f23f37b2e4f2255202adf3a",
      "task": {
        "taskId": "consumer-04-documentation",
        "repositoryId": "consumer-04",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 124007,
        "responseBytes": 170,
        "stderrBytes": 0,
        "stdoutHash": "ba71673c6df83f1f755228f076360dd5f75e5cfe52b232520ac2f0dd924f8edd",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 247887,
        "outputTokens": 2364,
        "tokenMethod": "provider",
        "toolCalls": 7
      },
      "contextBytes": 1326,
      "evidenceIds": [],
      "round": "reduced-sample-2026-08-31",
      "measurements": {
        "cachedInputTokens": 216064,
        "reasoningOutputTokens": 697,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "3a284245cb58172dc45513beb4ec60159df025204dfe27d2c2d2fac8e27006f5",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:18:22.539Z",
      "runId": "phase-2-reduced-sample-run-01",
      "planHash": "2b60c2f3ff9d042c77748e5cd4d1c615b6db349b0f23f37b2e4f2255202adf3a",
      "task": {
        "taskId": "consumer-04-implementation",
        "repositoryId": "consumer-04",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 43916,
        "responseBytes": 170,
        "stderrBytes": 0,
        "stdoutHash": "21e83ad96ec8d7273922f429543eb2965ceecc61a029f687b0ed69621d178e58",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 127588,
        "outputTokens": 1585,
        "tokenMethod": "provider",
        "toolCalls": 4
      },
      "contextBytes": 1364,
      "evidenceIds": [],
      "round": "reduced-sample-2026-08-31",
      "measurements": {
        "cachedInputTokens": 102144,
        "reasoningOutputTokens": 733,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "4bcd1effb5b5f0de98a3d473c8273bf889a073fe44dfdf00694e9fedb069a58b",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:19:23.256Z",
      "runId": "phase-2-reduced-sample-run-01",
      "planHash": "2b60c2f3ff9d042c77748e5cd4d1c615b6db349b0f23f37b2e4f2255202adf3a",
      "task": {
        "taskId": "consumer-05-discovery",
        "repositoryId": "consumer-05",
        "category": "discovery",
        "scenarioId": "registry-assisted",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "registry-assisted",
        "agentId": "ecosystem-doc-bridge-corpus-scanner",
        "agentVersion": "v1.0.0",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 60706,
        "responseBytes": 169,
        "stderrBytes": 0,
        "stdoutHash": "70e6bb409bf6535ab2a88c7be9820e2a2f71f979471e492c2bbebc3475721743",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 118208,
        "outputTokens": 1272,
        "tokenMethod": "provider",
        "toolCalls": 3
      },
      "contextBytes": 1127,
      "evidenceIds": [],
      "round": "reduced-sample-2026-08-31",
      "measurements": {
        "cachedInputTokens": 82944,
        "reasoningOutputTokens": 287,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "b9d8fe2c61e5847427a03af22fc42d4a5bc16b5298545a1197b3ff4b30846831",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:20:33.038Z",
      "runId": "phase-2-reduced-sample-run-01",
      "planHash": "2b60c2f3ff9d042c77748e5cd4d1c615b6db349b0f23f37b2e4f2255202adf3a",
      "task": {
        "taskId": "consumer-05-architecture",
        "repositoryId": "consumer-05",
        "category": "architecture",
        "scenarioId": "registry-assisted",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "registry-assisted",
        "agentId": "ecosystem-doc-bridge-corpus-scanner",
        "agentVersion": "v1.0.0",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 69763,
        "responseBytes": 172,
        "stderrBytes": 0,
        "stdoutHash": "cfdf5ee7e4458aae838d32ebb5bea063533ba92977c2b8f6aee63bc504803de7",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 368980,
        "outputTokens": 2916,
        "tokenMethod": "provider",
        "toolCalls": 10
      },
      "contextBytes": 1156,
      "evidenceIds": [],
      "round": "reduced-sample-2026-08-31",
      "measurements": {
        "cachedInputTokens": 323840,
        "reasoningOutputTokens": 1124,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "d76a93b2d3cf6e26c14d74e035cbb6dc0bae147b0027f424c4dba60f70c7a7d3",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:22:14.847Z",
      "runId": "phase-2-reduced-sample-run-01",
      "planHash": "2b60c2f3ff9d042c77748e5cd4d1c615b6db349b0f23f37b2e4f2255202adf3a",
      "task": {
        "taskId": "consumer-05-documentation",
        "repositoryId": "consumer-05",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 101806,
        "responseBytes": 170,
        "stderrBytes": 0,
        "stdoutHash": "85a0766f7948585c8b806406f1f65f7ffdbc567e1bfe27b81b18ef33200d8ed4",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 208958,
        "outputTokens": 1551,
        "tokenMethod": "provider",
        "toolCalls": 7
      },
      "contextBytes": 1317,
      "evidenceIds": [],
      "round": "reduced-sample-2026-08-31",
      "measurements": {
        "cachedInputTokens": 180224,
        "reasoningOutputTokens": 476,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "74b4e16648fce5702e3a33bc71287132060030085129781cb06d225dd402dd2e",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:24:02.783Z",
      "runId": "phase-2-reduced-sample-run-01",
      "planHash": "2b60c2f3ff9d042c77748e5cd4d1c615b6db349b0f23f37b2e4f2255202adf3a",
      "task": {
        "taskId": "consumer-05-implementation",
        "repositoryId": "consumer-05",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 107930,
        "responseBytes": 172,
        "stderrBytes": 0,
        "stdoutHash": "62713b15b3c7b713c868be23d7eb1d37128d61a11642f05e78ddb49652ae0de1",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 503869,
        "outputTokens": 3354,
        "tokenMethod": "provider",
        "toolCalls": 10,
        "errorCode": "token-budget"
      },
      "contextBytes": 1355,
      "evidenceIds": [],
      "round": "reduced-sample-2026-08-31",
      "measurements": {
        "cachedInputTokens": 442368,
        "reasoningOutputTokens": 1671,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "e15bb687b0ca4dc2768412ec31a0d34454f80443bac710b13bc79574f342a19c",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:25:13.344Z",
      "runId": "phase-2-reduced-sample-run-01",
      "planHash": "2b60c2f3ff9d042c77748e5cd4d1c615b6db349b0f23f37b2e4f2255202adf3a",
      "task": {
        "taskId": "consumer-06-discovery",
        "repositoryId": "consumer-06",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "invalid-output",
        "exitCode": 0,
        "signal": null,
        "durationMs": 70554,
        "responseBytes": 530,
        "stderrBytes": 0,
        "stdoutHash": "a3c5442b82f504048363feaafe6aec73f6dd9b414461a1890cadd0dcbd5dca97",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "errorCode": "invalid-metrics"
      },
      "contextBytes": 1134,
      "evidenceIds": [],
      "round": "reduced-sample-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "a5fe7d36088419d13fa1970cbbb1fe8dcb6b84be3c7df4779b83bc1fa816f769",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:26:36.705Z",
      "runId": "phase-2-reduced-sample-run-01",
      "planHash": "2b60c2f3ff9d042c77748e5cd4d1c615b6db349b0f23f37b2e4f2255202adf3a",
      "task": {
        "taskId": "consumer-06-architecture",
        "repositoryId": "consumer-06",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 83354,
        "responseBytes": 426,
        "stderrBytes": 0,
        "stdoutHash": "0a07b4a43960edc60311ef86940b50c2564203f01050231377339fd3653811a0",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 479463,
        "outputTokens": 2786,
        "tokenMethod": "provider",
        "toolCalls": 10,
        "errorCode": "token-budget"
      },
      "contextBytes": 1163,
      "evidenceIds": [
        "architecture-evidence"
      ],
      "round": "reduced-sample-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "measurements": {
        "architectureComponents": 8,
        "directionalRelations": 11,
        "attentionPoints": 1,
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 408832,
        "reasoningOutputTokens": 1394,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "e01ae3ebf141e416ec30149c07716163de2acdd6614dbcf85e3df324e4895b35",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:27:53.152Z",
      "runId": "phase-2-reduced-sample-run-01",
      "planHash": "2b60c2f3ff9d042c77748e5cd4d1c615b6db349b0f23f37b2e4f2255202adf3a",
      "task": {
        "taskId": "consumer-06-documentation",
        "repositoryId": "consumer-06",
        "category": "documentation",
        "scenarioId": "registry-assisted",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "registry-assisted",
        "agentId": "ecosystem-doc-bridge-corpus-scanner",
        "agentVersion": "v1.0.0",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 76431,
        "responseBytes": 170,
        "stderrBytes": 0,
        "stdoutHash": "271812ac11db96eb273ae3a1a046df6cadf71e64bc4c656e8dd6f1bd36abc93b",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 295545,
        "outputTokens": 1550,
        "tokenMethod": "provider",
        "toolCalls": 8
      },
      "contextBytes": 1319,
      "evidenceIds": [],
      "round": "reduced-sample-2026-08-31",
      "measurements": {
        "cachedInputTokens": 241408,
        "reasoningOutputTokens": 323,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "2dff238dc4e0b9175a25dc67dc67e4b57cac3787cb9af129c484d02de4fb2c7b",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:28:50.280Z",
      "runId": "phase-2-reduced-sample-run-01",
      "planHash": "2b60c2f3ff9d042c77748e5cd4d1c615b6db349b0f23f37b2e4f2255202adf3a",
      "task": {
        "taskId": "consumer-06-implementation",
        "repositoryId": "consumer-06",
        "category": "implementation",
        "scenarioId": "registry-assisted",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "registry-assisted",
        "agentId": "ecosystem-doc-bridge-corpus-scanner",
        "agentVersion": "v1.0.0",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 57122,
        "responseBytes": 171,
        "stderrBytes": 0,
        "stdoutHash": "b59f306820e1b927b31e2dc760a565c76a262c4008b369a6d433b959e9203c91",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 294599,
        "outputTokens": 2262,
        "tokenMethod": "provider",
        "toolCalls": 7
      },
      "contextBytes": 1357,
      "evidenceIds": [],
      "round": "reduced-sample-2026-08-31",
      "measurements": {
        "cachedInputTokens": 246784,
        "reasoningOutputTokens": 1314,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "9f51d7430e9a86f45c4f423e806c3d657d48a477ff2f000bf6ca559044bafd38",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:46:10.219Z",
      "runId": "phase-3-structured-provider-run-01",
      "planHash": "3122d65dbd7dd191e2ae8e3ad71f5487e6df759f421baaa5238e9e87fb851a8e",
      "task": {
        "taskId": "consumer-01-discovery",
        "repositoryId": "consumer-01",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 1,
        "signal": null,
        "durationMs": 9510,
        "responseBytes": 0,
        "stderrBytes": 0,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1397,
      "evidenceIds": [],
      "round": "structured-provider-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "3d8e039643fbc8b127cdfc539bc4b2f66bbe7525ae82bb64c419bce4af29770e",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:46:13.895Z",
      "runId": "phase-3-structured-provider-run-01",
      "planHash": "3122d65dbd7dd191e2ae8e3ad71f5487e6df759f421baaa5238e9e87fb851a8e",
      "task": {
        "taskId": "consumer-01-architecture",
        "repositoryId": "consumer-01",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 1,
        "signal": null,
        "durationMs": 3671,
        "responseBytes": 0,
        "stderrBytes": 0,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1426,
      "evidenceIds": [],
      "round": "structured-provider-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "fa3289799ae29a045720a8808cf8b3f12207bea5887cae5c9462d443f89008d7",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:46:16.758Z",
      "runId": "phase-3-structured-provider-run-01",
      "planHash": "3122d65dbd7dd191e2ae8e3ad71f5487e6df759f421baaa5238e9e87fb851a8e",
      "task": {
        "taskId": "consumer-01-documentation",
        "repositoryId": "consumer-01",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 1,
        "signal": null,
        "durationMs": 2858,
        "responseBytes": 0,
        "stderrBytes": 0,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1598,
      "evidenceIds": [],
      "round": "structured-provider-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "7a07486f0a1c48fba475d762930de6884a96fe0e52514ab8672a9b43edce2d7b",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:46:23.825Z",
      "runId": "phase-3-structured-provider-run-01",
      "planHash": "3122d65dbd7dd191e2ae8e3ad71f5487e6df759f421baaa5238e9e87fb851a8e",
      "task": {
        "taskId": "consumer-01-implementation",
        "repositoryId": "consumer-01",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 1,
        "signal": null,
        "durationMs": 7064,
        "responseBytes": 0,
        "stderrBytes": 0,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1636,
      "evidenceIds": [],
      "round": "structured-provider-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "d541e10d9f85e08667cf8f17eb4284e09a976e0f7f9d764371448a12c3396311",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:46:27.112Z",
      "runId": "phase-3-structured-provider-run-01",
      "planHash": "3122d65dbd7dd191e2ae8e3ad71f5487e6df759f421baaa5238e9e87fb851a8e",
      "task": {
        "taskId": "consumer-02-discovery",
        "repositoryId": "consumer-02",
        "category": "discovery",
        "scenarioId": "registry-assisted",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "registry-assisted",
        "agentId": "ecosystem-doc-bridge-corpus-scanner",
        "agentVersion": "v1.0.0",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 1,
        "signal": null,
        "durationMs": 3282,
        "responseBytes": 0,
        "stderrBytes": 0,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1399,
      "evidenceIds": [],
      "round": "structured-provider-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "90ddae1ec8b47660cfaa28054c3d58a9b0c2666e2433a26a4d56b72b2dbd0041",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:46:29.804Z",
      "runId": "phase-3-structured-provider-run-01",
      "planHash": "3122d65dbd7dd191e2ae8e3ad71f5487e6df759f421baaa5238e9e87fb851a8e",
      "task": {
        "taskId": "consumer-02-architecture",
        "repositoryId": "consumer-02",
        "category": "architecture",
        "scenarioId": "registry-assisted",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "registry-assisted",
        "agentId": "ecosystem-doc-bridge-corpus-scanner",
        "agentVersion": "v1.0.0",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 1,
        "signal": null,
        "durationMs": 2688,
        "responseBytes": 0,
        "stderrBytes": 0,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1428,
      "evidenceIds": [],
      "round": "structured-provider-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "316c6f8633f25cdbd91afc6497dddd6741f8c0184e94a585f424dbf8b7e7b5ba",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:46:32.654Z",
      "runId": "phase-3-structured-provider-run-01",
      "planHash": "3122d65dbd7dd191e2ae8e3ad71f5487e6df759f421baaa5238e9e87fb851a8e",
      "task": {
        "taskId": "consumer-02-documentation",
        "repositoryId": "consumer-02",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 1,
        "signal": null,
        "durationMs": 2845,
        "responseBytes": 0,
        "stderrBytes": 0,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1589,
      "evidenceIds": [],
      "round": "structured-provider-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "f7c35cb7de521900d825431837711cfb98d88cefcd241d840b5ef7ad77b8e434",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:46:35.887Z",
      "runId": "phase-3-structured-provider-run-01",
      "planHash": "3122d65dbd7dd191e2ae8e3ad71f5487e6df759f421baaa5238e9e87fb851a8e",
      "task": {
        "taskId": "consumer-02-implementation",
        "repositoryId": "consumer-02",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 1,
        "signal": null,
        "durationMs": 3229,
        "responseBytes": 0,
        "stderrBytes": 0,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1627,
      "evidenceIds": [],
      "round": "structured-provider-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "3ca7adc3c956e654c799b4a6de94e777e994191fa23d77e09ae1f2845969a6f8",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:46:38.756Z",
      "runId": "phase-3-structured-provider-run-01",
      "planHash": "3122d65dbd7dd191e2ae8e3ad71f5487e6df759f421baaa5238e9e87fb851a8e",
      "task": {
        "taskId": "consumer-03-discovery",
        "repositoryId": "consumer-03",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 1,
        "signal": null,
        "durationMs": 2867,
        "responseBytes": 0,
        "stderrBytes": 0,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1406,
      "evidenceIds": [],
      "round": "structured-provider-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "dc11fd883e1b831c92d505696a6670b7f7458a4544425c44aacb996ced98f5aa",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:46:41.430Z",
      "runId": "phase-3-structured-provider-run-01",
      "planHash": "3122d65dbd7dd191e2ae8e3ad71f5487e6df759f421baaa5238e9e87fb851a8e",
      "task": {
        "taskId": "consumer-03-architecture",
        "repositoryId": "consumer-03",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 1,
        "signal": null,
        "durationMs": 2670,
        "responseBytes": 0,
        "stderrBytes": 0,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1435,
      "evidenceIds": [],
      "round": "structured-provider-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "2d4d29f4186c2db9d28b0b29c9be8859fa0c3fac31d1c654f5bccf3ab367cfa6",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:46:44.399Z",
      "runId": "phase-3-structured-provider-run-01",
      "planHash": "3122d65dbd7dd191e2ae8e3ad71f5487e6df759f421baaa5238e9e87fb851a8e",
      "task": {
        "taskId": "consumer-03-documentation",
        "repositoryId": "consumer-03",
        "category": "documentation",
        "scenarioId": "registry-assisted",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "registry-assisted",
        "agentId": "ecosystem-doc-bridge-corpus-scanner",
        "agentVersion": "v1.0.0",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 1,
        "signal": null,
        "durationMs": 2965,
        "responseBytes": 0,
        "stderrBytes": 0,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1591,
      "evidenceIds": [],
      "round": "structured-provider-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "a9ce66a38c7e7d85bfc4e81aa326a1c4c306dc8e687e34e26d046a33393f09b4",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:46:47.151Z",
      "runId": "phase-3-structured-provider-run-01",
      "planHash": "3122d65dbd7dd191e2ae8e3ad71f5487e6df759f421baaa5238e9e87fb851a8e",
      "task": {
        "taskId": "consumer-03-implementation",
        "repositoryId": "consumer-03",
        "category": "implementation",
        "scenarioId": "registry-assisted",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "registry-assisted",
        "agentId": "ecosystem-doc-bridge-corpus-scanner",
        "agentVersion": "v1.0.0",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 1,
        "signal": null,
        "durationMs": 2748,
        "responseBytes": 0,
        "stderrBytes": 0,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1629,
      "evidenceIds": [],
      "round": "structured-provider-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "e19f68f9418c1f6eadd8e4c26285293cbc99e7e777044cd8f90b081343df1527",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:46:49.725Z",
      "runId": "phase-3-structured-provider-run-01",
      "planHash": "3122d65dbd7dd191e2ae8e3ad71f5487e6df759f421baaa5238e9e87fb851a8e",
      "task": {
        "taskId": "consumer-04-discovery",
        "repositoryId": "consumer-04",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 1,
        "signal": null,
        "durationMs": 2571,
        "responseBytes": 0,
        "stderrBytes": 0,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1397,
      "evidenceIds": [],
      "round": "structured-provider-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "a9fcbe33e93ada168f97e0d808f325dafa35dc3a0a161e96e3f692d70fb96104",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:46:52.626Z",
      "runId": "phase-3-structured-provider-run-01",
      "planHash": "3122d65dbd7dd191e2ae8e3ad71f5487e6df759f421baaa5238e9e87fb851a8e",
      "task": {
        "taskId": "consumer-04-architecture",
        "repositoryId": "consumer-04",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 1,
        "signal": null,
        "durationMs": 2898,
        "responseBytes": 0,
        "stderrBytes": 0,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1426,
      "evidenceIds": [],
      "round": "structured-provider-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "c41b171df3e59d95d80060eec49878c4c24488583207142fbcdc61eeb704d877",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:46:55.194Z",
      "runId": "phase-3-structured-provider-run-01",
      "planHash": "3122d65dbd7dd191e2ae8e3ad71f5487e6df759f421baaa5238e9e87fb851a8e",
      "task": {
        "taskId": "consumer-04-documentation",
        "repositoryId": "consumer-04",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 1,
        "signal": null,
        "durationMs": 2565,
        "responseBytes": 0,
        "stderrBytes": 0,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1598,
      "evidenceIds": [],
      "round": "structured-provider-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "0996ef292b2d6f1437a59e12a7bc3eadb8ba291702cb5a8b4f617332b455eeb6",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:47:02.526Z",
      "runId": "phase-3-structured-provider-run-01",
      "planHash": "3122d65dbd7dd191e2ae8e3ad71f5487e6df759f421baaa5238e9e87fb851a8e",
      "task": {
        "taskId": "consumer-04-implementation",
        "repositoryId": "consumer-04",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 1,
        "signal": null,
        "durationMs": 7328,
        "responseBytes": 0,
        "stderrBytes": 0,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1636,
      "evidenceIds": [],
      "round": "structured-provider-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "bc7d8785b411fa340cd440856931ca9ab23746940b691ca5dd8ccf45f1fe2b84",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:47:05.290Z",
      "runId": "phase-3-structured-provider-run-01",
      "planHash": "3122d65dbd7dd191e2ae8e3ad71f5487e6df759f421baaa5238e9e87fb851a8e",
      "task": {
        "taskId": "consumer-05-discovery",
        "repositoryId": "consumer-05",
        "category": "discovery",
        "scenarioId": "registry-assisted",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "registry-assisted",
        "agentId": "ecosystem-doc-bridge-corpus-scanner",
        "agentVersion": "v1.0.0",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 1,
        "signal": null,
        "durationMs": 2760,
        "responseBytes": 0,
        "stderrBytes": 0,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1399,
      "evidenceIds": [],
      "round": "structured-provider-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "684b367a24cc87c29c29925dfc15ee68f9ced7bc57cca2b39c2ca889bc93ca39",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:47:08.118Z",
      "runId": "phase-3-structured-provider-run-01",
      "planHash": "3122d65dbd7dd191e2ae8e3ad71f5487e6df759f421baaa5238e9e87fb851a8e",
      "task": {
        "taskId": "consumer-05-architecture",
        "repositoryId": "consumer-05",
        "category": "architecture",
        "scenarioId": "registry-assisted",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "registry-assisted",
        "agentId": "ecosystem-doc-bridge-corpus-scanner",
        "agentVersion": "v1.0.0",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 1,
        "signal": null,
        "durationMs": 2824,
        "responseBytes": 0,
        "stderrBytes": 0,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1428,
      "evidenceIds": [],
      "round": "structured-provider-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "20809ef4e3906161f7c3376d9311c7fb433abd21bebe656a03deace2a3854c76",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:47:10.669Z",
      "runId": "phase-3-structured-provider-run-01",
      "planHash": "3122d65dbd7dd191e2ae8e3ad71f5487e6df759f421baaa5238e9e87fb851a8e",
      "task": {
        "taskId": "consumer-05-documentation",
        "repositoryId": "consumer-05",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 1,
        "signal": null,
        "durationMs": 2547,
        "responseBytes": 0,
        "stderrBytes": 0,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1589,
      "evidenceIds": [],
      "round": "structured-provider-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "5535ff8af6715947eacd8801ff4a17fd72c15704a1af33ddba90d82243dc9997",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T14:47:13.508Z",
      "runId": "phase-3-structured-provider-run-01",
      "planHash": "3122d65dbd7dd191e2ae8e3ad71f5487e6df759f421baaa5238e9e87fb851a8e",
      "task": {
        "taskId": "consumer-05-implementation",
        "repositoryId": "consumer-05",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 1,
        "signal": null,
        "durationMs": 2836,
        "responseBytes": 0,
        "stderrBytes": 0,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1627,
      "evidenceIds": [],
      "round": "structured-provider-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "85643742545d961e30f17822e016cbb9fb47e81d4da32274e0e8746cec6556c4",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T15:06:41.365Z",
      "runId": "phase-4-structured-provider-recovery-01",
      "planHash": "64dd9541eb736b0290cd54e374b60de470f3c2d700f0c32bd683ac44b10f5124",
      "task": {
        "taskId": "consumer-01-discovery",
        "repositoryId": "consumer-01",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "invalid-output",
        "exitCode": 0,
        "signal": null,
        "durationMs": 53071,
        "responseBytes": 554,
        "stderrBytes": 0,
        "stdoutHash": "f03d0da7c3de0b167f8bc3ce082115f7413a1b476f0bf80d9ed3ae9efc789c10",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "errorCode": "invalid-metrics"
      },
      "contextBytes": 1427,
      "evidenceIds": [],
      "round": "structured-provider-recovery-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "bc2b8ead1ccd15e9fc5da9d81ff5578b84260ae319564c6bdec5e5c02dad3e5c",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T15:07:36.106Z",
      "runId": "phase-4-structured-provider-recovery-01",
      "planHash": "64dd9541eb736b0290cd54e374b60de470f3c2d700f0c32bd683ac44b10f5124",
      "task": {
        "taskId": "consumer-01-architecture",
        "repositoryId": "consumer-01",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "invalid-output",
        "exitCode": 0,
        "signal": null,
        "durationMs": 54732,
        "responseBytes": 1136,
        "stderrBytes": 0,
        "stdoutHash": "6f9188248204efdad7c39334f7fd77e2f71154f2a8bd0689b63170c7d7e0f8aa",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "errorCode": "invalid-metrics"
      },
      "contextBytes": 1456,
      "evidenceIds": [],
      "round": "structured-provider-recovery-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "628468634ca15200f922ca26073f664180b32cd16fd17600c3e09625ea640732",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T15:08:38.360Z",
      "runId": "phase-4-structured-provider-recovery-01",
      "planHash": "64dd9541eb736b0290cd54e374b60de470f3c2d700f0c32bd683ac44b10f5124",
      "task": {
        "taskId": "consumer-01-documentation",
        "repositoryId": "consumer-01",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "invalid-output",
        "exitCode": 0,
        "signal": null,
        "durationMs": 62246,
        "responseBytes": 1208,
        "stderrBytes": 0,
        "stdoutHash": "5e237dac71e03a986e8b01aad41aa0f4b612b9d0a0e8f47186968b36ab8a714b",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "errorCode": "invalid-metrics"
      },
      "contextBytes": 1628,
      "evidenceIds": [],
      "round": "structured-provider-recovery-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "4f8d5239991a64cc7bcdbfb26451440f868d9ee46b86da7c4e013ae809558d65",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T15:09:14.573Z",
      "runId": "phase-4-structured-provider-recovery-01",
      "planHash": "64dd9541eb736b0290cd54e374b60de470f3c2d700f0c32bd683ac44b10f5124",
      "task": {
        "taskId": "consumer-01-implementation",
        "repositoryId": "consumer-01",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "invalid-output",
        "exitCode": 0,
        "signal": null,
        "durationMs": 36204,
        "responseBytes": 447,
        "stderrBytes": 0,
        "stdoutHash": "c7f0e3c9bb4901bb3432a99dab847e948c9e12b7ea7594d95da153924c5bbb76",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "errorCode": "invalid-metrics"
      },
      "contextBytes": 1666,
      "evidenceIds": [],
      "round": "structured-provider-recovery-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "42635381d59c41c95a37c2763222eadcff640061994d4b72b68a7da80d2f2e0b",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T15:51:56.613Z",
      "runId": "phase-5-structured-provider-recovery-01",
      "planHash": "be4bb39c5cd42038c1f004671fd282c586903f498da615b67675107cd83fdb01",
      "task": {
        "taskId": "consumer-01-discovery",
        "repositoryId": "consumer-01",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 48209,
        "responseBytes": 762,
        "stderrBytes": 0,
        "stdoutHash": "b444f6d2ce4ee865fe415f6245bc4f189dfde75c5dd8c59cc02330ca8b50780f",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 111494,
        "outputTokens": 907,
        "tokenMethod": "provider",
        "toolCalls": 3
      },
      "contextBytes": 1427,
      "evidenceIds": [
        "entrypoint-evidence:package.json",
        "entrypoint-evidence:pnpm-workspace.yaml",
        "ownership-evidence:AGENTS.md",
        "canonical-docs-evidence:apps/docs-next/README.md",
        "canonical-docs-evidence:docs/architecture/adrs/0007-docs-platform-fumadocs.md",
        "discovery-evidence:sha256:3b5a3981abca11e57dc9790d0d1f949f5647d59bcf323781079c24d274709dce"
      ],
      "round": "structured-provider-recovery-2026-08-31",
      "taskOutcome": "success",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 1,
      "measurements": {
        "acceptanceChecksPassed": 1,
        "entrypointEvidencePaths": 2,
        "ownershipEvidencePaths": 1,
        "canonicalDocumentationPaths": 2,
        "filesModified": 0,
        "cachedInputTokens": 70912,
        "reasoningOutputTokens": 305,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "e842cdf4ae058e3d6df84dea8aa2e3e4fea82669c4ab86a15caf1814f6b1e302",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T15:52:23.969Z",
      "runId": "phase-5-structured-provider-recovery-01",
      "planHash": "be4bb39c5cd42038c1f004671fd282c586903f498da615b67675107cd83fdb01",
      "task": {
        "taskId": "consumer-01-architecture",
        "repositoryId": "consumer-01",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 27348,
        "responseBytes": 424,
        "stderrBytes": 0,
        "stdoutHash": "f2b22316b467b5b87ffd60d919c20172046c20001ee1dae1977e5e1be0d93659",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 112680,
        "outputTokens": 1013,
        "tokenMethod": "provider",
        "toolCalls": 3
      },
      "contextBytes": 1456,
      "evidenceIds": [
        "architecture-evidence"
      ],
      "round": "structured-provider-recovery-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "components_identified": 14,
        "directional_relationships_observed": 20,
        "attention_points_identified": 1,
        "cachedInputTokens": 75008,
        "reasoningOutputTokens": 406,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "2671f0e06ba10ea00a30cf65f39b3fbf2e77af68666bf4676c85d61756c88e8c",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T15:53:11.532Z",
      "runId": "phase-5-structured-provider-recovery-01",
      "planHash": "be4bb39c5cd42038c1f004671fd282c586903f498da615b67675107cd83fdb01",
      "task": {
        "taskId": "consumer-01-documentation",
        "repositoryId": "consumer-01",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 47556,
        "responseBytes": 900,
        "stderrBytes": 0,
        "stdoutHash": "2558edfca6801b104834bc98200ab96818a066ef51a6d8c185babdb515012f9a",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 147064,
        "outputTokens": 968,
        "tokenMethod": "provider",
        "toolCalls": 4
      },
      "contextBytes": 1628,
      "evidenceIds": [
        "documentation-evidence:stale:apps/docs-next/content/docs/reference/index.mdx:15 claims 69 published recipes; ecosystem-claims.json recipes claim reports 71 from scripts/compute-stats.mjs; recommended action: regenerate or update the reference index claim",
        "review-limitation:high-confidence direct numeric contradiction; ak-docs audit documentation --json could not execute because ak-docs was unavailable directly and pnpm exec required a sandbox-prohibited temporary write"
      ],
      "round": "structured-provider-recovery-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 1,
      "measurements": {
        "documentationFindings": 1,
        "documentedRecipeCount": 69,
        "generatedRecipeCount": 71,
        "confidence": 0.99,
        "acceptanceChecksPassed": 0,
        "cachedInputTokens": 102400,
        "reasoningOutputTokens": 357,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "b9c7027521c53f0d2723058df2d002566528aef20b14b7fcd7b2b312e17bc651",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T15:54:30.072Z",
      "runId": "phase-5-structured-provider-recovery-01",
      "planHash": "be4bb39c5cd42038c1f004671fd282c586903f498da615b67675107cd83fdb01",
      "task": {
        "taskId": "consumer-01-implementation",
        "repositoryId": "consumer-01",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 78531,
        "responseBytes": 784,
        "stderrBytes": 0,
        "stdoutHash": "b421dcd0aa0b5cfef1634c22850931a91278c5ffb1b2838691facaad06be4617",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 538325,
        "outputTokens": 3036,
        "tokenMethod": "provider",
        "toolCalls": 11,
        "errorCode": "token-budget"
      },
      "contextBytes": 1666,
      "evidenceIds": [
        "patch-evidence:docs/studies/v1-readiness-audit.md:29-31",
        "gap-evidence:docs/STABILITY.md:34-61",
        "proposal:mark-the-audit-finding-as-historical-and-align-its-package-count/tier-claims-with-the-current-stability-map;documentation-only;no-source-changes",
        "verification-plan:ak-docs-check-json-after-approval",
        "blocked:ak-docs-command-unavailable;direct-run-blocked-by-read-only-filesystem"
      ],
      "round": "structured-provider-recovery-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "proposed_document_targets": 1,
        "source_files_to_change": 0,
        "verification_command_exit_code": 2,
        "cachedInputTokens": 461056,
        "reasoningOutputTokens": 1325,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "9561afeee5c8cdf777b391dbe041b2eb4e332502d792cef66d66904f813160da",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T15:54:55.062Z",
      "runId": "phase-5-structured-provider-recovery-01",
      "planHash": "be4bb39c5cd42038c1f004671fd282c586903f498da615b67675107cd83fdb01",
      "task": {
        "taskId": "consumer-02-discovery",
        "repositoryId": "consumer-02",
        "category": "discovery",
        "scenarioId": "registry-assisted",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "registry-assisted",
        "agentId": "ecosystem-doc-bridge-corpus-scanner",
        "agentVersion": "v1.0.0",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 24982,
        "responseBytes": 375,
        "stderrBytes": 0,
        "stdoutHash": "cc08cb2252cd2f199168f9457a2fbb0c76963d83627126d0b0936acffa511278",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 65109,
        "outputTokens": 413,
        "tokenMethod": "provider",
        "toolCalls": 2
      },
      "contextBytes": 1429,
      "evidenceIds": [
        "entrypoint-evidence"
      ],
      "round": "structured-provider-recovery-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "low",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksAttempted": 1,
        "cachedInputTokens": 40448,
        "reasoningOutputTokens": 152,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "026467e2f1019fec0a7afd84b0c4bab67427c6464c721c2d43ebf18cffa4f141",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T15:56:07.039Z",
      "runId": "phase-5-structured-provider-recovery-01",
      "planHash": "be4bb39c5cd42038c1f004671fd282c586903f498da615b67675107cd83fdb01",
      "task": {
        "taskId": "consumer-02-architecture",
        "repositoryId": "consumer-02",
        "category": "architecture",
        "scenarioId": "registry-assisted",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "registry-assisted",
        "agentId": "ecosystem-doc-bridge-corpus-scanner",
        "agentVersion": "v1.0.0",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 71968,
        "responseBytes": 1280,
        "stderrBytes": 0,
        "stdoutHash": "3a524d8487b25eaa6e41f551321750cdd709ebcf7698ec3d426c636a24d5364c",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 220710,
        "outputTokens": 3181,
        "tokenMethod": "provider",
        "toolCalls": 18
      },
      "contextBytes": 1458,
      "evidenceIds": [
        "docs/architecture/overview.md:AgentsKit upstream primitives -> AgentsKit Chat application layer",
        "docs/architecture/overview.md:Core -> Protocol -> Server -> native renderers",
        "pnpm-workspace.yaml:TypeScript workspace monorepo",
        "packages/*/package.json:renderer/server/CLI/devtools dependency directions",
        "packages/chat/src:framework-neutral definitions and registries",
        "packages/protocol/src:versioned schemas and encode/decode boundaries",
        "packages/server/src:Web-standard handler over chat and protocol",
        "docs/for-agents/packages/chat.md:upstream-first and framework-neutral boundary",
        "Review attention: align observed @agentskit/core versions across package peer/dev dependencies and the upstream adoption matrix",
        "ak-docs map --json:blocked by EPERM creating .doc-bridge/workflow/.lock; existing .doc-bridge/index.json supplied artifact map evidence"
      ],
      "round": "structured-provider-recovery-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "bounded_architecture_component_groups": 8,
        "observed_directional_package_relations": 26,
        "acceptance_command_exit_code": 2,
        "cachedInputTokens": 159488,
        "reasoningOutputTokens": 1123,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "230cbcd5cf9beaf14731f23f41af95bbf28388953104960ce2d9ae370499f37e",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T15:57:21.807Z",
      "runId": "phase-5-structured-provider-recovery-01",
      "planHash": "be4bb39c5cd42038c1f004671fd282c586903f498da615b67675107cd83fdb01",
      "task": {
        "taskId": "consumer-02-documentation",
        "repositoryId": "consumer-02",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "invalid-output",
        "exitCode": 0,
        "signal": null,
        "durationMs": 74758,
        "responseBytes": 1464,
        "stderrBytes": 0,
        "stdoutHash": "7825ee0520dcbf200240353a53ad1839c9035ea0ca57e01dbbc84a57c9922ad3",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "errorCode": "invalid-metrics"
      },
      "contextBytes": 1619,
      "evidenceIds": [],
      "round": "structured-provider-recovery-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "0ac0b7c7f93de21f673a64bd545ee35389da00806c5193ee191ecfe27d448ba3",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T15:58:30.450Z",
      "runId": "phase-5-structured-provider-recovery-01",
      "planHash": "be4bb39c5cd42038c1f004671fd282c586903f498da615b67675107cd83fdb01",
      "task": {
        "taskId": "consumer-02-implementation",
        "repositoryId": "consumer-02",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 68630,
        "responseBytes": 641,
        "stderrBytes": 0,
        "stdoutHash": "8a4d0d8d761d396d6ff2216f7abe0bda12cef8268bddc77aac6ad1c349a3b571",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 480654,
        "outputTokens": 2849,
        "tokenMethod": "provider",
        "toolCalls": 9,
        "errorCode": "token-budget"
      },
      "contextBytes": 1657,
      "evidenceIds": [
        "patch-evidence:docs/for-agents/apps/docs.md:18",
        "patch-evidence:docs/architecture/adrs/0027-fumadocs-framework-dogfood.md:3",
        "proposed-change:replace-Proposed-ADR-0027-with-Accepted-ADR-0027-status",
        "verification-plan:ak-docs check --json",
        "verification-blocked:ak-docs-command-unavailable"
      ],
      "round": "structured-provider-recovery-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "proposed_files": 1,
        "verification_exit_code": 127,
        "cachedInputTokens": 410368,
        "reasoningOutputTokens": 1407,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "1ce6ec014864b14c409ead28e968cf25a3242ce1bb15880c153e1ee1dca028b9",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T15:58:53.511Z",
      "runId": "phase-5-structured-provider-recovery-01",
      "planHash": "be4bb39c5cd42038c1f004671fd282c586903f498da615b67675107cd83fdb01",
      "task": {
        "taskId": "consumer-03-discovery",
        "repositoryId": "consumer-03",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 23053,
        "responseBytes": 504,
        "stderrBytes": 0,
        "stdoutHash": "fa4847af1605fe7206a9c943dad79903ac5079634c51ac1d1462c9b6336743e5",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 79898,
        "outputTokens": 430,
        "tokenMethod": "provider",
        "toolCalls": 2
      },
      "contextBytes": 1436,
      "evidenceIds": [
        "AGENTS.md",
        "README.md",
        "package.json",
        "docs/for-agents/INDEX.md",
        "acceptance:ak-docs-discover-command-not-found"
      ],
      "round": "structured-provider-recovery-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksFailed": 1,
        "canonicalDocumentationPathsFound": 4,
        "cachedInputTokens": 50688,
        "reasoningOutputTokens": 127,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "1d5047ba4bee2b8a698163aa3f5fec18634165b62b08d365202a9a640b29fed6",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:00:41.224Z",
      "runId": "phase-5-structured-provider-recovery-01",
      "planHash": "be4bb39c5cd42038c1f004671fd282c586903f498da615b67675107cd83fdb01",
      "task": {
        "taskId": "consumer-03-architecture",
        "repositoryId": "consumer-03",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "invalid-output",
        "exitCode": 0,
        "signal": null,
        "durationMs": 107707,
        "responseBytes": 1382,
        "stderrBytes": 0,
        "stdoutHash": "a2e5f6930b031adeeab4e72c7e220de0d5ac55a6ba870d6485a5b3222277e986",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "errorCode": "invalid-metrics"
      },
      "contextBytes": 1465,
      "evidenceIds": [],
      "round": "structured-provider-recovery-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "2f9bb3c450767a422a7199d5a015142e30e1405d360592736298aedaa2d27230",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:01:22.984Z",
      "runId": "phase-5-structured-provider-recovery-01",
      "planHash": "be4bb39c5cd42038c1f004671fd282c586903f498da615b67675107cd83fdb01",
      "task": {
        "taskId": "consumer-03-documentation",
        "repositoryId": "consumer-03",
        "category": "documentation",
        "scenarioId": "registry-assisted",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "registry-assisted",
        "agentId": "ecosystem-doc-bridge-corpus-scanner",
        "agentVersion": "v1.0.0",
        "network": false
      },
      "execution": {
        "status": "invalid-output",
        "exitCode": 0,
        "signal": null,
        "durationMs": 41753,
        "responseBytes": 1312,
        "stderrBytes": 0,
        "stdoutHash": "cae7fb19b7d53725b2be22c195db35f1f79c1cc85d776e1885189af9bfba728f",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "errorCode": "invalid-metrics"
      },
      "contextBytes": 1621,
      "evidenceIds": [],
      "round": "structured-provider-recovery-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "82b74994ecf350c9d635cb6e5326d1180be3c2e50806edefa135cc0d16d23f98",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:01:49.461Z",
      "runId": "phase-5-structured-provider-recovery-01",
      "planHash": "be4bb39c5cd42038c1f004671fd282c586903f498da615b67675107cd83fdb01",
      "task": {
        "taskId": "consumer-03-implementation",
        "repositoryId": "consumer-03",
        "category": "implementation",
        "scenarioId": "registry-assisted",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "registry-assisted",
        "agentId": "ecosystem-doc-bridge-corpus-scanner",
        "agentVersion": "v1.0.0",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 26470,
        "responseBytes": 299,
        "stderrBytes": 0,
        "stdoutHash": "16587cd588875cfd860025cfb844682184c26f5490eaa9d0c6ed5b61894ad7fe",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 134157,
        "outputTokens": 918,
        "tokenMethod": "provider",
        "toolCalls": 4
      },
      "contextBytes": 1659,
      "evidenceIds": [],
      "round": "structured-provider-recovery-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "low",
      "safetyOutcome": "safe",
      "clarificationRequests": 1,
      "reworkCount": 0,
      "measurements": {
        "cachedInputTokens": 104448,
        "reasoningOutputTokens": 429,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "8010799e2ac687c7c14b8605b0083d21a1791ab6a914152777d9b9de0bfff2a3",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:02:21.850Z",
      "runId": "phase-5-structured-provider-recovery-01",
      "planHash": "be4bb39c5cd42038c1f004671fd282c586903f498da615b67675107cd83fdb01",
      "task": {
        "taskId": "consumer-04-discovery",
        "repositoryId": "consumer-04",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 32382,
        "responseBytes": 395,
        "stderrBytes": 0,
        "stdoutHash": "24b9543ae9af967ee0443f3d88aa77a443572d141ed6b257fba01515e579e031",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 93342,
        "outputTokens": 477,
        "tokenMethod": "provider",
        "toolCalls": 3
      },
      "contextBytes": 1427,
      "evidenceIds": [
        "entrypoint-evidence"
      ],
      "round": "structured-provider-recovery-2026-08-31",
      "taskOutcome": "success",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 1,
      "measurements": {
        "acceptanceChecksPassed": 1,
        "discoverySnapshotsProduced": 1,
        "filesModified": 0,
        "cachedInputTokens": 40448,
        "reasoningOutputTokens": 145,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "36a0d8644b15577bfecd8378f61cc57e63fef372a87e2a85f5c53158827b6a87",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:03:14.948Z",
      "runId": "phase-5-structured-provider-recovery-01",
      "planHash": "be4bb39c5cd42038c1f004671fd282c586903f498da615b67675107cd83fdb01",
      "task": {
        "taskId": "consumer-04-architecture",
        "repositoryId": "consumer-04",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "invalid-output",
        "exitCode": 0,
        "signal": null,
        "durationMs": 53088,
        "responseBytes": 1697,
        "stderrBytes": 0,
        "stdoutHash": "366b94b4984f60a4948cd1fd9142a519092812669deeeb7ee70b09b772054b24",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "errorCode": "invalid-metrics"
      },
      "contextBytes": 1456,
      "evidenceIds": [],
      "round": "structured-provider-recovery-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "b5bb3a9284ca1207db667b64fc907a97bd294e317ab59bb22f42deb421897ede",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:04:36.102Z",
      "runId": "phase-5-structured-provider-recovery-01",
      "planHash": "be4bb39c5cd42038c1f004671fd282c586903f498da615b67675107cd83fdb01",
      "task": {
        "taskId": "consumer-04-documentation",
        "repositoryId": "consumer-04",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "invalid-output",
        "exitCode": 0,
        "signal": null,
        "durationMs": 81146,
        "responseBytes": 1192,
        "stderrBytes": 0,
        "stdoutHash": "f113f959200817c9784a535a8749c9bda273d7613d8ed09cd60e689f66083539",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "errorCode": "invalid-metrics"
      },
      "contextBytes": 1628,
      "evidenceIds": [],
      "round": "structured-provider-recovery-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "7d3df84898323648bece2e3bf6d65a12c9f8997443245988d6f39fba1b807125",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:05:24.042Z",
      "runId": "phase-5-structured-provider-recovery-01",
      "planHash": "be4bb39c5cd42038c1f004671fd282c586903f498da615b67675107cd83fdb01",
      "task": {
        "taskId": "consumer-04-implementation",
        "repositoryId": "consumer-04",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 47934,
        "responseBytes": 1018,
        "stderrBytes": 0,
        "stdoutHash": "d14cbb7ed66784a09abe6d8d634285781c4d2c7fcdaab3be84da253f6371bed0",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 207875,
        "outputTokens": 1673,
        "tokenMethod": "provider",
        "toolCalls": 7
      },
      "contextBytes": 1666,
      "evidenceIds": [
        "patch-evidence:docs/for-agents/index.md; gap=the agent handoff index does not explicitly explain the Doc Bridge query/gate workflow or point contributors to the canonical registry documentation routes",
        "proposal:append a concise Doc Bridge workflow note to docs/for-agents/index.md, preserving existing headings and commands; no source-code or generated-artifact changes",
        "verification-plan:run ak-docs check --json after approval and confirm successful exit under the repository contract",
        "verification-blocked:ak-docs check --json could not create .doc-bridge/workflow/.lock because the workspace is read-only"
      ],
      "round": "structured-provider-recovery-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "proposed_files_changed": 0,
        "required_acceptance_checks_passed": 0,
        "required_acceptance_checks_blocked": 1,
        "cachedInputTokens": 165120,
        "reasoningOutputTokens": 739,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "c777a62668fef17adbf2584f0f211e422f4c8227e48b632f1bc0027d079d000a",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:05:51.175Z",
      "runId": "phase-5-structured-provider-recovery-01",
      "planHash": "be4bb39c5cd42038c1f004671fd282c586903f498da615b67675107cd83fdb01",
      "task": {
        "taskId": "consumer-05-discovery",
        "repositoryId": "consumer-05",
        "category": "discovery",
        "scenarioId": "registry-assisted",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "registry-assisted",
        "agentId": "ecosystem-doc-bridge-corpus-scanner",
        "agentVersion": "v1.0.0",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 27126,
        "responseBytes": 660,
        "stderrBytes": 0,
        "stdoutHash": "8088e1169c2ddfc5b4d763a7eba01a042d96b5880972f9746ec0bf812f940d63",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 63933,
        "outputTokens": 482,
        "tokenMethod": "provider",
        "toolCalls": 2
      },
      "contextBytes": 1429,
      "evidenceIds": [
        "entrypoint-evidence:AGENTS.md",
        "entrypoint-evidence:package.json",
        "entrypoint-evidence:README.md",
        "entrypoint-evidence:app/page.tsx",
        "entrypoint-evidence:app/docs/[[...slug]]/page.tsx",
        "entrypoint-evidence:content/docs/index.mdx",
        "entrypoint-evidence:docs/for-agents/INDEX.md"
      ],
      "round": "structured-provider-recovery-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptance_checks_passed": 0,
        "acceptance_checks_failed": 1,
        "repository_mutations": 0,
        "cachedInputTokens": 40448,
        "reasoningOutputTokens": 130,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "5382c630bf5f9923c6f7702cd1916f7be93036cc9a4d2344b53155da281005d5",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:06:50.323Z",
      "runId": "phase-5-structured-provider-recovery-01",
      "planHash": "be4bb39c5cd42038c1f004671fd282c586903f498da615b67675107cd83fdb01",
      "task": {
        "taskId": "consumer-05-architecture",
        "repositoryId": "consumer-05",
        "category": "architecture",
        "scenarioId": "registry-assisted",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "registry-assisted",
        "agentId": "ecosystem-doc-bridge-corpus-scanner",
        "agentVersion": "v1.0.0",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 59142,
        "responseBytes": 674,
        "stderrBytes": 0,
        "stdoutHash": "5aea5591cbeec28ab9e44513b83f79f2c4ffb148896f507acd0e3959e6bc9cde",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 339175,
        "outputTokens": 2371,
        "tokenMethod": "provider",
        "toolCalls": 10
      },
      "contextBytes": 1458,
      "evidenceIds": [
        "architecture-evidence",
        "artifact:.doc-bridge/index.json",
        "artifact:doc-bridge.config.json",
        "artifact:package.json",
        "artifact:README.md",
        "artifact:content/docs/agentskit-chat.md",
        "artifact:components/ask-widget.tsx",
        "artifact:lib/discovery.ts",
        "acceptance:ak-docs-map-json:blocked"
      ],
      "round": "structured-provider-recovery-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "identified_components": 8,
        "identified_directional_relationships": 10,
        "attention_points": 1,
        "cachedInputTokens": 282112,
        "reasoningOutputTokens": 910,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "16acc4295fff202e6ca71ec9ead4f60f36344b29b88df8b7de6640c99c4e7efa",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:07:54.390Z",
      "runId": "phase-5-structured-provider-recovery-01",
      "planHash": "be4bb39c5cd42038c1f004671fd282c586903f498da615b67675107cd83fdb01",
      "task": {
        "taskId": "consumer-05-documentation",
        "repositoryId": "consumer-05",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 64057,
        "responseBytes": 1078,
        "stderrBytes": 0,
        "stdoutHash": "143db47b398e463921169b5e64699300fa0149e580a5c07219bdff220cebafb6",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 160273,
        "outputTokens": 1265,
        "tokenMethod": "provider",
        "toolCalls": 5
      },
      "contextBytes": 1619,
      "evidenceIds": [
        "documentation-evidence",
        "content/docs/pillars/ai-collaboration/sub-agent-pattern.md:101",
        "content/docs/prompts/index.md:13",
        ".doc-bridge/llms.txt:114-124",
        "review-limitation",
        "audit-blocked:EPERM-read-only-filesystem"
      ],
      "round": "structured-provider-recovery-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "stale claim: sub-agent-pattern says prompt bodies are future work, while prompt index states 12 bodies shipped and generated Doc": 1,
        "classification confidence": 0.99,
        "prompt markdown files currently present": 15,
        "generated index prompt entries observed in .doc-bridge/llms.txt lines 114-124": 11,
        "limitation: ak-docs audit documentation --json could not run because it attempted a temporary write on a read-only filesystem": 1,
        "recommended next action: replace 'bodies in a future session' with wording that reflects shipped prompt bodies and correct the .": 1,
        "cachedInputTokens": 119552,
        "reasoningOutputTokens": 397,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "f03cbf797ba379eeb4515077f28910aac66f5e72eda9ad15f55ff582e4132341",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:09:20.538Z",
      "runId": "phase-5-structured-provider-recovery-01",
      "planHash": "be4bb39c5cd42038c1f004671fd282c586903f498da615b67675107cd83fdb01",
      "task": {
        "taskId": "consumer-05-implementation",
        "repositoryId": "consumer-05",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 86139,
        "responseBytes": 715,
        "stderrBytes": 0,
        "stdoutHash": "90c0aef3d445a9100d82cfad5202c1f155b4be0695686453fa3bb77bc1b13115",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 572321,
        "outputTokens": 3543,
        "tokenMethod": "provider",
        "toolCalls": 13,
        "errorCode": "token-budget"
      },
      "contextBytes": 1657,
      "evidenceIds": [
        "patch-evidence:preceding discovery evidence was unavailable; no knowledge gap could be verified",
        "patch-proposal:packages/harness/README.md; add a minimal link to packages/harness/CONTRIBUTING.md in the workflow/contribution guidance",
        "verification-plan:pnpm exec ak-docs check --json",
        "verification-result:blocked by read-only sandbox EPERM; no files modified"
      ],
      "round": "structured-provider-recovery-2026-08-31",
      "taskOutcome": "incomplete",
      "evidenceQuality": "low",
      "safetyOutcome": "safe",
      "clarificationRequests": 1,
      "reworkCount": 0,
      "measurements": {
        "files_modified": 0,
        "acceptance_checks_passed": 0,
        "cachedInputTokens": 508672,
        "reasoningOutputTokens": 1738,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "69e6ef8fb0ba54b966b72570b6eb6ebe05f77057f6b02cb61fdb75bf09a44e06",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:09:55.915Z",
      "runId": "phase-5-structured-provider-recovery-01",
      "planHash": "be4bb39c5cd42038c1f004671fd282c586903f498da615b67675107cd83fdb01",
      "task": {
        "taskId": "consumer-06-discovery",
        "repositoryId": "consumer-06",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 35367,
        "responseBytes": 409,
        "stderrBytes": 0,
        "stdoutHash": "f4d23234d472712b3b2547771cad14bcc3103f7974ec6d0b7b29160f568220ae",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 90418,
        "outputTokens": 680,
        "tokenMethod": "provider",
        "toolCalls": 3
      },
      "contextBytes": 1436,
      "evidenceIds": [
        "entrypoint-evidence"
      ],
      "round": "structured-provider-recovery-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksFailed": 1,
        "evidenceRequirementsSupported": 1,
        "cachedInputTokens": 62720,
        "reasoningOutputTokens": 244,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "32d006413cdb3695fcc44f63cddc6fdaf1cfa1db3a2eb537e6269fa58a21d174",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:11:07.633Z",
      "runId": "phase-5-structured-provider-recovery-01",
      "planHash": "be4bb39c5cd42038c1f004671fd282c586903f498da615b67675107cd83fdb01",
      "task": {
        "taskId": "consumer-06-architecture",
        "repositoryId": "consumer-06",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 71710,
        "responseBytes": 1157,
        "stderrBytes": 0,
        "stdoutHash": "ffc3a5f5f0c3eaa865a8e2cd94f1d18a121f5f7075672a939f36d6052fb4e9a2",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 516984,
        "outputTokens": 2991,
        "tokenMethod": "provider",
        "toolCalls": 9,
        "errorCode": "token-budget"
      },
      "contextBytes": 1465,
      "evidenceIds": [
        "architecture-evidence",
        "artifact:.doc-bridge/workflow/artifacts/collect-846267f038d3a6f3faa10cd1eb232f097f5d6700e2e59e60387867ca3c57d571",
        "observed:src/cli/program.ts->src/index-builder/build-index.ts,src/discovery/repository.ts,src/query/query.ts,src/mcp/server.ts,src/workflow/engine.ts",
        "observed:src/discovery/repository.ts emits contains,depends-on,imports,re-exports relations",
        "declared:doc-bridge.config.json routes CLI,index,query,MCP,gates,conformance,doctor,memory,chat",
        "attention:dynamic-import coverage partial; unresolved non-literal imports at src/agents/registry-adapter.ts:140 and src/intelligence/peers.ts:50",
        "validation:ak-docs map --json failed EPERM creating .doc-bridge/workflow/.lock"
      ],
      "round": "structured-provider-recovery-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "entity_count": 368,
        "relation_count": 1146,
        "static_import_export_coverage_complete": 1,
        "dynamic_import_coverage_partial": 1,
        "map_command_success": 0,
        "cachedInputTokens": 426752,
        "reasoningOutputTokens": 996,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "50ddcf6797d0c14b80fd399eccc47d3983733d92c3a1d4947fe97f1a5cedcdc7",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:12:09.454Z",
      "runId": "phase-5-structured-provider-recovery-01",
      "planHash": "be4bb39c5cd42038c1f004671fd282c586903f498da615b67675107cd83fdb01",
      "task": {
        "taskId": "consumer-06-documentation",
        "repositoryId": "consumer-06",
        "category": "documentation",
        "scenarioId": "registry-assisted",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "registry-assisted",
        "agentId": "ecosystem-doc-bridge-corpus-scanner",
        "agentVersion": "v1.0.0",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 61812,
        "responseBytes": 823,
        "stderrBytes": 0,
        "stdoutHash": "5dfb9f58a4a65e7f91b851c508d1dd06cfa95a487b78c5c7c28e5845cd232dcf",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 188197,
        "outputTokens": 1254,
        "tokenMethod": "provider",
        "toolCalls": 6
      },
      "contextBytes": 1621,
      "evidenceIds": [
        "documentation-evidence:docs/spec/config-v1.md:836-stale-planned-claim",
        "documentation-evidence:src/cli/program.ts:20,1598-implemented-chat-command",
        "documentation-evidence:src/intelligence/chat.ts:75-runChatOnce-implementation",
        "review-limitation:semantic-classification-high-confidence-but-ak-docs-audit-blocked-by-command-path-and-read-only-lock"
      ],
      "round": "structured-provider-recovery-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "documentationIssuesClassified": 1,
        "documentClaimsCompared": 1,
        "sourceImplementationsCompared": 3,
        "acceptanceChecksPassed": 0,
        "acceptanceChecksBlocked": 1,
        "confidencePercent": 95,
        "cachedInputTokens": 121600,
        "reasoningOutputTokens": 370,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "cc4f418fb0e17f6daa0ff1412d86f18beb59b19119f4ebaae308f28a6a2ce06a",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:12:55.442Z",
      "runId": "phase-5-structured-provider-recovery-01",
      "planHash": "be4bb39c5cd42038c1f004671fd282c586903f498da615b67675107cd83fdb01",
      "task": {
        "taskId": "consumer-06-implementation",
        "repositoryId": "consumer-06",
        "category": "implementation",
        "scenarioId": "registry-assisted",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "registry-assisted",
        "agentId": "ecosystem-doc-bridge-corpus-scanner",
        "agentVersion": "v1.0.0",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 45980,
        "responseBytes": 633,
        "stderrBytes": 0,
        "stdoutHash": "b23b0f1247a43872c3a3fae782a03fc004938fbe4bb9045d440fc86b3fe5808b",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 216986,
        "outputTokens": 1802,
        "tokenMethod": "provider",
        "toolCalls": 6
      },
      "contextBytes": 1659,
      "evidenceIds": [
        "patch-evidence:docs/spec/config-v1.md:836-stale-planned-claim",
        "patch-evidence:src/cli/program.ts:20,1598-implemented-chat-command",
        "patch-evidence:src/intelligence/chat.ts:75-runChatOnce-implementation",
        "verification-plan:ak-docs check --json"
      ],
      "round": "structured-provider-recovery-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "proposedDocumentationFiles": 1,
        "proposedChangedLines": 1,
        "acceptanceChecksExecuted": 0,
        "cachedInputTokens": 167424,
        "reasoningOutputTokens": 843,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "41825caf0675e5d64e42ad82a9a160b1f35c50b94359ab938082ba488bbc0687",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:19:11.971Z",
      "runId": "phase-6-structured-provider-normalized-01",
      "planHash": "8903a96835128a57d5d7981a0fb81a7ef8823b4603294a3ce7a196eb19fb258c",
      "task": {
        "taskId": "consumer-01-discovery",
        "repositoryId": "consumer-01",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 34486,
        "responseBytes": 639,
        "stderrBytes": 0,
        "stdoutHash": "6cc0ed559fb3259e8588e270b6af6deca002551eb7dd6f8b1b896779f14df00f",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 107479,
        "outputTokens": 751,
        "tokenMethod": "provider",
        "toolCalls": 3
      },
      "contextBytes": 1427,
      "evidenceIds": [
        "artifact:package.json",
        "artifact:pnpm-workspace.yaml",
        "artifact:README.md",
        "artifact:AGENTS.md",
        "artifact:.github/CODEOWNERS",
        "artifact:doc-bridge.config.json",
        "artifact:llms.txt",
        "acceptance:ak-docs-discover-json:exit-127-command-not-found"
      ],
      "round": "structured-provider-normalized-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "artifactPathsInspected": 7,
        "acceptanceChecksRun": 1,
        "acceptanceChecksPassed": 0,
        "filesModified": 0,
        "cachedInputTokens": 82944,
        "reasoningOutputTokens": 218,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "9a1e2a49e875329ec4404896ec21b5d8d7edb7a3e49cfd0d03e0fecf9cd58c64",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:20:43.151Z",
      "runId": "phase-6-structured-provider-normalized-01",
      "planHash": "8903a96835128a57d5d7981a0fb81a7ef8823b4603294a3ce7a196eb19fb258c",
      "task": {
        "taskId": "consumer-01-architecture",
        "repositoryId": "consumer-01",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 91170,
        "responseBytes": 677,
        "stderrBytes": 0,
        "stdoutHash": "57b2f2f3fa373bf3fd09044b1504f9bd500603fbde2a69e780b55f6196e0436a",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 817320,
        "outputTokens": 3259,
        "tokenMethod": "provider",
        "toolCalls": 16,
        "errorCode": "token-budget"
      },
      "contextBytes": 1456,
      "evidenceIds": [
        "architecture-evidence",
        "docs/architecture-at-a-glance.mdx",
        "pnpm-workspace.yaml",
        "packages/*/package.json",
        "packages/*/src imports",
        "attention:observability-topology-graph-runtime-shape-drift-risk",
        "ak-docs-map-command:blocked-by-EPERM-lock"
      ],
      "round": "structured-provider-normalized-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "components_identified": 8,
        "directional_relationships_identified": 12,
        "architecture_docs_observed": 1,
        "required_checks_emitted_map": 0,
        "cachedInputTokens": 739072,
        "reasoningOutputTokens": 980,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "18b07a707bf43dc7013e05d92362d5440ed1474aa31b7bc1157ae4e4e9c1e848",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:21:33.655Z",
      "runId": "phase-6-structured-provider-normalized-01",
      "planHash": "8903a96835128a57d5d7981a0fb81a7ef8823b4603294a3ce7a196eb19fb258c",
      "task": {
        "taskId": "consumer-01-documentation",
        "repositoryId": "consumer-01",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 50493,
        "responseBytes": 523,
        "stderrBytes": 0,
        "stdoutHash": "2ec8bcb8673a0130d85f5f15c1a78d4b35e68c17c98f4e5be4471590e62b20f1",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 153805,
        "outputTokens": 1018,
        "tokenMethod": "provider",
        "toolCalls": 5
      },
      "contextBytes": 1628,
      "evidenceIds": [
        "evidence-30dad67620105c00f9a51b003fe4925d",
        "evidence-913cee28261896ef5bc175c2080c2057"
      ],
      "round": "structured-provider-normalized-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 1,
      "measurements": {
        "staleClaimsFound": 4,
        "documentationFilesCompared": 1,
        "sourceArtifactsCompared": 1,
        "acceptanceChecksPassed": 0,
        "acceptanceChecksBlocked": 1,
        "cachedInputTokens": 127488,
        "reasoningOutputTokens": 243,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "9c0c4bab96c02aa892d341561a8b034c1103d691edd43944dce4a88c32996967",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:21:41.700Z",
      "runId": "phase-6-structured-provider-normalized-01",
      "planHash": "8903a96835128a57d5d7981a0fb81a7ef8823b4603294a3ce7a196eb19fb258c",
      "task": {
        "taskId": "consumer-01-implementation",
        "repositoryId": "consumer-01",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 8038,
        "responseBytes": 296,
        "stderrBytes": 0,
        "stdoutHash": "965ac4a4c5fe90e043f39268f1764636f6152626f358fa9f4e8a8c8f4975fbc4",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 20420,
        "outputTokens": 209,
        "tokenMethod": "provider",
        "toolCalls": 0
      },
      "contextBytes": 1666,
      "evidenceIds": [],
      "round": "structured-provider-normalized-2026-08-31",
      "taskOutcome": "incomplete",
      "evidenceQuality": "low",
      "safetyOutcome": "safe",
      "clarificationRequests": 1,
      "reworkCount": 0,
      "measurements": {
        "cachedInputTokens": 0,
        "reasoningOutputTokens": 157,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "d66229e3666038ed8daaf2a8aca700c3a1c5afa14c2d3427d25cead71bfca0e1",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:22:04.317Z",
      "runId": "phase-6-structured-provider-normalized-01",
      "planHash": "8903a96835128a57d5d7981a0fb81a7ef8823b4603294a3ce7a196eb19fb258c",
      "task": {
        "taskId": "consumer-02-discovery",
        "repositoryId": "consumer-02",
        "category": "discovery",
        "scenarioId": "registry-assisted",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "registry-assisted",
        "agentId": "ecosystem-doc-bridge-corpus-scanner",
        "agentVersion": "v1.0.0",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 22611,
        "responseBytes": 351,
        "stderrBytes": 0,
        "stdoutHash": "2fe89ca96d9868a3299631933e9415b13e4c791e0be165968e9798f222fff4ed",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 65507,
        "outputTokens": 420,
        "tokenMethod": "provider",
        "toolCalls": 2
      },
      "contextBytes": 1429,
      "evidenceIds": [],
      "round": "structured-provider-normalized-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "low",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksFailed": 1,
        "cachedInputTokens": 40448,
        "reasoningOutputTokens": 168,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "8b96192cd594a92e643011eafb823f9df49c1a55c40dca6b60e8891957390ab0",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:22:47.326Z",
      "runId": "phase-6-structured-provider-normalized-01",
      "planHash": "8903a96835128a57d5d7981a0fb81a7ef8823b4603294a3ce7a196eb19fb258c",
      "task": {
        "taskId": "consumer-02-architecture",
        "repositoryId": "consumer-02",
        "category": "architecture",
        "scenarioId": "registry-assisted",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "registry-assisted",
        "agentId": "ecosystem-doc-bridge-corpus-scanner",
        "agentVersion": "v1.0.0",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 43003,
        "responseBytes": 804,
        "stderrBytes": 0,
        "stdoutHash": "6c3df36391fd2c2e490c53bceb8fd2d27715a3df0cee7a65c3b0ff57789d95d7",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 137273,
        "outputTokens": 1482,
        "tokenMethod": "provider",
        "toolCalls": 4
      },
      "contextBytes": 1458,
      "evidenceIds": [
        "architecture-evidence",
        "architecture-overview:Application->Core->AgentsKit-runtime",
        "architecture-overview:Core->Protocol->Server->Native-renderers",
        "workspace-manifests:chat-and-framework-packages",
        "workspace-manifests:apps-consume-chat-and-framework-bindings",
        "boundary-review:core-version-range-drift",
        "ak-docs-map:blocked-command-unavailable-write-permission"
      ],
      "round": "structured-provider-normalized-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "architecture_components_observed": 7,
        "directional_relationships_observed": 6,
        "attention_points_identified": 1,
        "acceptance_checks_passed": 0,
        "cachedInputTokens": 117504,
        "reasoningOutputTokens": 686,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "3884c9c6f43bfb0e7081bef32b2bc5faf80c440d12cb18d29ec609f2e4765d78",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:23:46.146Z",
      "runId": "phase-6-structured-provider-normalized-01",
      "planHash": "8903a96835128a57d5d7981a0fb81a7ef8823b4603294a3ce7a196eb19fb258c",
      "task": {
        "taskId": "consumer-02-documentation",
        "repositoryId": "consumer-02",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 58811,
        "responseBytes": 743,
        "stderrBytes": 0,
        "stdoutHash": "97d3260d82de4e2e33ea05b960852701bd8540935b54102b08bb07a80ba13dae",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 182843,
        "outputTokens": 1235,
        "tokenMethod": "provider",
        "toolCalls": 6
      },
      "contextBytes": 1619,
      "evidenceIds": [
        "evidence-ebefca3342e6608037f75bfe8f413426",
        "generated-index-evidence:.doc-bridge/index.json:17 repeats the stale Proposed status",
        "review-limitation:high-confidence repository-local semantic comparison; ak-docs audit could not complete because the read-only sandbox denied creation of .doc-bridge/workflow/.lock"
      ],
      "round": "structured-provider-normalized-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 1,
      "measurements": {
        "documentationFindings": 1,
        "confidence": 0.99,
        "acceptanceChecksAttempted": 1,
        "acceptanceChecksPassed": 0,
        "acceptanceChecksBlocked": 1,
        "cachedInputTokens": 160000,
        "reasoningOutputTokens": 277,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "84b3154ec6d25070a34372d4d508a703cc5c707f77f70f92533e92715f2f1f57",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:24:07.396Z",
      "runId": "phase-6-structured-provider-normalized-01",
      "planHash": "8903a96835128a57d5d7981a0fb81a7ef8823b4603294a3ce7a196eb19fb258c",
      "task": {
        "taskId": "consumer-02-implementation",
        "repositoryId": "consumer-02",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 21242,
        "responseBytes": 503,
        "stderrBytes": 0,
        "stdoutHash": "05b230c4f24e4392dae5671290e47673c8e1886131015c06a7c136243f16fd03",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 92195,
        "outputTokens": 712,
        "tokenMethod": "provider",
        "toolCalls": 3
      },
      "contextBytes": 1657,
      "evidenceIds": [
        "artifact:preceding-evidence-unavailable",
        "contract:.codex/verification.json-missing",
        "runner:ak-docs-check-command-not-found"
      ],
      "round": "structured-provider-normalized-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "low",
      "safetyOutcome": "not-applicable",
      "clarificationRequests": 1,
      "reworkCount": 0,
      "measurements": {
        "required_contracts_available": 0,
        "verification_command_exit_code": 127,
        "cachedInputTokens": 73728,
        "reasoningOutputTokens": 317,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "71b7d9554a1ad8397d0f75dba6378d37af68b46997ba56c759925922e430f4f0",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:24:30.864Z",
      "runId": "phase-6-structured-provider-normalized-01",
      "planHash": "8903a96835128a57d5d7981a0fb81a7ef8823b4603294a3ce7a196eb19fb258c",
      "task": {
        "taskId": "consumer-03-discovery",
        "repositoryId": "consumer-03",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 23459,
        "responseBytes": 543,
        "stderrBytes": 0,
        "stdoutHash": "227fd6c9f9c2c3774751f71a96a797063f424bb350694768f18c6ab17e198d22",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 89161,
        "outputTokens": 454,
        "tokenMethod": "provider",
        "toolCalls": 2
      },
      "contextBytes": 1436,
      "evidenceIds": [
        "entrypoint-evidence:AGENTS.md",
        "entrypoint-evidence:package.json",
        "entrypoint-evidence:docs/for-agents/INDEX.md",
        "discovery-check:command-not-found"
      ],
      "round": "structured-provider-normalized-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksAttempted": 1,
        "canonicalDocumentationPathsFound": 3,
        "cachedInputTokens": 50688,
        "reasoningOutputTokens": 113,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "bafd91c04dfc8dd920face6a4e53ff997f303e32720cee4979e26835e261d57e",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:25:13.112Z",
      "runId": "phase-6-structured-provider-normalized-01",
      "planHash": "8903a96835128a57d5d7981a0fb81a7ef8823b4603294a3ce7a196eb19fb258c",
      "task": {
        "taskId": "consumer-03-architecture",
        "repositoryId": "consumer-03",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 42240,
        "responseBytes": 637,
        "stderrBytes": 0,
        "stdoutHash": "8c20ec9615932c878223ad44d44d687252e726bf7e9e921d83d0c4c410ceb0d2",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 241180,
        "outputTokens": 1685,
        "tokenMethod": "provider",
        "toolCalls": 5
      },
      "contextBytes": 1465,
      "evidenceIds": [
        "architecture-evidence",
        "docs/internal/architecture.md",
        "docs/for-agents/architecture.md",
        "packages/os-headless/package.json",
        "packages/os-contracts/package.json",
        "docs/for-agents/packages/os-contracts.md",
        "architecture-check-blocked"
      ],
      "round": "structured-provider-normalized-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "architecture_layers_observed": 4,
        "l3_composition_packages_observed": 11,
        "ak_docs_map_command_failed": 1,
        "cachedInputTokens": 197120,
        "reasoningOutputTokens": 697,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "0df327df5aca63462bd4c9d918dfa1bb68e55c282b37baa8ee7fc00834f94c5f",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:26:06.594Z",
      "runId": "phase-6-structured-provider-normalized-01",
      "planHash": "8903a96835128a57d5d7981a0fb81a7ef8823b4603294a3ce7a196eb19fb258c",
      "task": {
        "taskId": "consumer-03-documentation",
        "repositoryId": "consumer-03",
        "category": "documentation",
        "scenarioId": "registry-assisted",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "registry-assisted",
        "agentId": "ecosystem-doc-bridge-corpus-scanner",
        "agentVersion": "v1.0.0",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 53475,
        "responseBytes": 1285,
        "stderrBytes": 0,
        "stdoutHash": "a0c1d66e97d05fecefbb7a3e9d1f675acc9a9f0f7052e272b25d602f9aa95618",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 139828,
        "outputTokens": 1195,
        "tokenMethod": "provider",
        "toolCalls": 4
      },
      "contextBytes": 1621,
      "evidenceIds": [
        "evidence-4708828eb1a9a70c52a0f1be766e5361",
        "review-limitation:Semantic classification was manually inferred from the current filesystem and AGENTS.md; the audit explicitly reports natural-language semantics as not analyzed and requires human approval before editing.",
        "audit-evidence:.codex/verification/runs/1788114915060-15146/run.json documents a passing documentation-audit check with status needs-review, 1 not-analyzed limitation, and 1348 documents audited.",
        "execution-limitation:The requested bare ak-docs command was unavailable, while pnpm exec ak-docs could not create a temporary file under the read-only sandbox; cached current-repository audit evidence was inspected instead.",
        "source-revision:090bb13080dd298f14ee6509eed68d54c8977d65"
      ],
      "round": "structured-provider-normalized-2026-08-31",
      "taskOutcome": "success",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "documentedPackageDirectories": 76,
        "actualPackageDirectories": 77,
        "documentedPackDirectories": 2,
        "actualPackDirectories": 3,
        "packageCountDifference": 1,
        "auditDocumentCount": 1348,
        "auditNotAnalyzedCount": 1,
        "classifiedDocumentationIssues": 1,
        "cachedInputTokens": 123648,
        "reasoningOutputTokens": 240,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "a79d41764f3bc569d9ee9029b5404838fba4fbbbe328800520c6918e12eff361",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:26:48.960Z",
      "runId": "phase-6-structured-provider-normalized-01",
      "planHash": "8903a96835128a57d5d7981a0fb81a7ef8823b4603294a3ce7a196eb19fb258c",
      "task": {
        "taskId": "consumer-03-implementation",
        "repositoryId": "consumer-03",
        "category": "implementation",
        "scenarioId": "registry-assisted",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "registry-assisted",
        "agentId": "ecosystem-doc-bridge-corpus-scanner",
        "agentVersion": "v1.0.0",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 42357,
        "responseBytes": 862,
        "stderrBytes": 0,
        "stdoutHash": "35b84d70282f75a4a5a02a61d40eb5514b4675296f6fcfcd1546c879c438cf05",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 289328,
        "outputTokens": 1641,
        "tokenMethod": "provider",
        "toolCalls": 7
      },
      "contextBytes": 1659,
      "evidenceIds": [
        "patch-evidence:docs/security/connections-byok-containment.md — distinguishes general catalog certification from the MVP allowlist and records the verified allowlist gap",
        "patch-evidence:docs/testing/provider-live-certification.md — aligns the runbook with the MVP journey matrix and candidate-SHA evidence requirement",
        "verification-plan:ak-docs check --json after approval; review the two documentation files for scope, conventions, and source-evidence alignment"
      ],
      "round": "structured-provider-normalized-2026-08-31",
      "taskOutcome": "success",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "documentation_files_in_scope": 2,
        "source_code_files_changed": 0,
        "external_systems_changed": 0,
        "cachedInputTokens": 229632,
        "reasoningOutputTokens": 684,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "c617db114047be180f0651bdf2d4e3cd1cb646f7e4a6c9dc46d23c5c362aaf13",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:27:19.179Z",
      "runId": "phase-6-structured-provider-normalized-01",
      "planHash": "8903a96835128a57d5d7981a0fb81a7ef8823b4603294a3ce7a196eb19fb258c",
      "task": {
        "taskId": "consumer-04-discovery",
        "repositoryId": "consumer-04",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 30209,
        "responseBytes": 623,
        "stderrBytes": 0,
        "stdoutHash": "c1e9e01d056ee5b0179f2b8e7f2257f267619e6de406434d17a30c7af63482b9",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 64782,
        "outputTokens": 531,
        "tokenMethod": "provider",
        "toolCalls": 2
      },
      "contextBytes": 1427,
      "evidenceIds": [
        "entrypoint-evidence:package.json",
        "entrypoint-evidence:AGENTS.md",
        "entrypoint-evidence:docs/for-agents/index.md",
        "entrypoint-evidence:docs/architecture.md",
        "discovery-check:ak-docs-command-not-found"
      ],
      "round": "structured-provider-normalized-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "entrypointDeclarationsLocated": 1,
        "ownershipEvidencePathsLocated": 3,
        "acceptanceChecksPassed": 0,
        "acceptanceChecksBlocked": 1,
        "cachedInputTokens": 40448,
        "reasoningOutputTokens": 148,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "54ae8fa0f4a7c8fe9ccd64536c6e85af5393bbed27232f97b08f462ab5381d29",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:28:58.590Z",
      "runId": "phase-6-structured-provider-normalized-01",
      "planHash": "8903a96835128a57d5d7981a0fb81a7ef8823b4603294a3ce7a196eb19fb258c",
      "task": {
        "taskId": "consumer-04-architecture",
        "repositoryId": "consumer-04",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 99399,
        "responseBytes": 1077,
        "stderrBytes": 0,
        "stdoutHash": "171653e755a4e53683218522225a7dc02b54dffc03062f2278b576d726a60b64",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 693210,
        "outputTokens": 3939,
        "tokenMethod": "provider",
        "toolCalls": 13,
        "errorCode": "token-budget"
      },
      "contextBytes": 1456,
      "evidenceIds": [
        "architecture-evidence:docs/architecture.md:registry-source->scripts/build-registry.mjs->public/r-and-public/deterministic->external-consumers",
        "architecture-evidence:scripts/build-registry.mjs:14-18->scripts/lib/deterministic-discovery.mjs",
        "architecture-evidence:scripts/lib/deterministic-discovery.mjs:3-12->external:@agentskit/chat/protocol",
        "architecture-evidence:package.json:14-50->AgentsKit-package-dependencies",
        "architecture-attention:review-generated-artifact-drift-at-public/r-and-public/deterministic; canonical sources are scripts/ and registry/",
        "architecture-check:ak-docs-map-json-blocked-by-read-only-EPERM-on-.doc-bridge/workflow/.lock"
      ],
      "round": "structured-provider-normalized-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "observed_entities": 1489,
        "observed_relations": 5033,
        "observed_import_relations": 3943,
        "observed_external_dependencies": 29,
        "cachedInputTokens": 611840,
        "reasoningOutputTokens": 872,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "19a6bf67e2fb955850aa454a7101d2d1044a5d39fe9b02ea2c85283e52bc3738",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:30:02.277Z",
      "runId": "phase-6-structured-provider-normalized-01",
      "planHash": "8903a96835128a57d5d7981a0fb81a7ef8823b4603294a3ce7a196eb19fb258c",
      "task": {
        "taskId": "consumer-04-documentation",
        "repositoryId": "consumer-04",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 63678,
        "responseBytes": 913,
        "stderrBytes": 0,
        "stdoutHash": "dd9ff32288113c9f98d4483b559ca9944c50dcbad5546a499a0bad6215ef093e",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 116715,
        "outputTokens": 1381,
        "tokenMethod": "provider",
        "toolCalls": 4
      },
      "contextBytes": 1628,
      "evidenceIds": [
        "documentation-evidence:missing-generated-link-targets:docs/for-agents/registry-validation.md contains 3 source-of-truth link targets, while its .doc-bridge/index.json knowledge body renders all 3 bullets without targets",
        "review-limitation:semantic impact is strongly indicated for consumers of the normalized body, but intended parser behavior was not independently confirmed",
        "acceptance-check-limitation:ak-docs audit documentation --json is unsupported by the installed ak-docs CLI"
      ],
      "round": "structured-provider-normalized-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "documentationFindings": 1,
        "sourceLinkTargets": 3,
        "generatedBodyLinkTargets": 0,
        "acceptanceCommandSupported": 0,
        "confidence": 0.96,
        "cachedInputTokens": 62720,
        "reasoningOutputTokens": 559,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "4b2e7f06df162d1c02ae4967cec7e118f50da1ae9bda903388920a390a434580",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:31:16.066Z",
      "runId": "phase-6-structured-provider-normalized-01",
      "planHash": "8903a96835128a57d5d7981a0fb81a7ef8823b4603294a3ce7a196eb19fb258c",
      "task": {
        "taskId": "consumer-04-implementation",
        "repositoryId": "consumer-04",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 73778,
        "responseBytes": 447,
        "stderrBytes": 0,
        "stdoutHash": "81b6934aa115d23ddd39f0d24e38d804b061b75eefa84ff1784e9df0aa3a2508",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 686892,
        "outputTokens": 2833,
        "tokenMethod": "provider",
        "toolCalls": 13,
        "errorCode": "token-budget"
      },
      "contextBytes": 1666,
      "evidenceIds": [
        "patch-evidence",
        "verification-plan"
      ],
      "round": "structured-provider-normalized-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "target_document_count": 1,
        "source_code_files_changed": 0,
        "documentation_only": 1,
        "acceptance_check_passed": 0,
        "cachedInputTokens": 621056,
        "reasoningOutputTokens": 797,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "f7e49d6134d05b4c3f980c9ebe0fd6c67fc962865e49b47e0025f2a91118ce4e",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:31:50.042Z",
      "runId": "phase-6-structured-provider-normalized-01",
      "planHash": "8903a96835128a57d5d7981a0fb81a7ef8823b4603294a3ce7a196eb19fb258c",
      "task": {
        "taskId": "consumer-05-discovery",
        "repositoryId": "consumer-05",
        "category": "discovery",
        "scenarioId": "registry-assisted",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "registry-assisted",
        "agentId": "ecosystem-doc-bridge-corpus-scanner",
        "agentVersion": "v1.0.0",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 33958,
        "responseBytes": 764,
        "stderrBytes": 0,
        "stdoutHash": "69566b6a5f73699dfd7112e13ccf2e2185662bab6a9b7455e1fb3854ae3e0f6a",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 74538,
        "outputTokens": 670,
        "tokenMethod": "provider",
        "toolCalls": 2
      },
      "contextBytes": 1429,
      "evidenceIds": [
        "entrypoint-evidence:AGENTS.md",
        "entrypoint-evidence:README.md",
        "entrypoint-evidence:package.json",
        "entrypoint-evidence:source.config.ts",
        "entrypoint-evidence:app/page.tsx",
        "entrypoint-evidence:app/docs/[[...slug]]/page.tsx",
        "entrypoint-evidence:content/docs/index.mdx",
        "entrypoint-evidence:docs/for-agents/INDEX.md"
      ],
      "round": "structured-provider-normalized-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "entrypointsLocated": 6,
        "ownershipBoundariesLocated": 2,
        "canonicalDocumentationPathsLocated": 3,
        "acceptanceChecksPassed": 0,
        "acceptanceChecksBlocked": 1,
        "cachedInputTokens": 42496,
        "reasoningOutputTokens": 133,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "12d2765cf23da11f6c969c43824c0c30985f0990b108b3847f3fa1b4747833ad",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:32:47.025Z",
      "runId": "phase-6-structured-provider-normalized-01",
      "planHash": "8903a96835128a57d5d7981a0fb81a7ef8823b4603294a3ce7a196eb19fb258c",
      "task": {
        "taskId": "consumer-05-architecture",
        "repositoryId": "consumer-05",
        "category": "architecture",
        "scenarioId": "registry-assisted",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "registry-assisted",
        "agentId": "ecosystem-doc-bridge-corpus-scanner",
        "agentVersion": "v1.0.0",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 56961,
        "responseBytes": 1483,
        "stderrBytes": 0,
        "stdoutHash": "5cfe1893c86bcd88a2088d8855b92b3a7ff30d6048489bc454a3c89c151d7e86",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 267923,
        "outputTokens": 2333,
        "tokenMethod": "provider",
        "toolCalls": 6
      },
      "contextBytes": 1458,
      "evidenceIds": [
        "architecture-evidence:README.md#L90-L108 identifies Markdown source -> Fumadocs, Doc Bridge, raw/LLM routes, deterministic artifact -> AgentsKit Chat -> optional backend, and quality gates",
        "architecture-evidence:package.json#L41-L57 confirms Next/Fumadocs, AgentsKit Chat/Core/React, Doc Bridge, and Harness dependencies",
        "architecture-evidence:doc-bridge.config.json#L6-L23 binds content/docs to Fumadocs and machine routing",
        "architecture-evidence:components/ask-widget.tsx#L17-L33 imports AgentsKit Chat/Core/React and local discovery",
        "architecture-evidence:lib/discovery.ts#L19-L35 and #L78-L107 show verified local knowledge with backend fallback",
        "architecture-evidence:AGENTS.md#L5-L9 states the boundary: this host owns corpus/configuration while chat state, lifecycle, streaming, memory, cancellation, and portable components belong to AgentsKit Chat",
        "acceptance-check:ak-docs map --json blocked because ak-docs is not on PATH and direct invocation attempted a lock-directory write rejected with EPERM"
      ],
      "round": "structured-provider-normalized-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "bounded_components_identified": 7,
        "directional_relations_identified": 8,
        "evidence_backed_attention_points": 1,
        "acceptance_checks_passed": 0,
        "acceptance_checks_blocked": 1,
        "cachedInputTokens": 200960,
        "reasoningOutputTokens": 954,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "b39a84983d413fb320652a2856b07eab38db7393eea8c934d7310bd6771c56d3",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:34:04.275Z",
      "runId": "phase-6-structured-provider-normalized-01",
      "planHash": "8903a96835128a57d5d7981a0fb81a7ef8823b4603294a3ce7a196eb19fb258c",
      "task": {
        "taskId": "consumer-05-documentation",
        "repositoryId": "consumer-05",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 77240,
        "responseBytes": 511,
        "stderrBytes": 0,
        "stdoutHash": "2d0573779f75a0e48f858acb998bc4e93df6a97f495d90a841aac0ee02c32331",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 214917,
        "outputTokens": 1672,
        "tokenMethod": "provider",
        "toolCalls": 7
      },
      "contextBytes": 1619,
      "evidenceIds": [
        "evidence-b94c4329a0f92ce211de29832e4f1a23",
        "evidence-87d506eeed554a1bb7242c51bade2f48"
      ],
      "round": "structured-provider-normalized-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "documentationIssues": 1,
        "documentClaimsCompared": 2,
        "sourceArtifactsChecked": 2,
        "confidence": 0.88,
        "auditFindingsEmitted": 0,
        "cachedInputTokens": 189440,
        "reasoningOutputTokens": 580,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "f86bdd7ab4498f22af231acc79563dde61d4b3c9338681b46d3fa598c98e9854",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:34:42.641Z",
      "runId": "phase-6-structured-provider-normalized-01",
      "planHash": "8903a96835128a57d5d7981a0fb81a7ef8823b4603294a3ce7a196eb19fb258c",
      "task": {
        "taskId": "consumer-05-implementation",
        "repositoryId": "consumer-05",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 38357,
        "responseBytes": 387,
        "stderrBytes": 0,
        "stdoutHash": "91e9938f227927c7eacabbec5a66955f4a5972c962f692f068aa5e87556ae681",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 211465,
        "outputTokens": 1310,
        "tokenMethod": "provider",
        "toolCalls": 6
      },
      "contextBytes": 1657,
      "evidenceIds": [
        "patch-evidence",
        "verification-plan"
      ],
      "round": "structured-provider-normalized-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "files_modified": 0,
        "acceptance_checks_passed": 0,
        "cachedInputTokens": 166400,
        "reasoningOutputTokens": 573,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "31b8d5953bbddf11d62abff6625a53a8d3f9f780171633c0c3c814d7f2f41620",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:35:28.990Z",
      "runId": "phase-6-structured-provider-normalized-01",
      "planHash": "8903a96835128a57d5d7981a0fb81a7ef8823b4603294a3ce7a196eb19fb258c",
      "task": {
        "taskId": "consumer-06-discovery",
        "repositoryId": "consumer-06",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 46336,
        "responseBytes": 1135,
        "stderrBytes": 0,
        "stdoutHash": "725caee47d07263a52fb3e042a2e76d881a946c82bcd8feefbeb7a2001eccaa9",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 97472,
        "outputTokens": 904,
        "tokenMethod": "provider",
        "toolCalls": 3
      },
      "contextBytes": 1436,
      "evidenceIds": [
        "entrypoint-evidence:package.json (CLI bin, library exports, repository metadata)",
        "entrypoint-evidence:bin/ak-docs.js (primary CLI entrypoint)",
        "entrypoint-evidence:src/index.ts (primary library entrypoint)",
        "entrypoint-evidence:src/cli/program.ts (CLI ownership boundary)",
        "entrypoint-evidence:src/mcp/server.ts (MCP ownership boundary)",
        "entrypoint-evidence:doc-bridge.config.json (canonical ownership routing and documentation paths)",
        "entrypoint-evidence:README.md (canonical project and usage documentation)",
        "entrypoint-evidence:GOVERNANCE.md (maintainer ownership and decision boundaries)",
        "entrypoint-evidence:failed:ak-docs discover --json (command unavailable, exit 127)"
      ],
      "round": "structured-provider-normalized-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "declared_primary_entrypoints": 3,
        "declared_ownership_boundaries": 10,
        "canonical_evidence_paths": 8,
        "acceptance_checks_passed": 0,
        "acceptance_checks_failed": 1,
        "cachedInputTokens": 65792,
        "reasoningOutputTokens": 301,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "822833e0d299de323372c6b4d4b1ec0cb7027a68c50eab178619b9070edc7d0f",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:36:47.398Z",
      "runId": "phase-6-structured-provider-normalized-01",
      "planHash": "8903a96835128a57d5d7981a0fb81a7ef8823b4603294a3ce7a196eb19fb258c",
      "task": {
        "taskId": "consumer-06-architecture",
        "repositoryId": "consumer-06",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 78396,
        "responseBytes": 1724,
        "stderrBytes": 0,
        "stdoutHash": "ac8a83693617537c56a460cb496d2b95b6ed0aec63b95c2884ff168e5eae687c",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 527844,
        "outputTokens": 3302,
        "tokenMethod": "provider",
        "toolCalls": 9,
        "errorCode": "token-budget"
      },
      "contextBytes": 1465,
      "evidenceIds": [
        "evidence-b54bba27347cb9234968dd29e09873dd",
        "component-edges:src/cli/program.ts imports discovery/repository, index-builder/build-index, query, mcp/server, gates, memory, intelligence, reconciliation, workflow; evidence=src/cli/program.ts:12-52",
        "component-edges:index-builder/build-index imports handoff builder, human adapters, pnpm/Nx plugins, agent corpus scanner, capabilities, hashes, llms renderer; evidence=src/index-builder/build-index.ts:4-14",
        "component-edges:mcp/server imports retriever, gates, memory pipeline, query, workflow, reconciliation validation, fixes, registry adapter, discovery, safety; evidence=src/mcp/server.ts:6-23",
        "documentation-claims:doc-bridge.config.json defines CLI, index, query, MCP, gates, conformance, doctor, memory, and optional intelligence ownership; evidence=doc-bridge.config.json:24-96; README.md:113-125",
        "attention-point:static-imports-and-exports coverage complete, but dynamic-imports coverage partial; unresolved/non-literal loading is evidenced at src/agents/registry-adapter.ts:140 and src/intelligence/peers.ts:50",
        "acceptance-check:node bin/ak-docs.js map --json failed with EPERM while creating .doc-bridge/workflow/.lock; existing workflow artifact was read-only evidence, not a fresh command result"
      ],
      "round": "structured-provider-normalized-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "module_entities": 199,
        "document_entities": 85,
        "package_entities": 1,
        "external_entities": 83,
        "import_relations": 850,
        "reexport_relations": 62,
        "dependency_relations": 36,
        "cachedInputTokens": 440832,
        "reasoningOutputTokens": 932,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "47ad17ff451848b3bbc5762c751e05099d250e49ce8ead4818ec2c2fbfa65564",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:38:00.205Z",
      "runId": "phase-6-structured-provider-normalized-01",
      "planHash": "8903a96835128a57d5d7981a0fb81a7ef8823b4603294a3ce7a196eb19fb258c",
      "task": {
        "taskId": "consumer-06-documentation",
        "repositoryId": "consumer-06",
        "category": "documentation",
        "scenarioId": "registry-assisted",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "20897bddf03704d26f975003bf3ba0d4d8cb38559b382603d7a03cb17b487a7c"
      },
      "scenario": {
        "id": "registry-assisted",
        "agentId": "ecosystem-doc-bridge-corpus-scanner",
        "agentVersion": "v1.0.0",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 72795,
        "responseBytes": 676,
        "stderrBytes": 0,
        "stdoutHash": "f7f4406818ca021dc443b2cb332a851db4126cb0dc33ddcdf33ff15148ff2a53",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 279513,
        "outputTokens": 1473,
        "tokenMethod": "provider",
        "toolCalls": 7
      },
      "contextBytes": 1621,
      "evidenceIds": [
        "evidence-bb79e1095a8116baada55af32de7cece",
        "evidence-839480e522bbb39a7a2029e38b9077ea",
        "acceptance-check:node bin/ak-docs.js audit documentation --json attempted and blocked with EPERM while creating .doc-bridge/workflow/.lock"
      ],
      "round": "structured-provider-normalized-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "documentationFindings": 1,
        "confidence": 0.98,
        "acceptanceChecksAttempted": 1,
        "acceptanceChecksPassed": 0,
        "acceptanceChecksBlocked": 1,
        "filesCompared": 5,
        "cachedInputTokens": 236544,
        "reasoningOutputTokens": 289,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "410eff87459e246d999b543d4c090dc5cff8a35d0e037bd2fadcc17394a10d05",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:38:58.835Z",
      "runId": "phase-6-structured-provider-normalized-01",
      "planHash": "8903a96835128a57d5d7981a0fb81a7ef8823b4603294a3ce7a196eb19fb258c",
      "task": {
        "taskId": "consumer-06-implementation",
        "repositoryId": "consumer-06",
        "category": "implementation",
        "scenarioId": "registry-assisted",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "a0a90a4141327fef23e2d7ef99da79cd44cfe66fa65d7757484e95629b3fe7bd"
      },
      "scenario": {
        "id": "registry-assisted",
        "agentId": "ecosystem-doc-bridge-corpus-scanner",
        "agentVersion": "v1.0.0",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 58617,
        "responseBytes": 766,
        "stderrBytes": 0,
        "stdoutHash": "09b5ff9b6551d50cca7d49d2aaaba5a7bd020b73c9358d09c454adde1c62f84a",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 364671,
        "outputTokens": 2197,
        "tokenMethod": "provider",
        "toolCalls": 8
      },
      "contextBytes": 1659,
      "evidenceIds": [
        "patch-evidence:docs/spec/config-v1.md:836 — remove the stale “planned” qualifier from the ak-docs chat mapping; current implementation is evidenced by src/cli/program.ts:20,1598 and src/intelligence/chat.ts:75",
        "verification-plan:after approval, run ak-docs check --json and review the resulting documentation report"
      ],
      "round": "structured-provider-normalized-2026-08-31",
      "taskOutcome": "success",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "documentation_files_in_scope": 1,
        "proposed_changed_lines": 1,
        "source_code_files_changed": 0,
        "files_modified": 0,
        "acceptance_checks_executed": 0,
        "cachedInputTokens": 302080,
        "reasoningOutputTokens": 1016,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "b237955c45f63dfbe0be8a5494d57e4518db4735d36276ec996a8e215143a24d",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:43:44.043Z",
      "runId": "phase-7-canonical-metrics-01",
      "planHash": "29e95cc2d8fa7f3e6a789884943f9d2cf20896ddb5405196b5ae869e87577d28",
      "task": {
        "taskId": "consumer-01-discovery",
        "repositoryId": "consumer-01",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 26567,
        "responseBytes": 581,
        "stderrBytes": 0,
        "stdoutHash": "7ca7a7e6f938720bfb87954d1d7e41fc034d4b0fb31f3c09aa585b42558296e8",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 68806,
        "outputTokens": 509,
        "tokenMethod": "provider",
        "toolCalls": 2
      },
      "contextBytes": 1855,
      "evidenceIds": [
        "entrypoint-evidence:package.json",
        "entrypoint-evidence:README.md",
        "entrypoint-evidence:AGENTS.md",
        "entrypoint-evidence:CONTRIBUTING.md",
        "acceptance:ak-docs-discover-command-not-found"
      ],
      "round": "canonical-metrics-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 4,
        "cachedInputTokens": 42496,
        "reasoningOutputTokens": 175,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "60f33cb65e734398ec42a64fd51b47520ade08871c00462d12a24e8b91ca0fd0",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:45:06.483Z",
      "runId": "phase-7-canonical-metrics-01",
      "planHash": "29e95cc2d8fa7f3e6a789884943f9d2cf20896ddb5405196b5ae869e87577d28",
      "task": {
        "taskId": "consumer-01-architecture",
        "repositoryId": "consumer-01",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 82424,
        "responseBytes": 965,
        "stderrBytes": 0,
        "stdoutHash": "d4afaeac6ebbe231df82b19c017e7dba25bf41a773ec7ff404469f804d5fb6a5",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 448237,
        "outputTokens": 3203,
        "tokenMethod": "provider",
        "toolCalls": 9,
        "errorCode": "token-budget"
      },
      "contextBytes": 1884,
      "evidenceIds": [
        "architecture-evidence",
        "artifact:.doc-bridge/index.json",
        "artifact:docs/architecture/adrs/0009-composition-rules.md",
        "artifact:docs/architecture/adrs/0006-runtime-contract.md",
        "artifact:docs/architecture/adrs/0031-cli-adapter-execution-boundary.md",
        "observed-relation:@agentskit/core->adapters,runtime,tools,memory,rag,skills,react,ink",
        "observed-relation:@agentskit/runtime->@agentskit/core",
        "observed-relation:@agentskit/cli->adapters,core,ink,integrations,memory,rag,runtime,skills,tools",
        "attention-point:CLI adapter process execution is Node-only; trusted-local is explicitly not an isolation boundary"
      ],
      "round": "canonical-metrics-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 384512,
        "reasoningOutputTokens": 1124,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "9976b1dbc63270059060c87687162227e84497927461eb8add99f0eab5171cc4",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:46:06.805Z",
      "runId": "phase-7-canonical-metrics-01",
      "planHash": "29e95cc2d8fa7f3e6a789884943f9d2cf20896ddb5405196b5ae869e87577d28",
      "task": {
        "taskId": "consumer-01-documentation",
        "repositoryId": "consumer-01",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 60306,
        "responseBytes": 808,
        "stderrBytes": 0,
        "stdoutHash": "bc45f96c5c6ffa7ab814730450a7ab038c9b28f09d29f5f1eb8102e00e2df9bc",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 152622,
        "outputTokens": 1232,
        "tokenMethod": "provider",
        "toolCalls": 4
      },
      "contextBytes": 2056,
      "evidenceIds": [
        "documentation-evidence:AGENTS.md:11 claims @agentskit/core is v1.0.0; packages/core/package.json reports version 1.12.8; classify as stale with high confidence and recommend updating the status claim or generating it from package metadata.",
        "review-limitation:The required audit command is unavailable in the installed ak-docs CLI, so this semantic finding was established by read-only comparison rather than audit output."
      ],
      "round": "canonical-metrics-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 1,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 127488,
        "reasoningOutputTokens": 514,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "8de5223c1e74d759c82117ac74db09463467041c277507f990380f074bb268f9",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:47:04.830Z",
      "runId": "phase-7-canonical-metrics-01",
      "planHash": "29e95cc2d8fa7f3e6a789884943f9d2cf20896ddb5405196b5ae869e87577d28",
      "task": {
        "taskId": "consumer-01-implementation",
        "repositoryId": "consumer-01",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 58012,
        "responseBytes": 506,
        "stderrBytes": 0,
        "stdoutHash": "34e3163a524eea55980abed09c30d6d52d007083ab2521a74d38bc15d037af71",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 451232,
        "outputTokens": 2100,
        "tokenMethod": "provider",
        "toolCalls": 10,
        "errorCode": "token-budget"
      },
      "contextBytes": 2094,
      "evidenceIds": [
        "apps/docs-next/content/docs/reference/packages/observability.mdx:maxUsd/onExceed-conflict",
        "packages/observability/src/cost-guard.ts:budgetUsd/onExceeded",
        "verification:ak-docs-check-json-after-approval"
      ],
      "round": "canonical-metrics-2026-08-31",
      "taskOutcome": "success",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "cachedInputTokens": 367616,
        "reasoningOutputTokens": 755,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "87bff6f51d92fd47908dc7323da98694ca20591f518a2dfbf23bed4395ff71ac",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:47:36.810Z",
      "runId": "phase-7-canonical-metrics-01",
      "planHash": "29e95cc2d8fa7f3e6a789884943f9d2cf20896ddb5405196b5ae869e87577d28",
      "task": {
        "taskId": "consumer-02-discovery",
        "repositoryId": "consumer-02",
        "category": "discovery",
        "scenarioId": "registry-assisted",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "registry-assisted",
        "agentId": "ecosystem-doc-bridge-corpus-scanner",
        "agentVersion": "v1.0.0",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 31968,
        "responseBytes": 520,
        "stderrBytes": 0,
        "stdoutHash": "bce103fb4017254fb6d1d301cfacf3cfcf260bef0ef2d807e77a2863c729cf18",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 73377,
        "outputTokens": 650,
        "tokenMethod": "provider",
        "toolCalls": 2
      },
      "contextBytes": 1857,
      "evidenceIds": [
        "package.json",
        "pnpm-workspace.yaml",
        "README.md",
        "AGENTS.md",
        "docs/for-agents/index.md",
        "docs/for-agents/architecture.md",
        "docs/architecture/overview.md"
      ],
      "round": "canonical-metrics-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 42496,
        "reasoningOutputTokens": 220,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "2afc0f87ee186774b4b782262935334fac43acbf335b0a26ef83e73e334b7128",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:48:16.459Z",
      "runId": "phase-7-canonical-metrics-01",
      "planHash": "29e95cc2d8fa7f3e6a789884943f9d2cf20896ddb5405196b5ae869e87577d28",
      "task": {
        "taskId": "consumer-02-architecture",
        "repositoryId": "consumer-02",
        "category": "architecture",
        "scenarioId": "registry-assisted",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "registry-assisted",
        "agentId": "ecosystem-doc-bridge-corpus-scanner",
        "agentVersion": "v1.0.0",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 39637,
        "responseBytes": 714,
        "stderrBytes": 0,
        "stdoutHash": "9e62bd103c8aeafae566ff7ed02886484c5f1cbead9c19014e458050442c74f2",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 178926,
        "outputTokens": 1465,
        "tokenMethod": "provider",
        "toolCalls": 5
      },
      "contextBytes": 1886,
      "evidenceIds": [
        "artifact:docs/architecture/overview.md#product-boundary",
        "artifact:docs/architecture/overview.md#reference-turn-pipeline",
        "artifact:packages/chat/package.json#dependencies",
        "artifact:packages/server/src/index.ts#handler-controller-session-boundary",
        "artifact:packages/react/src/index.tsx#upstream-renderer-binding",
        "acceptance:ak-docs-map-command-not-found"
      ],
      "round": "canonical-metrics-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 127744,
        "reasoningOutputTokens": 668,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "33373cf3f2e3d6047dde92c68eb11140c37537103758e4bce481f1e598e72b35",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:49:38.882Z",
      "runId": "phase-7-canonical-metrics-01",
      "planHash": "29e95cc2d8fa7f3e6a789884943f9d2cf20896ddb5405196b5ae869e87577d28",
      "task": {
        "taskId": "consumer-02-documentation",
        "repositoryId": "consumer-02",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 82410,
        "responseBytes": 444,
        "stderrBytes": 0,
        "stdoutHash": "4f7ff82693d218295b23db01b524f6739161056c38e957c1f09f3adefab6188b",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 254962,
        "outputTokens": 1729,
        "tokenMethod": "provider",
        "toolCalls": 8
      },
      "contextBytes": 2047,
      "evidenceIds": [
        "documentation-evidence",
        "review-limitation"
      ],
      "round": "canonical-metrics-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 218880,
        "reasoningOutputTokens": 535,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "db69c5571261e28cae1d9ab47ce88b545549e7d86bfd061a49fc9633be97a47d",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:51:15.670Z",
      "runId": "phase-7-canonical-metrics-01",
      "planHash": "29e95cc2d8fa7f3e6a789884943f9d2cf20896ddb5405196b5ae869e87577d28",
      "task": {
        "taskId": "consumer-02-implementation",
        "repositoryId": "consumer-02",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 96775,
        "responseBytes": 543,
        "stderrBytes": 0,
        "stdoutHash": "055f432842e3b914ce7581f98dc9152e24fefe19a83d77e45ed750b612541c77",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 893818,
        "outputTokens": 3662,
        "tokenMethod": "provider",
        "toolCalls": 14,
        "errorCode": "token-budget"
      },
      "contextBytes": 2085,
      "evidenceIds": [
        "patch-evidence:docs/for-agents/index.md",
        "artifact:reconcile-915252b168168470a2747278f2352d0035b4ee762982d067ec765d020356a83d",
        "verification-plan:ak-docs check --json blocked by EPERM"
      ],
      "round": "canonical-metrics-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 812288,
        "reasoningOutputTokens": 1304,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "013a6b01fee02f2d7daa26da3151c68ab7eb061b0d54f003aa3a6581740e4b0a",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:51:43.861Z",
      "runId": "phase-7-canonical-metrics-01",
      "planHash": "29e95cc2d8fa7f3e6a789884943f9d2cf20896ddb5405196b5ae869e87577d28",
      "task": {
        "taskId": "consumer-03-discovery",
        "repositoryId": "consumer-03",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 28178,
        "responseBytes": 557,
        "stderrBytes": 0,
        "stdoutHash": "edc5955c8353e5d521d19497af96d23eeadd5c48b53babb9dd7c24b754ce5f6f",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 80619,
        "outputTokens": 522,
        "tokenMethod": "provider",
        "toolCalls": 2
      },
      "contextBytes": 1864,
      "evidenceIds": [
        "entrypoint-evidence:AGENTS.md",
        "entrypoint-evidence:README.md",
        "entrypoint-evidence:package.json",
        "entrypoint-evidence:docs/for-agents/INDEX.md",
        "acceptance:ak-docs-discover-command-not-found"
      ],
      "round": "canonical-metrics-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "low",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 61696,
        "reasoningOutputTokens": 204,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "cb2c426d8242002e1befa4ecdd43b6f4c1d6d3a2e27c2c7947bb56607d9fa2fb",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:52:50.412Z",
      "runId": "phase-7-canonical-metrics-01",
      "planHash": "29e95cc2d8fa7f3e6a789884943f9d2cf20896ddb5405196b5ae869e87577d28",
      "task": {
        "taskId": "consumer-03-architecture",
        "repositoryId": "consumer-03",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 66535,
        "responseBytes": 1079,
        "stderrBytes": 0,
        "stdoutHash": "b57746d7ef7c870ad175bc5ce66ca4d6fe7c16a9328590e07d37abe60756569a",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 635825,
        "outputTokens": 2565,
        "tokenMethod": "provider",
        "toolCalls": 9,
        "errorCode": "token-budget"
      },
      "contextBytes": 1893,
      "evidenceIds": [
        "architecture:apps-console->desktop-shell->os-desktop->desktop-screen-cluster",
        "architecture:desktop-screen-cluster->desktop-sidecar-bridge->os-headless->os-contracts->os-core",
        "architecture:os-headless->os-runtime->os-flow",
        "architecture:os-headless->os-storage-or-os-store-postgres",
        "architecture:apps-cloud->os-cloud-sync-and-os-core",
        "attention:desktop-sidecar-bridge-json-rpc-boundary-layer-enforcement-and-no-reverse-deps",
        "artifact:docs-for-agents-architecture.md",
        "artifact:docs-for-agents-packages-os-contracts.md",
        "artifact:docs-for-agents-packages-os-headless.md",
        "artifact:adr-0107-desktop-sidecar-bridge-contract.md",
        "artifact:package-json-workspace-dependencies",
        "acceptance:ak-docs-map-json-blocked-by-EPERM"
      ],
      "round": "canonical-metrics-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 546304,
        "reasoningOutputTokens": 809,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "559ee964da1d0ebb8559bba9349eb0be5371442e447974a82cde66903e28c090",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:54:58.371Z",
      "runId": "phase-7-canonical-metrics-01",
      "planHash": "29e95cc2d8fa7f3e6a789884943f9d2cf20896ddb5405196b5ae869e87577d28",
      "task": {
        "taskId": "consumer-03-documentation",
        "repositoryId": "consumer-03",
        "category": "documentation",
        "scenarioId": "registry-assisted",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "registry-assisted",
        "agentId": "ecosystem-doc-bridge-corpus-scanner",
        "agentVersion": "v1.0.0",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 127945,
        "responseBytes": 428,
        "stderrBytes": 0,
        "stdoutHash": "c342a00999778fc5fa6e5872b95d4dce2da65ffd9022ee7abff96b459114e71f",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 207713,
        "outputTokens": 1300,
        "tokenMethod": "provider",
        "toolCalls": 6
      },
      "contextBytes": 2049,
      "evidenceIds": [
        "documentation-evidence",
        "review-limitation"
      ],
      "round": "canonical-metrics-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 170496,
        "reasoningOutputTokens": 394,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "dbcc7734c8b486ab92df3be7bf8f5c2943dc08abb960a7a5e91d8b301f3f4d73",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:56:14.180Z",
      "runId": "phase-7-canonical-metrics-01",
      "planHash": "29e95cc2d8fa7f3e6a789884943f9d2cf20896ddb5405196b5ae869e87577d28",
      "task": {
        "taskId": "consumer-03-implementation",
        "repositoryId": "consumer-03",
        "category": "implementation",
        "scenarioId": "registry-assisted",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "registry-assisted",
        "agentId": "ecosystem-doc-bridge-corpus-scanner",
        "agentVersion": "v1.0.0",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 75793,
        "responseBytes": 661,
        "stderrBytes": 0,
        "stdoutHash": "d61ea38eba705979389da3417ee65f285cbb8f570ac401415c72acac6e645278",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 508785,
        "outputTokens": 3067,
        "tokenMethod": "provider",
        "toolCalls": 11,
        "errorCode": "token-budget"
      },
      "contextBytes": 2087,
      "evidenceIds": [
        "patch-evidence:docs/rfc/0056-doc-bridge-ecosystem-dogfood.md:28-39 omits ak-docs check --json from the operational route; proposed documentation-only addition",
        "verification-plan:run ak-docs check --json after approval",
        "audit-finding:DOCUMENTATION_SEMANTICS_NOT_ANALYZED"
      ],
      "round": "canonical-metrics-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 439552,
        "reasoningOutputTokens": 1543,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "70ec7779b948506d9ea84c6b94f6ef7d9fd82ce184061c29ae4a12edec78be63",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:56:43.855Z",
      "runId": "phase-7-canonical-metrics-01",
      "planHash": "29e95cc2d8fa7f3e6a789884943f9d2cf20896ddb5405196b5ae869e87577d28",
      "task": {
        "taskId": "consumer-04-discovery",
        "repositoryId": "consumer-04",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 29660,
        "responseBytes": 590,
        "stderrBytes": 0,
        "stdoutHash": "6dcf9ced85c68d5e48bdf45509061595c35bd07ea214afe89a7cb72e1b09a48f",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 72578,
        "outputTokens": 620,
        "tokenMethod": "provider",
        "toolCalls": 2
      },
      "contextBytes": 1855,
      "evidenceIds": [
        "entrypoint-evidence:docs/for-agents/index.md",
        "entrypoint-evidence:docs/architecture.md",
        "entrypoint-evidence:CONTRIBUTING.md",
        "entrypoint-evidence:package.json",
        "entrypoint-evidence:scripts/generate-doc-bridge-config.mjs"
      ],
      "round": "canonical-metrics-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 40448,
        "reasoningOutputTokens": 205,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "de0e8bec380bc3059a4926381f5c026869e42287f1d105c13c2b3feca3f121c6",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:57:35.408Z",
      "runId": "phase-7-canonical-metrics-01",
      "planHash": "29e95cc2d8fa7f3e6a789884943f9d2cf20896ddb5405196b5ae869e87577d28",
      "task": {
        "taskId": "consumer-04-architecture",
        "repositoryId": "consumer-04",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 51540,
        "responseBytes": 465,
        "stderrBytes": 0,
        "stdoutHash": "3cabddc9e35cff72948a4f86ca0ad1583deaaca524ae14463ece9c576cf22710",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 232335,
        "outputTokens": 1983,
        "tokenMethod": "provider",
        "toolCalls": 7
      },
      "contextBytes": 1884,
      "evidenceIds": [
        "docs/architecture.md",
        "package.json",
        "pnpm-workspace.yaml",
        "doc-bridge.config.json",
        "ak-docs-map:EPERM-lock"
      ],
      "round": "canonical-metrics-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 203776,
        "reasoningOutputTokens": 923,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "379835fa5e8ba18ed0b4039c1e163b3da08db37465a4c124f965cc103f3e2839",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T16:59:14.118Z",
      "runId": "phase-7-canonical-metrics-01",
      "planHash": "29e95cc2d8fa7f3e6a789884943f9d2cf20896ddb5405196b5ae869e87577d28",
      "task": {
        "taskId": "consumer-04-documentation",
        "repositoryId": "consumer-04",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 98698,
        "responseBytes": 830,
        "stderrBytes": 0,
        "stdoutHash": "17f026dc55aa5caaea21156a78ca07d4d06c3741db23fa173ad5de2947fa0f33",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 274024,
        "outputTokens": 2179,
        "tokenMethod": "provider",
        "toolCalls": 9
      },
      "contextBytes": 2056,
      "evidenceIds": [
        "documentation-evidence:contradictory:docs/getting-started.md:40-41:says-all-346-evals-are-replayed",
        "source-evidence:package.json:19:eval-run-requires-explicit-selector",
        "source-evidence:scripts/run-eval.mjs:39-42:no-selector-exits-with-usage-error",
        "review-limitation:audit-blocked-by-read-only-filesystem-at-.doc-bridge/workflow/.lock",
        "recommended-next-action:clarify-doc-claim-or-add-and-document-an-all-agents-eval-selector"
      ],
      "round": "canonical-metrics-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 233216,
        "reasoningOutputTokens": 905,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "eb28d6494c3f2ae65f77a06ecfa97c64fa37a82b06efe6a128ba4df36133c03d",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:00:27.773Z",
      "runId": "phase-7-canonical-metrics-01",
      "planHash": "29e95cc2d8fa7f3e6a789884943f9d2cf20896ddb5405196b5ae869e87577d28",
      "task": {
        "taskId": "consumer-04-implementation",
        "repositoryId": "consumer-04",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 73640,
        "responseBytes": 746,
        "stderrBytes": 0,
        "stdoutHash": "a5bf19ed1f3bd9a621ba074124d86b1fef3a2db24c16ab729afd2a3e633be4eb",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 424908,
        "outputTokens": 3070,
        "tokenMethod": "provider",
        "toolCalls": 10,
        "errorCode": "token-budget"
      },
      "contextBytes": 2094,
      "evidenceIds": [
        "patch-evidence:docs/for-agents/index.md:26-36 omits verification-contract checks docs:bridge:conformance and check:readme-standard",
        "proposal:docs/for-agents/index.md:add npm run docs:bridge:conformance and npm run check:readme-standard",
        "verification-plan:ak-docs check --json",
        "validation-blocker:ak-docs unavailable exit-127",
        "source-revision:341b22ebd72838137f644e731de2f7d52607ed9b"
      ],
      "round": "canonical-metrics-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 372992,
        "reasoningOutputTokens": 1241,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "2c9c654b41175811a6eb070d13baeb55b9205a4bb5eda746286dd9cec1af0bf6",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:01:07.775Z",
      "runId": "phase-7-canonical-metrics-01",
      "planHash": "29e95cc2d8fa7f3e6a789884943f9d2cf20896ddb5405196b5ae869e87577d28",
      "task": {
        "taskId": "consumer-05-discovery",
        "repositoryId": "consumer-05",
        "category": "discovery",
        "scenarioId": "registry-assisted",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "registry-assisted",
        "agentId": "ecosystem-doc-bridge-corpus-scanner",
        "agentVersion": "v1.0.0",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 39989,
        "responseBytes": 643,
        "stderrBytes": 0,
        "stdoutHash": "657f4aaa6523e1b03f32ed044486f3c5d75f01bd3208e7f57b396cbdf5f9e755",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 90660,
        "outputTokens": 781,
        "tokenMethod": "provider",
        "toolCalls": 3
      },
      "contextBytes": 1857,
      "evidenceIds": [
        "entrypoint-evidence:package.json",
        "entrypoint-evidence:README.md",
        "entrypoint-evidence:AGENTS.md",
        "entrypoint-evidence:app/page.tsx",
        "entrypoint-evidence:app/docs/[[...slug]]/page.tsx",
        "entrypoint-evidence:docs/for-agents/INDEX.md",
        "discovery-check:ak-docs-command-not-found"
      ],
      "round": "canonical-metrics-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 61696,
        "reasoningOutputTokens": 276,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "bbaf6b2077e3fce8618c750194815b9521441ebaeb329b3498c6cdd9b1ec6722",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:02:02.439Z",
      "runId": "phase-7-canonical-metrics-01",
      "planHash": "29e95cc2d8fa7f3e6a789884943f9d2cf20896ddb5405196b5ae869e87577d28",
      "task": {
        "taskId": "consumer-05-architecture",
        "repositoryId": "consumer-05",
        "category": "architecture",
        "scenarioId": "registry-assisted",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "registry-assisted",
        "agentId": "ecosystem-doc-bridge-corpus-scanner",
        "agentVersion": "v1.0.0",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 54650,
        "responseBytes": 600,
        "stderrBytes": 0,
        "stdoutHash": "664c2aab518be40a823e2b58b5ad81f73f4fc523647f496be446477d87e8a759",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 376080,
        "outputTokens": 2043,
        "tokenMethod": "provider",
        "toolCalls": 9
      },
      "contextBytes": 1886,
      "evidenceIds": [
        "architecture-evidence",
        "packages/harness/src/index.ts",
        "packages/harness/src/verification.ts",
        "packages/harness/src/context.ts",
        "packages/harness/src/plugins.ts",
        "packages/harness/README.md",
        ".doc-bridge/index.json",
        "ak-docs-map-blocked-EPERM"
      ],
      "round": "canonical-metrics-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 314880,
        "reasoningOutputTokens": 566,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "f140bf142e7b5ea26c29f6e61687e47b6576bd552217c39479bd6d623ab913d7",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:03:15.921Z",
      "runId": "phase-7-canonical-metrics-01",
      "planHash": "29e95cc2d8fa7f3e6a789884943f9d2cf20896ddb5405196b5ae869e87577d28",
      "task": {
        "taskId": "consumer-05-documentation",
        "repositoryId": "consumer-05",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 73464,
        "responseBytes": 820,
        "stderrBytes": 0,
        "stdoutHash": "75e704d67c08b9ac54d327476beaa93f71ac9b96ade3cdf70ed014de960a0b05",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 226088,
        "outputTokens": 1498,
        "tokenMethod": "provider",
        "toolCalls": 7
      },
      "contextBytes": 2047,
      "evidenceIds": [
        "evidence-e2a6074a088d214f404e7e65afe145b4",
        "review-limitation:medium-confidence:semantic classification is based on repository source and the existing generated index; the required audit could not complete because read-only execution blocked creation of .doc-bridge/workflow/.lock",
        "acceptance-check:documentation-check:blocked:ak-docs audit documentation --json exited 2 with EPERM creating .doc-bridge/workflow/.lock"
      ],
      "round": "canonical-metrics-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 165376,
        "reasoningOutputTokens": 399,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "b817aad733b89f5de359dda7c239b4c6c637037f552d00a1dcfc143ca816f71e",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:04:02.150Z",
      "runId": "phase-7-canonical-metrics-01",
      "planHash": "29e95cc2d8fa7f3e6a789884943f9d2cf20896ddb5405196b5ae869e87577d28",
      "task": {
        "taskId": "consumer-05-implementation",
        "repositoryId": "consumer-05",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 46204,
        "responseBytes": 383,
        "stderrBytes": 0,
        "stdoutHash": "a7807f128e164289de81df9d120551a8265f8b06242a40ddb08a2849805915a1",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 311831,
        "outputTokens": 1658,
        "tokenMethod": "provider",
        "toolCalls": 7
      },
      "contextBytes": 2085,
      "evidenceIds": [],
      "round": "canonical-metrics-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "low",
      "safetyOutcome": "safe",
      "clarificationRequests": 1,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "documentationFindingCount": 0,
        "cachedInputTokens": 263168,
        "reasoningOutputTokens": 682,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "0b093341b2d50438059fad71fbcff39b3c4ad44f3251dec092af40db271b2493",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:04:34.766Z",
      "runId": "phase-7-canonical-metrics-01",
      "planHash": "29e95cc2d8fa7f3e6a789884943f9d2cf20896ddb5405196b5ae869e87577d28",
      "task": {
        "taskId": "consumer-06-discovery",
        "repositoryId": "consumer-06",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 32582,
        "responseBytes": 431,
        "stderrBytes": 0,
        "stdoutHash": "5fdf6f4ce2d6551a45a8b849232ebb6bd2ce01ef05d88cdca278ee9e98be602e",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 90081,
        "outputTokens": 566,
        "tokenMethod": "provider",
        "toolCalls": 3
      },
      "contextBytes": 1864,
      "evidenceIds": [
        "package.json",
        "README.md",
        "docs/index.md",
        "apps/docs/AGENTS.md"
      ],
      "round": "canonical-metrics-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 61696,
        "reasoningOutputTokens": 160,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "934f81bf90af464f46532680b34cda4d0ce91be96f7c1b2f7d2ae49ce9109f8a",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:05:53.744Z",
      "runId": "phase-7-canonical-metrics-01",
      "planHash": "29e95cc2d8fa7f3e6a789884943f9d2cf20896ddb5405196b5ae869e87577d28",
      "task": {
        "taskId": "consumer-06-architecture",
        "repositoryId": "consumer-06",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 78934,
        "responseBytes": 1397,
        "stderrBytes": 0,
        "stdoutHash": "d4a475f7e4f2bff6dfec5743fb03d97c583733265b5751ea47b2624f683d3e6e",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 547126,
        "outputTokens": 2843,
        "tokenMethod": "provider",
        "toolCalls": 11,
        "errorCode": "token-budget"
      },
      "contextBytes": 1893,
      "evidenceIds": [
        "architecture-evidence: .doc-bridge/workflow/artifacts/normalize-50e826ab8cd8e98c3f507d62e5e2cdbf53d8b394267e6ca214671b5219b3d7ce.json; 368 entities and 1147 relations",
        "component: src/cli/program.ts -> src/index-builder/build-index.ts, src/discovery/repository.ts, src/reconciliation/reconcile.ts, src/query/query.ts, src/mcp/server.ts",
        "component: src/index-builder/build-index.ts -> corpus scanning, package discovery, handoff/index generation, llms.txt and capabilities outputs",
        "component: src/discovery/repository.ts -> static JS/TS module import and export discovery plus workspace metadata",
        "component: src/reconciliation/reconcile.ts -> observed-versus-declared relation comparison and diagnostics",
        "attention: dynamic-import coverage is partial; non-literal dynamic imports remain unresolved in src/agents/registry-adapter.ts:140, src/intelligence/peers.ts:50, and .mcpb-runtime/ak-docs.js",
        "acceptance-command: node bin/ak-docs.js map --json failed with EPERM creating .doc-bridge/workflow/.lock; ak-docs was unavailable on PATH"
      ],
      "round": "canonical-metrics-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 423936,
        "reasoningOutputTokens": 730,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "06d7cf774340dff9e61bf8dc4c1fdc79c00e857c4be94624b666c53284d83049",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:06:53.982Z",
      "runId": "phase-7-canonical-metrics-01",
      "planHash": "29e95cc2d8fa7f3e6a789884943f9d2cf20896ddb5405196b5ae869e87577d28",
      "task": {
        "taskId": "consumer-06-documentation",
        "repositoryId": "consumer-06",
        "category": "documentation",
        "scenarioId": "registry-assisted",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "registry-assisted",
        "agentId": "ecosystem-doc-bridge-corpus-scanner",
        "agentVersion": "v1.0.0",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 60217,
        "responseBytes": 644,
        "stderrBytes": 0,
        "stdoutHash": "96e6703fa97fabb509519d60541beda7b08f933757fb58d0d5250ba5afd8e1e9",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 185833,
        "outputTokens": 1235,
        "tokenMethod": "provider",
        "toolCalls": 6
      },
      "contextBytes": 2049,
      "evidenceIds": [
        "evidence-d6644944c2f9afede3d97d3bafda1401",
        "review-limitation:semantic classification manually established; configured audit could not run because ak-docs was unavailable and the source-tree fallback required a prohibited workspace lock write"
      ],
      "round": "canonical-metrics-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 135680,
        "reasoningOutputTokens": 387,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "3158c612a900495184472772d1d82a8c690822b322a566794f4824a4b9033773",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:07:40.508Z",
      "runId": "phase-7-canonical-metrics-01",
      "planHash": "29e95cc2d8fa7f3e6a789884943f9d2cf20896ddb5405196b5ae869e87577d28",
      "task": {
        "taskId": "consumer-06-implementation",
        "repositoryId": "consumer-06",
        "category": "implementation",
        "scenarioId": "registry-assisted",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "registry-assisted",
        "agentId": "ecosystem-doc-bridge-corpus-scanner",
        "agentVersion": "v1.0.0",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 46498,
        "responseBytes": 955,
        "stderrBytes": 0,
        "stdoutHash": "e62c75c1452b1d18bf1ba7e4e3563a86a0532fcf8d11625910f1808e16a49ee1",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 345861,
        "outputTokens": 1764,
        "tokenMethod": "provider",
        "toolCalls": 8
      },
      "contextBytes": 2087,
      "evidenceIds": [
        "patch-evidence:docs/spec/config-v1.md:836 — replace the stale `ak-docs chat | planned;` claim with an implemented/available chat command entry, preserving the existing CLI mapping table convention",
        "patch-evidence:src/cli/program.ts:20,1598 — chat command is implemented",
        "patch-evidence:src/intelligence/chat.ts:75 — runChatOnce implementation",
        "verification-plan:ak-docs check --json after approval",
        "acceptance-result:ak-docs check --json could not run because the binary was unavailable; pnpm exec ak-docs check --json was blocked by EPERM during pnpm install"
      ],
      "round": "canonical-metrics-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 288768,
        "reasoningOutputTokens": 654,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "ad63eaa02c9af046fcdc970abe13fb6d28f6d64556dc2cbba4665a48b91c1e4b",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:21.162Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-05-discovery",
        "repositoryId": "consumer-05",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 102,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1865,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "b9bc85a53cc900779cafc6bf64ecfc07214997a7f3642290450a65dca29dddc4",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:21.310Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-03-implementation",
        "repositoryId": "consumer-03",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 123,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2084,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "05287982eff18e900ee204024ee89b2782dcae63acb92585d417f913b65205e5",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:21.467Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-06-implementation",
        "repositoryId": "consumer-06",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 130,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2085,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "388c379d79ac80bb1366fe72cee93fa24bc3926086320c49f6496862dde1f8fd",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:21.603Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-01-discovery",
        "repositoryId": "consumer-01",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 122,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1856,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "26d007293237a0d36a09f3023aa27432b86283bfdf3d8385af655cdcedddbd32",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:21.781Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-01-implementation",
        "repositoryId": "consumer-01",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 145,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2094,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "b153942f5f7e64064293606d1b0b01f83eab4f13b4545677fd2f542a3970dbd3",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:21.909Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-04-documentation",
        "repositoryId": "consumer-04",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 112,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2048,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "f45ded396d90e8c686373ae9b6eeee105bc3e2035ca126f28cd8d0080b92c44f",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:22.050Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-01-implementation",
        "repositoryId": "consumer-01",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 129,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2085,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "542e2815f845f1be60f1e28ebc5f55f71015b3dfdfb2e8636fb68ce99a7b2144",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:22.187Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-03-discovery",
        "repositoryId": "consumer-03",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 120,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1864,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "94705d1b3cf3494af1148df1692b1677719cf55e2ba7ef8ad089e2fb9eca8cb1",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:22.361Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-06-discovery",
        "repositoryId": "consumer-06",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 159,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1864,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "84208a09e96218bee42d63f018a0750b8e10119a2d80983e3d6fc85bd71151a2",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:22.490Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-05-architecture",
        "repositoryId": "consumer-05",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 119,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1884,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "0f17d338d8a5aa7f672e656f5ff6f7e8a926b35ff81d7e99afbd8aad9d4b5bee",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:22.614Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-06-discovery",
        "repositoryId": "consumer-06",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 112,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1856,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "714c6c07a20bf8c8b8fb8d49f0f2bffebb1a76b1ae3e097d9738173e4c3008bd",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:22.739Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-06-implementation",
        "repositoryId": "consumer-06",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 110,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2084,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "27859ac0d79dcffc208292cb3f6ca898ca10091b6f9113d4a7244b44d9165426",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:22.876Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-05-documentation",
        "repositoryId": "consumer-05",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 122,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2056,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "c9e35948ef3ed2a3967298ff0fc588fa3f3eb58d0da3b27bf27d8bd71827b929",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:22.965Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-02-architecture",
        "repositoryId": "consumer-02",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 77,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1883,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "914cc41262e6653e972427040a645e4fa171d177e17fc8211070eba405c02d21",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:23.086Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-05-documentation",
        "repositoryId": "consumer-05",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 101,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2057,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "df32e194c0c26a665a55dc522f5000cbf4f077a49d3e5ebc260f63d3003edbaf",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:23.177Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-06-architecture",
        "repositoryId": "consumer-06",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 80,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1893,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "f0e482fb78042463ad0d99cffdd84096c2d4ff29153c4ad6c3e4ad0e5cededc4",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:23.274Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-01-implementation",
        "repositoryId": "consumer-01",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 86,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2084,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "a822fcea6496d2aea9650e31103a4ef392b95b54a139ffea0489978f8c7c7498",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:23.352Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-01-architecture",
        "repositoryId": "consumer-01",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 67,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1893,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "ab26cb46199309ea8b54b1f2bba870157b0dd7ed0de40c45358d66f80bce7bf0",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:23.436Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-06-discovery",
        "repositoryId": "consumer-06",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 74,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1865,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "006460a4b08f0f5323be249f6989b936c7d8579b6771936417569b6bbc59b6a8",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:23.519Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-01-discovery",
        "repositoryId": "consumer-01",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 73,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1864,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "26fd75dd9bfe034c2c241123a912946695ddb622aef9b85cd90defc92c555c2a",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:23.592Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-01-architecture",
        "repositoryId": "consumer-01",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 63,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1884,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "11df15b3e41032aed626d27af898cb5f84de3228e463baebf761546092f9705a",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:23.668Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-04-architecture",
        "repositoryId": "consumer-04",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 65,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1893,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "95c73428b851acb0e9e0f16fed6c28b5c9bcfea5f0e3bd1eddd6ac657621d8f0",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:23.747Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-01-discovery",
        "repositoryId": "consumer-01",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 69,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1855,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "d0fa6c01dac2c498c879f6775b171b9590d8e12a6b0690e68c7fa35b663798dd",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:23.836Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-01-architecture",
        "repositoryId": "consumer-01",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 77,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1892,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "ec93987616bfa2059839e7c7d84cb1a12e614626bac72a3ff45261de2aa94de5",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:23.912Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-02-architecture",
        "repositoryId": "consumer-02",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 66,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1893,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "b0028ea0c74b8eb5c2eee7046c0bdb283622da5f187ed66d436a19e6c0e482e2",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:23.999Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-06-implementation",
        "repositoryId": "consumer-06",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 75,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2093,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "670bab3aa8f3070e0a713b160aa11df297e7a44cc0b72c48d87618ca7bae06e0",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:24.119Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-05-implementation",
        "repositoryId": "consumer-05",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 108,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2093,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "3353db3ffbc7cca8b8f35618ebafcb2eabd6264d7a87baafb77273f08985ff31",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:24.225Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-05-architecture",
        "repositoryId": "consumer-05",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 93,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1892,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "9f6cdc34266b5f3619739d63c583cc648bfff50c0421919b9be37410be9ea31e",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:24.341Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-05-architecture",
        "repositoryId": "consumer-05",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 106,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1883,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "78022c91d4b7815db4018ea1f8f5f4409975875a6d4a79dd49e2bb81f0354480",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:24.438Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-02-discovery",
        "repositoryId": "consumer-02",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 87,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1855,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "d50c17ac6f637661256fe24117ffc18ab4f54c0fce64441efddf7799c5ce00d9",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:24.548Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-02-architecture",
        "repositoryId": "consumer-02",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 97,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1892,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "abe63073ec0a7aa4adff2dbb67f388adab24ee078d3ef8fa5612476076f29329",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:24.696Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-03-implementation",
        "repositoryId": "consumer-03",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 134,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2093,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "fc938382dce215543c48684b341a3e9743625ade83aed4428b79f5896f425205",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:24.832Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-04-implementation",
        "repositoryId": "consumer-04",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 116,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2085,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "1b22a588026e04ca375a09c7a3f1a91637beaf456a2173c843027dbab008ec21",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:24.931Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-02-architecture",
        "repositoryId": "consumer-02",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 87,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1884,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "ea1f21dfed97d528f5fd1f31ab40abbb618e715997fefcce4c230aff0334fe36",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:25.028Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-03-discovery",
        "repositoryId": "consumer-03",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 84,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1855,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "06b906c2ce1bd0ddff3e17e7b168c262b56457523eb2dda1087bc66f9782dedd",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:25.146Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-02-documentation",
        "repositoryId": "consumer-02",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 105,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2057,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "c30fa3f6d4273e7da432ac55e73f490577800bab35ec59ecb2065fb00246e95b",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:25.258Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-06-architecture",
        "repositoryId": "consumer-06",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 98,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1892,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "a1f6bd4b721bd69290d32eee18cbad298c1e6c30dc11187b545a0149b42c4771",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:25.354Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-02-implementation",
        "repositoryId": "consumer-02",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 84,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2084,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "4981737c7eb8026c4d6fe421206dca3f3ac2a6612f8726e270b12401c6992345",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:25.462Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-06-documentation",
        "repositoryId": "consumer-06",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 95,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2047,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "982b0f5c00c986d44ea1843bb8774b65546d76166ea24fcb160e89bb3d267d2e",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:25.556Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-01-documentation",
        "repositoryId": "consumer-01",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 82,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2056,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "9089d75ddb1d99f51e867c1bc750e0fc309bc6dda1dcc429d860b8d1c4376e2a",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:25.643Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-06-documentation",
        "repositoryId": "consumer-06",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 76,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2048,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "9563c435ff6dc19d95abe537811682c33450b82a7f64b775416eb10eb431e77e",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:25.741Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-03-discovery",
        "repositoryId": "consumer-03",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 85,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1856,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "b0e3dd167f189f3508da4e843776f7b5a7a7d107f9181bf7c6ad0f8bdb991f72",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:25.820Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-02-discovery",
        "repositoryId": "consumer-02",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 69,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1856,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "f5f79285f24c88c53a13d56a4b0c5379c105f8af5893a75db27c77b3b5e7f081",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:25.890Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-05-discovery",
        "repositoryId": "consumer-05",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 60,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1855,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "7a7d391391a1f04654c106908df7e2db9524f9c4d4cb88c5dd52553d53fad0dc",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:25.968Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-03-architecture",
        "repositoryId": "consumer-03",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 66,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1893,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "337cbcfee9f751db674b84b2bd73b1356aa8d15938345bd55c2570e8d77943b7",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:26.044Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-01-architecture",
        "repositoryId": "consumer-01",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 65,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1883,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "9e4f61338ed19706b1f3aee0c40e67363571c68939c0878bc703882976146130",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:26.122Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-01-documentation",
        "repositoryId": "consumer-01",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 67,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2047,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "893f216849b4c37e4cb0c2f6b5b80487b70d3306bc80f314abed2f41df50a49e",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:26.202Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-05-implementation",
        "repositoryId": "consumer-05",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 68,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2094,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "198d4b0fe72ec0e803193a52f1ba0ccde3caf72f89d0f591eec9a2f62f483d49",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:26.282Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-05-implementation",
        "repositoryId": "consumer-05",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 68,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2085,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "5d0d52fdaf2c8456e1feab9e58549ee2f02e7629a95fdf7b7c20c2ff324f06e8",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:26.360Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-04-documentation",
        "repositoryId": "consumer-04",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 66,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2057,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "369ca3b7c8672e2769fdd9bc09da23d920ae59f7df020bd31c31e7a4e85df091",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:26.454Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-06-architecture",
        "repositoryId": "consumer-06",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 79,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1883,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "d6580eaca6d51fe17e00cbb7d661a290a191981a58142dfb5948d7a6fdc5cd82",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:26.545Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-05-discovery",
        "repositoryId": "consumer-05",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 77,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1864,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "9d720ba93c05ec0700c9bf987de072d37b43c0b574a7f09fe3d9f8d5e619f5a0",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:26.658Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-04-documentation",
        "repositoryId": "consumer-04",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 101,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2056,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "555f7e08f40f33930917e500a8d4e0589216b6fd6ed16985f529d25888f22e12",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:26.762Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-03-documentation",
        "repositoryId": "consumer-03",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 84,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2047,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "5edcd9162bab7d70c86a075f2a13ca7ec5d6d39a0d52618de9d27355670921eb",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:26.857Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-02-documentation",
        "repositoryId": "consumer-02",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 79,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2048,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "0ba729c22851f21bfa7e8e00af022ed728f0bbe497d0629b1c2ad1d9bdcb73a4",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:26.965Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-02-discovery",
        "repositoryId": "consumer-02",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 97,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1865,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "50fc626c86ad7b20d9bdef48f8402c3a09c1591418c305c2adf1f2a7e6d55400",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:27.063Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-04-implementation",
        "repositoryId": "consumer-04",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 86,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2094,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "b5d27f76c6fa8f5b311245b64ea5050bf74659d2a322c98036c0aaaf1b36ae67",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:27.167Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-01-documentation",
        "repositoryId": "consumer-01",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 90,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2048,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "220ea7062d794cf7a21a20b62e952032d23fb331e4cb1ac56908e6ddfaa4a9a0",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:27.277Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-03-implementation",
        "repositoryId": "consumer-03",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 96,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2085,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "25149ce86e4dd2eefc21da1b9d18ceb76a22e31be3e654d107ab076210538621",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:27.386Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-04-architecture",
        "repositoryId": "consumer-04",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 98,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1892,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "cad78d40f1fd80fc6b9cb98fec2fa679ca175acf9c2af8a473dbabb3ac4e899c",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:27.486Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-03-implementation",
        "repositoryId": "consumer-03",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 85,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2094,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "add6bd1f1035f63a9d91ac55bf4d674a956f12983e521a7c27bbe362cf557e64",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:27.579Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-03-architecture",
        "repositoryId": "consumer-03",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 82,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1883,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "a9f07d9912b60542cd5375a7adff1d92437cc7d860804ad82718b05298dcb011",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:27.663Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-04-architecture",
        "repositoryId": "consumer-04",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 72,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1883,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "9da3b2db5bae303bc9e4e7fbc38b6b512424aeec45448aa5673749dff6741e25",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:27.755Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-01-documentation",
        "repositoryId": "consumer-01",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 77,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2057,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "f0ffb82c7eb501bb1c68873631aaffb82098b27e5ac9466355df0a5366ae3c63",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:27.848Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-04-implementation",
        "repositoryId": "consumer-04",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 79,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2093,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "a7682bbba33e9f18e79ff836ce5be082681232e6cb4db53404fa686e1df691ee",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:27.928Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-04-implementation",
        "repositoryId": "consumer-04",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 67,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2084,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "84f178740d63f4ea1593cd30b24f46a3936467c2cec0141bcba888ea5ade0e36",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:28.004Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-06-architecture",
        "repositoryId": "consumer-06",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 63,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1884,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "d6b2e612c306dd7ecad8794deb14234716bbcde98997871ed9eecfe3d265979c",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:28.080Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-01-discovery",
        "repositoryId": "consumer-01",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 64,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1865,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "f20757677ff82678fbb681c7f5e9199399710f7f1cc2ce341aa634752730756e",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:28.158Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-05-architecture",
        "repositoryId": "consumer-05",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 65,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1893,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "2dbd7f6f921ff048a1bb17d00d56e8f03e7494387bde46f304f37041732c69dd",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:28.237Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-04-discovery",
        "repositoryId": "consumer-04",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 66,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1864,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "7a67b66eb4a2b08a509324664c9774ca95f1d69eb3c867b5eef9cc4f9594dd5b",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:28.343Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-06-discovery",
        "repositoryId": "consumer-06",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 93,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1855,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "9bf69375837e3ab9294e1379f4dd2958cfb5a1c36e1ed0b36617a7b48803bbbd",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:28.476Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-05-documentation",
        "repositoryId": "consumer-05",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 116,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2048,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "49d2715fe2213590cffd6a71d81fd09060d9ac6671a25ce8562275ed296c4a13",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:28.634Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-05-discovery",
        "repositoryId": "consumer-05",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 145,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1856,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "f9d473c2f7e196129366399add70756a01d1c3f30df7035da651818c251704e3",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:28.775Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-04-architecture",
        "repositoryId": "consumer-04",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 116,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1884,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "045bb0f52722a1f067b0c580111a483933c5a6ac608db7057ae57b5634ce895c",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:28.902Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-04-discovery",
        "repositoryId": "consumer-04",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 106,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1855,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "28d057e0f34e540bff268dbcf02560f971ebb9755bbee11ba571a43ce35ebbe8",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:29.018Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-04-discovery",
        "repositoryId": "consumer-04",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 94,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1856,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "d6f846ec28addde04a65405f8afd66a0d1b6e631465161197e49d66c8ad0c8be",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:29.125Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-03-documentation",
        "repositoryId": "consumer-03",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 92,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2057,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "3e3c6e13b1a7dfe4dba4dbd7addd66e75eb42914ea946b0fc217a62b6ee77086",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:29.249Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-03-architecture",
        "repositoryId": "consumer-03",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 105,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1884,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "c48e414c07fa3b516492518885bab003f2448cb948c5fa3efb943208f5e1db8d",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:29.356Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-03-documentation",
        "repositoryId": "consumer-03",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 92,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2056,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "3c60bff5a1fb257296ca382f8e9b42ca934facbfb6635ba5e1c0abbcab560012",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:29.446Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-03-architecture",
        "repositoryId": "consumer-03",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 75,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1892,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "d27dcacb69d8369e7121f0aebacc3323f3600854376c2e2cb934d1e45cfb4c27",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:29.542Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-02-implementation",
        "repositoryId": "consumer-02",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 83,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2093,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "91a7356f202041582c5f6910b0f87a97416c0216604ebd68acfc79b813141d67",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:29.630Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-01-implementation",
        "repositoryId": "consumer-01",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 75,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2093,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "78f92a2fb682087d70724cde2d991b4594d2d11568990dada3dc8979dd884f53",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:29.725Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-02-discovery",
        "repositoryId": "consumer-02",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 80,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1864,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "ee0c47c623519450ac96c66bfed59303690be28598d1b1db74c3f050befb88f8",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:29.802Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-02-documentation",
        "repositoryId": "consumer-02",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 64,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2047,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "f399fa51ba44bd2c1fe0fbf1bcd6487ed70eb9a527eee72b30484e89143979cb",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:29.876Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-05-implementation",
        "repositoryId": "consumer-05",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 61,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2084,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "cffa6101275d0ae98ad4582bd195dcf2062b16e08264461ee2ce87879e912e80",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:29.952Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-06-implementation",
        "repositoryId": "consumer-06",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 63,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2094,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "1477f4d0efb17f542b447207e10fb6a29bb63d25e6d72f76ae9b282d03009bd5",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:30.029Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-04-discovery",
        "repositoryId": "consumer-04",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 64,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1865,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "18914f630707cffa1a116153035f0b4d721bbd6412f426a3023ea4c32b7a49a5",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:30.104Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-02-documentation",
        "repositoryId": "consumer-02",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 62,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2056,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "1618515fc7386bb869f612f26b736c6fe8b4cf14eb3f730b8ef3c69903b3bf4c",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:30.182Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-04-documentation",
        "repositoryId": "consumer-04",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 65,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2047,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "3cb837e4b1f7f0ecf769ce97b31a088b28d03ce815646255e48ddb01c99b27d5",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:30.259Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-06-documentation",
        "repositoryId": "consumer-06",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 63,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2057,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "27195d7a99f55f774cc2233ac8ef41380846cdfbf6e272618cc3fa80fe26772b",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:30.336Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-03-discovery",
        "repositoryId": "consumer-03",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 64,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 1865,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "d39a1b43edde14e71e206c6bbadc8ab42b4f914dca617d31cfc9e8ad8e40b91c",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:30.413Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-05-documentation",
        "repositoryId": "consumer-05",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 63,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2047,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "3aeccb6d742cfef58b17297ce047e583457ecd0e465d4f66769e62921bd08dc3",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:30.505Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-02-implementation",
        "repositoryId": "consumer-02",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 79,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2094,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "39cc80c686fa9f7c5a9f6d595af032817e78d8d0c9235583f3fadca2c7dd6ed9",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:30.612Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-06-documentation",
        "repositoryId": "consumer-06",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 93,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2056,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "7e0a723b6b6e38d140abd8dc1bd221bf746c99fe65e8ba3040906b1625dbea67",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:30.719Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-02-implementation",
        "repositoryId": "consumer-02",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 90,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2085,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "c31c4ebe629dbc54f0ccd6e6ddfb190a635c38b520ded9896301f72c3ac1d363",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:24:30.841Z",
      "runId": "phase-8-ab-baseline-01",
      "planHash": "ab735034299c9c44f464d31f2e95e637dc436416d1eb156b5e75f743b4c4ecc5",
      "task": {
        "taskId": "consumer-03-documentation",
        "repositoryId": "consumer-03",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "failed",
        "exitCode": 2,
        "signal": null,
        "durationMs": 108,
        "responseBytes": 0,
        "stderrBytes": 33,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "a9ef8671cf61f4684e5ae981f7f52c6188dcfc7adcd140db8450d763ac22d8f7",
        "errorCode": "non-zero-exit"
      },
      "contextBytes": 2048,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "ec200d054644fe0834d0290c26ba7348c5abe4df2da2affba819534610177bc0",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:27:46.267Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-05-discovery",
        "repositoryId": "consumer-05",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 92080,
        "responseBytes": 663,
        "stderrBytes": 0,
        "stdoutHash": "282782e813e562310f484879ab2695c7a5ceb00074dfad4990f9b6bafb4584b8",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 501580,
        "outputTokens": 2657,
        "tokenMethod": "provider",
        "toolCalls": 12,
        "errorCode": "token-budget"
      },
      "contextBytes": 1865,
      "evidenceIds": [
        "entrypoint-evidence:package.json:14-37",
        "entrypoint-evidence:docs/for-agents/index.md:13-24",
        "entrypoint-evidence:docs/architecture.md:21-30",
        "entrypoint-evidence:scripts/build-discovery.mjs:5-20",
        "entrypoint-evidence:scripts/lib/deterministic-discovery.mjs:212-230",
        "discovery-check:ak-docs-discover-json"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "success",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 1,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 422656,
        "reasoningOutputTokens": 1057,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "a8fb14a9b4ee0fa0dc96d068d08bb01632bc707990f8d511f9af2183bd350990",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:28:00.529Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-03-implementation",
        "repositoryId": "consumer-03",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 14241,
        "responseBytes": 296,
        "stderrBytes": 0,
        "stdoutHash": "1657e6f97333f6d6d93d2b3d9df62e5703d2361faa4021c2f79a3828ef01e48e",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 21182,
        "outputTokens": 233,
        "tokenMethod": "provider",
        "toolCalls": 0
      },
      "contextBytes": 2084,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "incomplete",
      "evidenceQuality": "low",
      "safetyOutcome": "safe",
      "clarificationRequests": 1,
      "reworkCount": 0,
      "measurements": {
        "cachedInputTokens": 0,
        "reasoningOutputTokens": 181,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "8fad2ba334200ea0110dd495d2de7de69457a27667a827fdc5d664e179e4d910",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:29:02.130Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-06-implementation",
        "repositoryId": "consumer-06",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 61582,
        "responseBytes": 602,
        "stderrBytes": 0,
        "stdoutHash": "e58dbcd68f0dac6e42367726af8f430b14100879617222abcd2f9dd361dc15a4",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 586087,
        "outputTokens": 2407,
        "tokenMethod": "provider",
        "toolCalls": 11,
        "errorCode": "token-budget"
      },
      "contextBytes": 2085,
      "evidenceIds": [
        "docs/security/connections-byok-containment.md:3-8",
        "docs/testing/provider-live-certification.md:3-10",
        "packages/os-integrations/src/integration-product-metadata.ts:399-424",
        "verification-plan:ak-docs check --json"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "documentationFindingCount": 0,
        "cachedInputTokens": 526336,
        "reasoningOutputTokens": 1083,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "660c3e82c7613f58d077d4329f11e1948a663df48a40b63eebdef7d78bdc9d0e",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:30:04.737Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-01-discovery",
        "repositoryId": "consumer-01",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 62584,
        "responseBytes": 1084,
        "stderrBytes": 0,
        "stdoutHash": "a94e407c4f263fc4832b403fe1e167f33056bd6558b44dff85b47efffe260f76",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 238252,
        "outputTokens": 2575,
        "tokenMethod": "provider",
        "toolCalls": 5
      },
      "contextBytes": 1856,
      "evidenceIds": [
        "entrypoint-evidence:package.json:7-16 (ak-docs, ak-verify, package exports)",
        "entrypoint-evidence:bin/ak-docs.js:1-5 (CLI runtime entrypoint)",
        "entrypoint-evidence:src/index.ts:1-12 (library source entrypoint)",
        "entrypoint-evidence:doc-bridge.config.json:24-96 (ownership boundaries, checks, agent and human docs)",
        "entrypoint-evidence:docs/agent-corpus/OVERVIEW.md:1-9 (canonical agent corpus and maintainer owner)",
        "entrypoint-evidence:docs/guides/cli-map.md:6-10 (canonical CLI documentation)",
        "entrypoint-evidence:docs/for-agents.md:8-38 (canonical agent routing and machine documentation)",
        "discovery-check:node bin/ak-docs.js discover --json succeeded with structured discovery evidence; exact ak-docs command unavailable"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 188928,
        "reasoningOutputTokens": 1158,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "e257853996cf486812655c28727a1e023594d6086cde4022ea4f3564737ad494",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:30:55.888Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-01-implementation",
        "repositoryId": "consumer-01",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 51116,
        "responseBytes": 606,
        "stderrBytes": 0,
        "stdoutHash": "da429d85618fccb61de4ad8ef31d1ad1bd9b052a8334229849080993a8a506ed",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 288306,
        "outputTokens": 1735,
        "tokenMethod": "provider",
        "toolCalls": 7
      },
      "contextBytes": 2094,
      "evidenceIds": [
        "patch-evidence:docs/studies/v1-readiness-audit.md:29-31",
        "gap-evidence:docs/STABILITY.md:34-61",
        "proposal:documentation-only;no-source-changes",
        "verification-plan:ak-docs-check-json-after-approval",
        "blocked:ak-docs-command-unavailable"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 250880,
        "reasoningOutputTokens": 647,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "1177c28855a03831a49d6e530c1820439f5247d15ef4b5dae19757638d940acd",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:31:56.464Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-04-documentation",
        "repositoryId": "consumer-04",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 60555,
        "responseBytes": 1056,
        "stderrBytes": 0,
        "stdoutHash": "1d9db40a872af3b3194b6ce4a410fbab205521659c0405d2b70a094045444cc3",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 276455,
        "outputTokens": 2341,
        "tokenMethod": "provider",
        "toolCalls": 7
      },
      "contextBytes": 2048,
      "evidenceIds": [
        "documentation-evidence:content/docs/discovery.mdx:33 claims 13 executable gate scripts; packages/playbook/gate-manifest.json:1-13 and packages/playbook/gates count show 12",
        "documentation-evidence:content/docs/scripts/index.md:49 claims 12 reference implementations plus orchestrator while its table includes missing sanity.example.mjs",
        "generated-index-evidence:.doc-bridge/index.json:33 preserves the stale 13-script claim",
        "recommended-next-action:reconcile gate count and regenerate affected indexes",
        "review-limitation:semantic classification is based on static artifact comparison; mandated ak-docs audit could not run because ak-docs failed with EPERM"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "not-applicable",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "documentationFindingCount": 1,
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 228352,
        "reasoningOutputTokens": 861,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "67b21ae4b3db2de69b189dfbc76285bf3be09533733e9e8cc523b2eece5529e6",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:33:09.045Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-01-implementation",
        "repositoryId": "consumer-01",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 72555,
        "responseBytes": 583,
        "stderrBytes": 0,
        "stdoutHash": "e678bb3c1c9d65d03366040ac498904bb2bd7e45e9b01e874bd348cfefabdc70",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 503536,
        "outputTokens": 2632,
        "tokenMethod": "provider",
        "toolCalls": 11,
        "errorCode": "token-budget"
      },
      "contextBytes": 2085,
      "evidenceIds": [
        "README.md:109 references missing llms-install.md",
        "docs/guides/mcp-agents.md provides existing MCP setup guidance",
        "verification-plan:ak-docs check --json after approval",
        "acceptance-check-blocked:read-only filesystem EPERM"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 446464,
        "reasoningOutputTokens": 1161,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "6f0d1b913b0523aa65d5952ba57cf9416bda8883bdd909a35f44cf2ecf0d0c48",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:33:38.894Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-03-discovery",
        "repositoryId": "consumer-03",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 29829,
        "responseBytes": 648,
        "stderrBytes": 0,
        "stdoutHash": "f63196dc31fc9af31159078b2daf54a36601f5519f4570584b2fa6716904a125",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 72381,
        "outputTokens": 634,
        "tokenMethod": "provider",
        "toolCalls": 2
      },
      "contextBytes": 1864,
      "evidenceIds": [
        "entrypoint-evidence:package.json",
        "entrypoint-evidence:pnpm-workspace.yaml",
        "entrypoint-evidence:README.md",
        "entrypoint-evidence:AGENTS.md",
        "entrypoint-evidence:docs/architecture/overview.md",
        "entrypoint-evidence:docs/architecture/upstream-adoption.md"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 6,
        "cachedInputTokens": 42496,
        "reasoningOutputTokens": 220,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "17e13045dc3c96beaa812e75cbe10523f467ccacd8715c123d585dab67e35e98",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:34:06.899Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-06-discovery",
        "repositoryId": "consumer-06",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 27961,
        "responseBytes": 388,
        "stderrBytes": 0,
        "stdoutHash": "93674e9b1c96ca533447529463ba187ea984854a44775ee1b627c5c8903857a6",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 80832,
        "outputTokens": 524,
        "tokenMethod": "provider",
        "toolCalls": 2
      },
      "contextBytes": 1864,
      "evidenceIds": [
        "entrypoint-evidence"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 35328,
        "reasoningOutputTokens": 251,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "df68f2f1458a050272e43f74bc1d57391b35dc433ec3efb04e3fc35c7bc9edf2",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:35:44.408Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-05-architecture",
        "repositoryId": "consumer-05",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 97489,
        "responseBytes": 680,
        "stderrBytes": 0,
        "stdoutHash": "031cf4c3735fedc73ce287530518f7ca62c64d1cf29cabf353473cdcb4bc012c",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 583373,
        "outputTokens": 3657,
        "tokenMethod": "provider",
        "toolCalls": 14,
        "errorCode": "token-budget"
      },
      "contextBytes": 1884,
      "evidenceIds": [
        "architecture-evidence",
        "docs/architecture.md",
        "docs/for-agents/index.md",
        ".doc-bridge/workflow/artifacts/normalize-015404de24fb9d344a33e4eefdf6cb9e2ae1136213ccd427fbad8df1ecb8581a.json",
        "scripts/lib/eval-runner.mjs:129",
        "scripts/lib/eval-runner.mjs:189",
        "scripts/lib/eval-runner.mjs:198",
        "scripts/lib/eval-runner.mjs:219"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 503296,
        "reasoningOutputTokens": 1111,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "a94be9953b21a47f6b9f40c052d633589d8a37ce37de5a52dc574d9c9d7e997c",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:36:06.541Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-06-discovery",
        "repositoryId": "consumer-06",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 22096,
        "responseBytes": 503,
        "stderrBytes": 0,
        "stdoutHash": "1899e98c9b4f95d6947410b42723a1bcffb1c16a53258ca3af4577868a8c0aca",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 87220,
        "outputTokens": 760,
        "tokenMethod": "provider",
        "toolCalls": 2
      },
      "contextBytes": 1856,
      "evidenceIds": [
        "entrypoint-evidence",
        "AGENTS.md",
        "README.md",
        "package.json",
        "docs/for-agents/INDEX.md",
        "docs/internal/README.md",
        "ak-docs-discover:command-not-found"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 64768,
        "reasoningOutputTokens": 404,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "dc783d503e6072f9b2bf453fd7ab95722ca3ed857ad540c5c1c75a8f90af2ffe",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:36:57.615Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-06-implementation",
        "repositoryId": "consumer-06",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 51054,
        "responseBytes": 594,
        "stderrBytes": 0,
        "stdoutHash": "73fcef2179297a832055b1750c2f4715e41e7db1e37359f883e4111a5521a6a4",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 191826,
        "outputTokens": 1104,
        "tokenMethod": "provider",
        "toolCalls": 5
      },
      "contextBytes": 2084,
      "evidenceIds": [
        "patch-evidence:docs/testing/provider-live-certification.md",
        "patch-evidence:docs/security/connections-byok-containment.md",
        "verification-plan:ak-docs-check-json-blocked-by-read-only-filesystem"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 144128,
        "reasoningOutputTokens": 423,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "0f9fbb591fce470088cc068817360b7c78d3dbfa5fde4cca607f27054bb51624",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:38:08.934Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-05-documentation",
        "repositoryId": "consumer-05",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 71299,
        "responseBytes": 730,
        "stderrBytes": 0,
        "stdoutHash": "ca788515499fbd73961bbc68e49ba40c6efc062a6feb1b396e7d75af8abd8930",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 151420,
        "outputTokens": 1508,
        "tokenMethod": "provider",
        "toolCalls": 5
      },
      "contextBytes": 2056,
      "evidenceIds": [
        "documentation-evidence:contradictory:registry.schema.json:19|README.md:157-159",
        "review-limitation:production-ready-versus-beta-is-a-semantic-judgment-requiring-human-confirmation",
        "recommended-next-action:align-validated-status-description-with-beta-maturity-contract",
        "acceptance-check:blocked:EPERM-.doc-bridge/workflow/.lock"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 117504,
        "reasoningOutputTokens": 490,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "a460195c1ee92f2eb0d3e885db1c7b983dac9509ab2bd1409cb42abcb4d14796",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:38:49.182Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-02-architecture",
        "repositoryId": "consumer-02",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 40223,
        "responseBytes": 508,
        "stderrBytes": 0,
        "stdoutHash": "d7cb6578c0e62eab3c02ba641a58b16b3fa293d2e099cfd1d378ad1a685cb7aa",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 115294,
        "outputTokens": 634,
        "tokenMethod": "provider",
        "toolCalls": 4
      },
      "contextBytes": 1883,
      "evidenceIds": [
        "architecture-check:ak-docs-map-json:command-not-found",
        "architecture-check:node_modules-bin-ak-docs-map-json:blocked-by-read-only-filesystem"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "low",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 105216,
        "reasoningOutputTokens": 172,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "afe53fc1058dfb11f06e36b3cde45455cf34bdb5c1a82ee580be5bf7c7aa5e4f",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:40:20.209Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-05-documentation",
        "repositoryId": "consumer-05",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 90999,
        "responseBytes": 729,
        "stderrBytes": 0,
        "stdoutHash": "90b64471a73f1b5e090599d04bf861c19d1c69b0d943e484588c406c08442dbb",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 568493,
        "outputTokens": 3348,
        "tokenMethod": "provider",
        "toolCalls": 12,
        "errorCode": "token-budget"
      },
      "contextBytes": 2057,
      "evidenceIds": [
        "finding:catalog-manifest-stats-contradict-source",
        "catalog/manifest.json:5-7",
        "public/r/index.json:4",
        "source-count:registry-immediate-directories=346",
        "scripts/generate-catalog.mjs:stats-total-validated-draft",
        "limitation:semantic-audit-blocked-by-eperm-lock",
        "confidence:high",
        "next-action:regenerate-catalog-manifest-and-rerun-doc-audit"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "documentationFindingCount": 1,
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 494592,
        "reasoningOutputTokens": 1449,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "ebcc785cc321c09c0212297f4f1d85c2d64e8259f98be786324b1a35c2e8fb8e",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:40:59.327Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-06-architecture",
        "repositoryId": "consumer-06",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 39081,
        "responseBytes": 897,
        "stderrBytes": 0,
        "stdoutHash": "bead1b8a645dd18807388dccee78b69e8a0258e8ab49be975df596f8c095e951",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 184203,
        "outputTokens": 1473,
        "tokenMethod": "provider",
        "toolCalls": 5
      },
      "contextBytes": 1893,
      "evidenceIds": [
        "docs/architecture/layers.md:L0 os-core -> L1 domain -> L2 adapters -> L3 apps",
        "packages/os-headless/package.json:L78-L103 depends on os-core, os-flow, os-runtime, os-runtime-agentskit, os-rag-adapters, os-connectors",
        "apps/console/package.json:L17-L18 depends on desktop-shell and os-core",
        "packages/desktop-shell/package.json:L24 depends on os-desktop",
        "review-boundary:L1 domain packages importing L2 adapter packages would violate documented dependency direction",
        "ak-docs map --json:blocked by EPERM while opening repository temp file"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 152064,
        "reasoningOutputTokens": 626,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "ae12bd0f066b1ef91ed6ac67056fa889dce35d422ab9df56cece6fb7feac508a",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:42:02.651Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-01-implementation",
        "repositoryId": "consumer-01",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 63301,
        "responseBytes": 743,
        "stderrBytes": 0,
        "stdoutHash": "19db4e9f9303b2ab6eeddb6e580beabcf084fa45c1065329d1a8b619d37718b9",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 250610,
        "outputTokens": 1336,
        "tokenMethod": "provider",
        "toolCalls": 6
      },
      "contextBytes": 2084,
      "evidenceIds": [
        "patch-evidence-blocked:consumer-01 target documentation and preceding discovery/architecture/documentation artifacts were not available in the declared repository context, so no verified gap or safe patch location could be established",
        "verification-plan:after the target document and approved documentation-only patch are available, run ak-docs check --json and require a successful exit"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "low",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 200192,
        "reasoningOutputTokens": 423,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "3b6cd28f681a2f394f7929e15e5ab7dde6e2c27a95ff0b2c36a42042334767f4",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:43:16.955Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-01-architecture",
        "repositoryId": "consumer-01",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 74285,
        "responseBytes": 667,
        "stderrBytes": 0,
        "stdoutHash": "99e0b5b78acc7a598688f01a1aa94959c312529c5419ccbed24b04f4a4a51594",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 532793,
        "outputTokens": 2924,
        "tokenMethod": "provider",
        "toolCalls": 9,
        "errorCode": "token-budget"
      },
      "contextBytes": 1893,
      "evidenceIds": [
        "architecture-evidence",
        ".doc-bridge/workflow/artifacts/collect-926f9a23ad15b7a91c7394af75c8cae9e17522c3398ab35f1e9055f758e3c7cf.json",
        "doc-bridge.config.json",
        "src/cli/program.ts",
        "src/discovery/repository.ts",
        "src/index-builder/build-index.ts",
        "src/query/query.ts",
        "src/mcp/server.ts",
        "src/workflow/engine.ts"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 461056,
        "reasoningOutputTokens": 834,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "996c26afa5ed2899c01c40b9b24e99658a0ff8dfcfcba0c94dea7f09b0af47bc",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:44:06.352Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-06-discovery",
        "repositoryId": "consumer-06",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 49378,
        "responseBytes": 947,
        "stderrBytes": 0,
        "stdoutHash": "5ead6a3b663579be6dba03bb80ff63e6d49e5676baac264f7846f56544a8ca76",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 294777,
        "outputTokens": 1922,
        "tokenMethod": "provider",
        "toolCalls": 7
      },
      "contextBytes": 1865,
      "evidenceIds": [
        "entrypoint-evidence:README.md",
        "entrypoint-evidence:package.json",
        "entrypoint-evidence:apps/admin/app/layout.tsx",
        "entrypoint-evidence:apps/cloud/src/server.ts",
        "entrypoint-evidence:apps/console/src/main.tsx",
        "entrypoint-evidence:apps/desktop/src/main.tsx",
        "entrypoint-evidence:apps/license-service/src/index.ts",
        "entrypoint-evidence:apps/web/app/layout.tsx",
        "entrypoint-evidence:packages/os-cli/package.json",
        "canonical-docs:docs/for-agents/INDEX.md",
        "canonical-docs:docs/for-agents/packages/private-repository.md",
        "canonical-docs:docs/for-agents/apps/*.md",
        "discovery-check:ak-docs unavailable"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 259072,
        "reasoningOutputTokens": 780,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "7a272cff172de8506a810c6a70dd36d0ef14b449588862d2763de0a8a41d5c82",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:44:31.211Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-01-discovery",
        "repositoryId": "consumer-01",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 24825,
        "responseBytes": 364,
        "stderrBytes": 0,
        "stdoutHash": "6c35ee403d38a712048e6ca808990b03493d95b351bd2b1017ab680d1e6113e6",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 63225,
        "outputTokens": 453,
        "tokenMethod": "provider",
        "toolCalls": 2
      },
      "contextBytes": 1864,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "low",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 40448,
        "reasoningOutputTokens": 169,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "c4486c542576a0b7fe675b44a971b2a8d353353845a5ba0457da5f0af8186158",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:45:16.950Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-01-architecture",
        "repositoryId": "consumer-01",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 45710,
        "responseBytes": 531,
        "stderrBytes": 0,
        "stdoutHash": "82341bd077a21dd38e5e1f8ee1fcb0b373d777101c4cafb9b70a7ff9c4774ce1",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 294339,
        "outputTokens": 1615,
        "tokenMethod": "provider",
        "toolCalls": 7
      },
      "contextBytes": 1884,
      "evidenceIds": [
        "architecture-evidence",
        ".doc-bridge/index.json",
        "doc-bridge.config.json",
        "src/cli/program.ts",
        "src/discovery/repository.ts",
        "src/index-builder/build-index.ts",
        "docs/index.md"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 214528,
        "reasoningOutputTokens": 628,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "b6fb5a9190055fb7b62ad2e3c74e7512d9a5e1a86c2148ce399f421ab80bdf65",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:46:00.529Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-04-architecture",
        "repositoryId": "consumer-04",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 43538,
        "responseBytes": 929,
        "stderrBytes": 0,
        "stdoutHash": "2f8184ae0066a2b78eb0863d9d964383c369aaf0e4b89e9a514aa5542ebac803",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 203075,
        "outputTokens": 1524,
        "tokenMethod": "provider",
        "toolCalls": 5
      },
      "contextBytes": 1893,
      "evidenceIds": [
        "architecture-evidence:package.json:workspace-packages-and-dependencies",
        "architecture-evidence:components/ask-widget.tsx->@agentskit/chat,@agentskit/chat/react,@agentskit/core,@agentskit/react",
        "architecture-evidence:components/ask-widget.tsx->lib/discovery.ts",
        "architecture-evidence:app/docs/[[...slug]]/page.tsx->lib/source.ts->.source/Fumadocs",
        "architecture-evidence:packages/harness/src/adapters/doc-bridge.ts->.doc-bridge/index.json",
        "attention-boundary:components/ask-widget.tsx-host-shell-vs-AgentsKit-chat-owned-state",
        "acceptance:ak-docs-map:blocked-EPERM-lock"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 146176,
        "reasoningOutputTokens": 587,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "eecdf2ba6a744543de5fc120c77cd1b5eb981f788f91f35bb52133e63711355e",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:46:35.050Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-01-discovery",
        "repositoryId": "consumer-01",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 34495,
        "responseBytes": 511,
        "stderrBytes": 0,
        "stdoutHash": "178dca9bf01057261346460aaa563cb5df6dbe89ca61b7358ead92b6c6a1a63b",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 89937,
        "outputTokens": 707,
        "tokenMethod": "provider",
        "toolCalls": 3
      },
      "contextBytes": 1855,
      "evidenceIds": [
        "entrypoint-evidence:package.json",
        "entrypoint-evidence:README.md",
        "entrypoint-evidence:docs/index.md",
        "entrypoint-evidence:apps/docs/AGENTS.md"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 61696,
        "reasoningOutputTokens": 256,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "a0512dc85797ba39e73c45facbb6a986a74b48e24caa51d91b635a3bcb2dfdb2",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:47:11.321Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-01-architecture",
        "repositoryId": "consumer-01",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 36250,
        "responseBytes": 364,
        "stderrBytes": 0,
        "stdoutHash": "907a1e374f4a4e009093b8112c0e65979cd68e45b13b64e2c5222f71b4de306e",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 87056,
        "outputTokens": 588,
        "tokenMethod": "provider",
        "toolCalls": 3
      },
      "contextBytes": 1892,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "low",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 1,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 78848,
        "reasoningOutputTokens": 280,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "fdc5fe11539b4a06d406cde96e49d298b1635cc193af9f88c039ed3e1e979509",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:48:08.810Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-02-architecture",
        "repositoryId": "consumer-02",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 57465,
        "responseBytes": 841,
        "stderrBytes": 0,
        "stdoutHash": "c322abb67a68cf7249e1ff6a89440ac70d92b4dfcc75ab76e13dc53f1d2c2faf",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 287829,
        "outputTokens": 2266,
        "tokenMethod": "provider",
        "toolCalls": 7
      },
      "contextBytes": 1893,
      "evidenceIds": [
        "architecture-evidence:pnpm-workspace.yaml:packages-and-apps",
        "architecture-evidence:packages/core->runtime-adapters-tools-memory-rag-react",
        "architecture-evidence:README.md:208-235:package-responsibilities",
        "architecture-evidence:docs/architecture/adrs/0009-composition-rules.md:19-31:dependency-direction",
        "architecture-evidence:apps/ask-backend/src/server.ts:22-35:cross-app-docs-next-imports",
        "acceptance-check:architecture-check:blocked-by-read-only-temp-file-EPERM"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 234496,
        "reasoningOutputTokens": 683,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "6d822ee64ef642a9e51d07707270dfaf64d796d74156209f13b35623ceed0fa3",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:49:37.609Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-06-implementation",
        "repositoryId": "consumer-06",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 88743,
        "responseBytes": 356,
        "stderrBytes": 0,
        "stdoutHash": "1a1237c483994b0abbf100f5fcb6cfbd84658120b33ce981382179efce0a9413",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 260303,
        "outputTokens": 1501,
        "tokenMethod": "provider",
        "toolCalls": 6
      },
      "contextBytes": 2093,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "incomplete",
      "evidenceQuality": "low",
      "safetyOutcome": "safe",
      "clarificationRequests": 1,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 217344,
        "reasoningOutputTokens": 661,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "a9bc914c58fe2cc0fc0b2ae24cc37670bb6777485b5e06440c55c426b1dd6144",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:50:59.073Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-05-implementation",
        "repositoryId": "consumer-05",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 81093,
        "responseBytes": 626,
        "stderrBytes": 0,
        "stdoutHash": "7f355774183460c5e2bf2541155f11bd14c953f187b2ca554bf8d909aee846c4",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 202337,
        "outputTokens": 1588,
        "tokenMethod": "provider",
        "toolCalls": 5
      },
      "contextBytes": 2093,
      "evidenceIds": [
        "docs/for-agents/registry-discovery.md",
        "docs/for-agents/index.md#change-routes",
        "doc-bridge:query:change-discovery",
        "doc-bridge:query:understand-discovery",
        "verification-plan:ak-docs-check-json",
        "verification-blocker:read-only-workflow-lock"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 167424,
        "reasoningOutputTokens": 703,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "5a65f54c4a7ec8d7a6073fa756d234a74e019ec901ce5edb5bb690a3f3034f83",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:51:42.304Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-05-architecture",
        "repositoryId": "consumer-05",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 43209,
        "responseBytes": 388,
        "stderrBytes": 0,
        "stdoutHash": "48e9f69e628aa752062e354b706b4732ae34a2218126a6e30ded4ccc5ed4b308",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 138689,
        "outputTokens": 852,
        "tokenMethod": "provider",
        "toolCalls": 5
      },
      "contextBytes": 1892,
      "evidenceIds": [
        "architecture-evidence"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "low",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 86016,
        "reasoningOutputTokens": 345,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "981a0d9e62ad6aa72192ed307f9e03fa1f9abe3668ff0a5ae61e15e7387b5c01",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:52:45.491Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-05-architecture",
        "repositoryId": "consumer-05",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 63168,
        "responseBytes": 605,
        "stderrBytes": 0,
        "stdoutHash": "e999fe7f0c5e94f4bb3a87a8a414373902a888406e288af44fb880e958eb2718",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 208558,
        "outputTokens": 1291,
        "tokenMethod": "provider",
        "toolCalls": 6
      },
      "contextBytes": 1883,
      "evidenceIds": [
        "docs/architecture.md",
        "docs/for-agents/registry-architecture.md",
        "package.json",
        "scripts/lib/deterministic-discovery.mjs",
        "registry/research/agent.ts",
        "architecture-check:EPERM:.doc-bridge/workflow/.lock"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 1,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 180480,
        "reasoningOutputTokens": 375,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "e66b0d43939e99d63945aea16785e3660d2d5e3c273c8125d70a8513762cf7e0",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:53:18.290Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-02-discovery",
        "repositoryId": "consumer-02",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 32780,
        "responseBytes": 599,
        "stderrBytes": 0,
        "stdoutHash": "1240ae4e834316a7e948b6aa80b91df26738e4b377fd2c6cc3a7f26f4e6381e4",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 100576,
        "outputTokens": 679,
        "tokenMethod": "provider",
        "toolCalls": 3
      },
      "contextBytes": 1855,
      "evidenceIds": [
        "entrypoint-evidence:package.json",
        "entrypoint-evidence:AGENTS.md",
        "entrypoint-evidence:.github/CODEOWNERS",
        "entrypoint-evidence:apps/docs-next/content/docs/for-agents/doc-bridge.mdx"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 1,
      "measurements": {
        "searchHitRate": 1,
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 4,
        "cachedInputTokens": 64768,
        "reasoningOutputTokens": 287,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "174764a6ecb10375b7f3bb9f5c0df81a5b97fc52b93ba14255d600e43a4de6ad",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:54:14.113Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-02-architecture",
        "repositoryId": "consumer-02",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 55805,
        "responseBytes": 555,
        "stderrBytes": 0,
        "stdoutHash": "e26ed03317fb2a3e6293252a83054a303ce8bff4455ff908ede4171db91f99c6",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 170231,
        "outputTokens": 937,
        "tokenMethod": "provider",
        "toolCalls": 6
      },
      "contextBytes": 1892,
      "evidenceIds": [
        "architecture-evidence:doc-bridge.config.json",
        "architecture-evidence:package.json",
        "architecture-evidence:pnpm-workspace.yaml",
        "architecture-check:blocked:EPERM:.doc-bridge/workflow/.lock"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "low",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 133632,
        "reasoningOutputTokens": 237,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "eafece4a1c4d577f5b80e64d125845858d328de08a4da754dec85378498eb0a6",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:55:13.084Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-03-implementation",
        "repositoryId": "consumer-03",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 58948,
        "responseBytes": 629,
        "stderrBytes": 0,
        "stdoutHash": "5dd6c9e41add943311183b91cd9bb2d4cbc5022a66858fcd3d481750c4893441",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 184460,
        "outputTokens": 1228,
        "tokenMethod": "provider",
        "toolCalls": 6
      },
      "contextBytes": 2093,
      "evidenceIds": [
        "evidence-76b58a72c26234a6b54edde85fc36a66",
        "verification-plan:after approval run ak-docs check --json and review the corrected example against DeterministicAnswerAdapterOptions; current attempt exited 127 because ak-docs is unavailable on PATH"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 148992,
        "reasoningOutputTokens": 395,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "e810d21c6e27d1a828e1aee7c127f914a2a9d5345dd615cebf0fbc7ffdf19aeb",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:56:24.038Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-04-implementation",
        "repositoryId": "consumer-04",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 70920,
        "responseBytes": 606,
        "stderrBytes": 0,
        "stdoutHash": "7856ac14ca37534e656c2102c3c296093cde8b276eb96441f4888bb525618e2e",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 618993,
        "outputTokens": 2673,
        "tokenMethod": "provider",
        "toolCalls": 11,
        "errorCode": "token-budget"
      },
      "contextBytes": 2085,
      "evidenceIds": [
        ".doc-bridge/workflow/artifacts/report-5ecbc246bfdbfcef0cf05dde5d4bc742617dd47c05d981b8c99bbc582bdbe7fa.json",
        ".doc-bridge/workflow/artifacts/reconcile-9777cd3e46fbfecd325ed4caccf5776be28dfcf6d5af61f77378782180f6afeb.json"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "low",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "documentationFindingCount": 0,
        "cachedInputTokens": 555776,
        "reasoningOutputTokens": 570,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "020b521eadf1c202684b4c8a658eba4dcbf6b3d943decf9159671e48e61010c9",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:57:18.228Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-02-architecture",
        "repositoryId": "consumer-02",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 54156,
        "responseBytes": 379,
        "stderrBytes": 0,
        "stdoutHash": "dc7617e2367c743091d2a231c92dd88c5cf85567fc4d8960a352eb78e4473ac3",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 320227,
        "outputTokens": 2190,
        "tokenMethod": "provider",
        "toolCalls": 6
      },
      "contextBytes": 1884,
      "evidenceIds": [
        "architecture-evidence"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 262400,
        "reasoningOutputTokens": 777,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "584d88d88ce4e31b626ae227ac41da4b4e87f5001d33dc37e8e60436f3c35a4e",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:57:50.256Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-03-discovery",
        "repositoryId": "consumer-03",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 32007,
        "responseBytes": 627,
        "stderrBytes": 0,
        "stdoutHash": "353e50eed9f65bfb1711947106e9b6fffc3e8501e7bb1df2eabb6169cbe31d81",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 68916,
        "outputTokens": 519,
        "tokenMethod": "provider",
        "toolCalls": 2
      },
      "contextBytes": 1855,
      "evidenceIds": [
        "entrypoint-evidence:README.md",
        "entrypoint-evidence:package.json",
        "entrypoint-evidence:pnpm-workspace.yaml",
        "entrypoint-evidence:docs/for-agents/index.md",
        "entrypoint-evidence:docs/for-agents/architecture.md",
        "discovery-check:ak-docs-command-not-found-exit-127"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 47360,
        "reasoningOutputTokens": 190,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "2a264eed2a7ed714b637ab71f4fa0d754ff6a4f41174408f7ad65f365751bb41",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T17:58:41.221Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-02-documentation",
        "repositoryId": "consumer-02",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 50944,
        "responseBytes": 659,
        "stderrBytes": 0,
        "stdoutHash": "ddbc217800e85e70258cfceb30c9c6adefc6cc275a5db88de9faa40f96884401",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 225313,
        "outputTokens": 1778,
        "tokenMethod": "provider",
        "toolCalls": 6
      },
      "contextBytes": 2057,
      "evidenceIds": [
        "documentation-evidence:AGENTS.md:11-v1.0.0-conflicts-with-packages/core/package.json:3-v1.12.8; classify-stale; recommend-update-or-clarify-versioning",
        "review-limitation:high-confidence-file-conflict; audit-blocked-by-read-only-filesystem-and-ak-docs-command-unavailable"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "documentationFindingCount": 1,
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 178432,
        "reasoningOutputTokens": 909,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "4a60f1bbd7ce0838a224fd41e73c1711f7e2bc0ea8194a1be0ab557117414bbd",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:00:04.251Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-06-architecture",
        "repositoryId": "consumer-06",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 83005,
        "responseBytes": 604,
        "stderrBytes": 0,
        "stdoutHash": "e79ffea115253b540d3a440cd872b52a11f352d9f4526d78013dac6319c13d0e",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 209744,
        "outputTokens": 890,
        "tokenMethod": "provider",
        "toolCalls": 5
      },
      "contextBytes": 1892,
      "evidenceIds": [
        "architecture-evidence:package.json",
        "architecture-evidence:pnpm-workspace.yaml",
        "acceptance-check:ak-docs-map-command-not-found",
        "acceptance-check:pnpm-exec-read-only-filesystem-EPERM",
        "acceptance-check:direct-cli-workflow-lock-EPERM"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 2,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 147200,
        "reasoningOutputTokens": 280,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "cd0162c3a6ae2ca5aabde1de088dafff41918047ee60e5df05a8af0dd8126652",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:01:48.713Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-02-implementation",
        "repositoryId": "consumer-02",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 104438,
        "responseBytes": 485,
        "stderrBytes": 0,
        "stdoutHash": "a4148dd23fe7eb15d6fb73bde353cbe8a11c5d0dcba6d9d526d6b1937fb15661",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 280302,
        "outputTokens": 1812,
        "tokenMethod": "provider",
        "toolCalls": 9
      },
      "contextBytes": 2084,
      "evidenceIds": [
        "evidence-e8cf4bb94a30203b3013953fe27770c2",
        "evidence-8513981f86a46af4c9d459bfa5cdb94a"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 238336,
        "reasoningOutputTokens": 593,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "d084028d9d4c47f764248af889ba0a3dde83f42cf313f4bd48a13bfe66f3c1ac",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:02:28.719Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-06-documentation",
        "repositoryId": "consumer-06",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 39972,
        "responseBytes": 1136,
        "stderrBytes": 0,
        "stdoutHash": "b74516d2b9a53c2b751b5473d69a7fec303aaa5a788ceb9e9711e1bb49a027c4",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 111303,
        "outputTokens": 813,
        "tokenMethod": "provider",
        "toolCalls": 3
      },
      "contextBytes": 2047,
      "evidenceIds": [
        "documentation-evidence:AGENTS.md:3 claims 76 package dirs, but repository inspection finds 77 directories under packages; packages/pack-bundle is the extra artifact-only directory without package.json.",
        "documentation-evidence:docs/internal/assistant-confidence-demo-2026-07-09.md:28 independently classifies the same wording as P2 and explains that the accurate claim is 76 package workspaces, not dirs.",
        "review-limitation:ak-docs audit documentation --json could not run because ak-docs is unavailable, so the repository-wide audit and its measurable semantic limitations were not obtained.",
        "recommended-next-action:Change AGENTS.md:3 from 76 package dirs to 76 package workspaces, preserving the existing workspace composition claim."
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 78080,
        "reasoningOutputTokens": 283,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "90c686db30dbe6b21d21e8b390ec6c70d4f244f0b7f9586ff07c1bf612263ce6",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:03:24.733Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-01-documentation",
        "repositoryId": "consumer-01",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 55973,
        "responseBytes": 1023,
        "stderrBytes": 0,
        "stdoutHash": "d1cb570c5287dd0102afed0036cb85ffd4197aec3a94cd164af375d57c33728a",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 212789,
        "outputTokens": 1074,
        "tokenMethod": "provider",
        "toolCalls": 6
      },
      "contextBytes": 2056,
      "evidenceIds": [
        "documentation-evidence:stale-claim:docs/spec/config-v1.md:836 labels ak-docs chat as planned, but src/cli/program.ts:1600-1612 implements the chat command and src/intelligence/chat.ts:75-88 implements runChatOnce",
        "review-limitation:high-confidence direct source comparison; repository-wide audit was not completed because bare ak-docs was unavailable and node bin/ak-docs.js audit documentation --json was blocked by EPERM creating .doc-bridge/workflow/.lock",
        "recommended-next-action:remove the stale planned qualifier from docs/spec/config-v1.md:836, then rerun ak-docs audit documentation --json in a writable workspace"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 168448,
        "reasoningOutputTokens": 269,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "d84e0ba889f55afc082e443545b4f816b4c4bdbf20b5790412ab8071d83ad364",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:03:39.313Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-06-documentation",
        "repositoryId": "consumer-06",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 14559,
        "responseBytes": 297,
        "stderrBytes": 0,
        "stdoutHash": "e8aee5c4150b04efde7d1ec0287494fe55b3d337b242672281a47c4aed3e93e4",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 51221,
        "outputTokens": 483,
        "tokenMethod": "provider",
        "toolCalls": 1
      },
      "contextBytes": 2048,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "low",
      "safetyOutcome": "safe",
      "clarificationRequests": 1,
      "reworkCount": 0,
      "measurements": {
        "cachedInputTokens": 30208,
        "reasoningOutputTokens": 274,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "645dcc53642672b794370c5aa3c1a581024eaf15809b7fecb382d58efa402383",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:04:45.144Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-03-discovery",
        "repositoryId": "consumer-03",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 65811,
        "responseBytes": 573,
        "stderrBytes": 0,
        "stdoutHash": "8f9b94aa3f87605df4144a271af46f0e90e020871b1a966106d74141253dbec3",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 364718,
        "outputTokens": 2403,
        "tokenMethod": "provider",
        "toolCalls": 9
      },
      "contextBytes": 1856,
      "evidenceIds": [
        "discovery-check",
        "entrypoint-evidence",
        "package.json",
        "pnpm-workspace.yaml",
        "docs/for-agents/index.md",
        "docs/architecture/overview.md",
        "docs/for-agents/packages/chat.md",
        "packages/*/package.json",
        "apps/*/package.json"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "success",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 1,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 315904,
        "reasoningOutputTokens": 1246,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "5dfe9355ef1c4c143579346ab05b7fb5b326ef69c9e4aa1a53f74e3d948f72c4",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:05:33.217Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-02-discovery",
        "repositoryId": "consumer-02",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 48048,
        "responseBytes": 982,
        "stderrBytes": 0,
        "stdoutHash": "593bbed8e9641984a3d1ff3c36fec6c6d4e912032bbf1ae521a67c063fa68b90",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 239951,
        "outputTokens": 1921,
        "tokenMethod": "provider",
        "toolCalls": 6
      },
      "contextBytes": 1856,
      "evidenceIds": [
        "entrypoint-evidence:packages/core/package.json",
        "entrypoint-evidence:packages/adapters/package.json",
        "entrypoint-evidence:packages/runtime/package.json",
        "entrypoint-evidence:packages/tools/package.json",
        "entrypoint-evidence:packages/cli/package.json",
        "entrypoint-evidence:apps/docs-next/package.json",
        "ownership-evidence:AGENTS.md",
        "canonical-docs:doc-bridge.config.json",
        "canonical-docs:apps/docs-next/content/docs/for-agents/index.mdx",
        "canonical-docs:docs/product/surfaces.md",
        "canonical-docs:docs/architecture/adrs/0007-docs-platform-fumadocs.md",
        "discovery-check:ak-docs-discover-command-unavailable-or-blocked-by-EPERM"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 193792,
        "reasoningOutputTokens": 768,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "a31ade9a43b50c75326a1daa09b1aa7bf837028acd654f43e905e5bef5cb24e0",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:06:03.521Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-05-discovery",
        "repositoryId": "consumer-05",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 30228,
        "responseBytes": 543,
        "stderrBytes": 0,
        "stdoutHash": "fa7b95180e292e136ca08de5d305733d422bebbb04af61f5b47a7cfc21afb158",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 66810,
        "outputTokens": 554,
        "tokenMethod": "provider",
        "toolCalls": 2
      },
      "contextBytes": 1855,
      "evidenceIds": [
        "docs/for-agents/index.md",
        "docs/architecture.md",
        "docs/for-agents/registry-discovery.md",
        "package.json",
        "scripts/build-discovery.mjs",
        "scripts/lib/deterministic-discovery.mjs"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 41472,
        "reasoningOutputTokens": 213,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "a28f903c10da1e7d4afd610bf7bc2ef9751e10043d8a90b49c45d28d60bdc957",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:06:34.985Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-03-architecture",
        "repositoryId": "consumer-03",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 31435,
        "responseBytes": 911,
        "stderrBytes": 0,
        "stdoutHash": "a8d6b6fa6c1dd847fea2cb17da015f9d4107d5f41a09a38ab30b5fe15d1e3547",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 110998,
        "outputTokens": 1258,
        "tokenMethod": "provider",
        "toolCalls": 3
      },
      "contextBytes": 1893,
      "evidenceIds": [
        "architecture-evidence",
        "docs/architecture/overview.md",
        "docs/for-agents/architecture.md",
        "docs/architecture/upstream-adoption.md",
        "observed-components:Core,Protocol,Server,NativeRenderers,CLI,DevtoolsEval,AgentsKit-upstream,ExampleApps,DocsHost",
        "observed-relations:NativeRenderers->Core;Server->Core+Protocol;Core->AgentsKit;ExampleApps->Core;DocsHost->Core+Protocol",
        "attention-point:upstream-adoption-matrix-is-explicitly-proposed-for-human-acceptance-and-revalidation",
        "acceptance-check:ak-docs-map-blocked-command-unavailable-and-lock-creation-EPERM"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 70912,
        "reasoningOutputTokens": 564,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "03cb09b61358f1e01641b712db3b1b1dc48ff16d823933bd061ef549fd492f21",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:07:13.294Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-01-architecture",
        "repositoryId": "consumer-01",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 38291,
        "responseBytes": 668,
        "stderrBytes": 0,
        "stdoutHash": "f23754f85e47f3949ce741014f51b02108f28f99d896bfecd8b0744d6fd0895b",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 97113,
        "outputTokens": 771,
        "tokenMethod": "provider",
        "toolCalls": 3
      },
      "contextBytes": 1883,
      "evidenceIds": [
        "architecture-evidence:package.json",
        "architecture-evidence:pnpm-workspace.yaml",
        "architecture-evidence:doc-bridge.config.json",
        "architecture-evidence:src/index.ts",
        "architecture-evidence:src/cli/program.ts",
        "architecture-evidence:README.md",
        "architecture-check:node-bin-ak-docs-map-json-failed-EPERM-workflow-lock"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 62720,
        "reasoningOutputTokens": 312,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "addd53a5c752d5304e19a5441e97378fd2f1092a608003b4abd3674daba90ed3",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:07:58.382Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-01-documentation",
        "repositoryId": "consumer-01",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 45058,
        "responseBytes": 606,
        "stderrBytes": 0,
        "stdoutHash": "3f0d7c7b9cc34adcef708538bc54d0bb2279e920d3699a764f145e3f5af6b61c",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 132478,
        "outputTokens": 893,
        "tokenMethod": "provider",
        "toolCalls": 4
      },
      "contextBytes": 2047,
      "evidenceIds": [
        "evidence-6f610915ed4f9f2bda5c431ea99acfd0",
        "review-limitation:high-confidence repository comparison, but release/tag availability was not verified and ak-docs audit was blocked by read-only filesystem EPERM"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 97280,
        "reasoningOutputTokens": 284,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "c191f5e6dd138e0a15d159a2f6ea5372ee2d97e9fe853159e91fb63c3e8fc44a",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:09:05.369Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-05-implementation",
        "repositoryId": "consumer-05",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 66968,
        "responseBytes": 595,
        "stderrBytes": 0,
        "stdoutHash": "612224c9e50c14df2071e7525235c660adb4a7295a2eeff4e166236899216fd0",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 379971,
        "outputTokens": 2180,
        "tokenMethod": "provider",
        "toolCalls": 9
      },
      "contextBytes": 2094,
      "evidenceIds": [
        ".codex/verification/latest.json:documentation-audit",
        "docs/for-agents/index.md:required-checks",
        "verification-command:pnpm-exec-ak-docs-check-json-blocked-by-EPERM"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "low",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "documentationFindingCount": 1,
        "documentationExampleRate": 0.23389021479713604,
        "cachedInputTokens": 307968,
        "reasoningOutputTokens": 920,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "d9f6480e543925c0a249fb1e8b0bce950280407d172990c7074e4f8b8433f4c3",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:09:57.137Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-05-implementation",
        "repositoryId": "consumer-05",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 51745,
        "responseBytes": 577,
        "stderrBytes": 0,
        "stdoutHash": "5f17e7697a6b5774b9919d9340706d46593692cce79726ef134f7459a156fd57",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 206734,
        "outputTokens": 1938,
        "tokenMethod": "provider",
        "toolCalls": 6
      },
      "contextBytes": 2085,
      "evidenceIds": [
        "patch-evidence:docs/for-agents/index.md:required-checks guidance omits ak-docs check --json",
        "verification-plan:run ak-docs check --json after approval; current run blocked by .doc-bridge/workflow/.lock permission error"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 157184,
        "reasoningOutputTokens": 956,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "2ecb49c9325833e6f3981b7d7daa0f5d3a3d1690fd5442402ec4c4bda6064d0d",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:10:46.606Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-04-documentation",
        "repositoryId": "consumer-04",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 49448,
        "responseBytes": 742,
        "stderrBytes": 0,
        "stdoutHash": "19f40ecad4fecaa9812019ef5ef629a724d1049a9e4ab9a899737be3a5ebd5ea",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 275780,
        "outputTokens": 1955,
        "tokenMethod": "provider",
        "toolCalls": 8
      },
      "contextBytes": 2057,
      "evidenceIds": [
        "documentation-evidence:.codex/verification.json:documentation-audit configures unsupported command `ak-docs audit documentation --json`; `node node_modules/@agentskit/doc-bridge/bin/ak-docs.js ...` help exposes no `audit` command",
        "review-limitation:semantic documentation findings could not be measured because the required audit command is unavailable in installed Doc Bridge 1.7.45"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 221184,
        "reasoningOutputTokens": 887,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "283d29b890e78e582661d9e4d849ddba002f411b0a66120c5a3bf3ffa7e27517",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:11:27.862Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-06-architecture",
        "repositoryId": "consumer-06",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 41235,
        "responseBytes": 656,
        "stderrBytes": 0,
        "stdoutHash": "a0b069abd04fc330bdbd0552c746bbd339f0c683a7622ed5bd0208a4696dfcad",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 119980,
        "outputTokens": 843,
        "tokenMethod": "provider",
        "toolCalls": 3
      },
      "contextBytes": 1883,
      "evidenceIds": [
        "pnpm-workspace.yaml",
        "package.json",
        "AGENTS.md",
        "docs/for-agents/INDEX.md",
        "packages/desktop-shell/src/index.ts",
        "apps/desktop/src/main.tsx",
        "apps/console/src/main.tsx",
        "packages/os-runtime/src/registry.ts",
        "packages/os-flow/src/runner.ts",
        "acceptance:ak-docs-map-json:command-not-found"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 79104,
        "reasoningOutputTokens": 332,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "9b687c54de3f99e0ad819fdf6bb19f66d5fcbc855d6746b8ecf5c230e5645854",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:11:53.125Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-05-discovery",
        "repositoryId": "consumer-05",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 25244,
        "responseBytes": 593,
        "stderrBytes": 143,
        "stdoutHash": "1a59bbe92d337debd27c4be6cdf8bc1be901d20633a91cfef227d63a53d17313",
        "stderrHash": "ad3b3119fb5619601a4d87eaca4d6bb72f0c12b1138d333f816a1e310ac3de5f",
        "inputTokens": 46131,
        "outputTokens": 524,
        "tokenMethod": "provider",
        "toolCalls": 1
      },
      "contextBytes": 1864,
      "evidenceIds": [
        "entrypoint-evidence:AGENTS.md",
        "entrypoint-evidence:docs/for-agents/index.md",
        "entrypoint-evidence:package.json",
        "entrypoint-evidence:.codex/verification.json",
        "discovery-check:command-not-found"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 4,
        "cachedInputTokens": 27136,
        "reasoningOutputTokens": 177,
        "stderrBytes": 143
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "653425b239e9192d8d45ed77a91ae59c9f406d0fdeaf18e6427f9bbefc8b1fd1",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:13:19.304Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-04-documentation",
        "repositoryId": "consumer-04",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 86159,
        "responseBytes": 794,
        "stderrBytes": 0,
        "stdoutHash": "c01c8fb93d8903773a5dee2af739e25f25cd7765bfb3245174d1ab0a6644a5f9",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 275483,
        "outputTokens": 1739,
        "tokenMethod": "provider",
        "toolCalls": 8
      },
      "contextBytes": 2056,
      "evidenceIds": [
        "documentation-evidence:missing-command:the-required `ak-docs audit documentation --json` command is absent from @agentskit/doc-bridge 1.7.45; the installed CLI help lists no `audit` command and executing it prints help instead of audit JSON",
        "review-limitation:confidence-high-for-command-absence-but-semantic-documentation-audit-was-blocked-by-the-missing-command-and-read-only-workflow-lock"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 241408,
        "reasoningOutputTokens": 563,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "049039b9ca314e7cc68289aa682215c48e20f31c40b649cf1c2c98c6e87068a9",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:14:12.392Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-03-documentation",
        "repositoryId": "consumer-03",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 53064,
        "responseBytes": 699,
        "stderrBytes": 0,
        "stdoutHash": "d22cee49e00882e2962d90e756b1993c919dd5b680fef9257e185144401ff84c",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 153660,
        "outputTokens": 947,
        "tokenMethod": "provider",
        "toolCalls": 5
      },
      "contextBytes": 2047,
      "evidenceIds": [
        "evidence-ef4e460be8409cb30a2fbb4e1ca05ec4",
        "review-limitation:high-confidence direct version contradiction, but runtime compatibility and generated-index freshness were not independently verified because ak-docs was unavailable",
        "acceptance-check:documentation-check:blocked:ak-docs command not found"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 119552,
        "reasoningOutputTokens": 247,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "ef4c9b12ac21ac3aa3c00af21f6cb09812f7de624f1b4cfd35adaa1b2982ee2f",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:15:24.935Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-02-documentation",
        "repositoryId": "consumer-02",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 72521,
        "responseBytes": 1076,
        "stderrBytes": 0,
        "stdoutHash": "683086daf51348f38119b8d5391b6d38bd0629443ae11eec8c4a31017940adf4",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 408776,
        "outputTokens": 2711,
        "tokenMethod": "provider",
        "toolCalls": 8,
        "errorCode": "token-budget"
      },
      "contextBytes": 2048,
      "evidenceIds": [
        "evidence-8ae46e84cff0d0a9883293d2c583426e",
        "classification: missing and contradictory documentation claim; recommended next action is update the reference landing page package-link list and add a check that its links cover all published non-private packages.",
        "confidence: high for the line-level omission based on direct comparison of current package metadata and documentation; lower for broader documentation freshness because the full audit could not execute.",
        "review-limitation: ak-docs audit documentation --json failed before producing findings because the read-only environment denied creation of .doc-bridge/workflow/.lock (EPERM); no semantic audit report or generated-index freshness result was available."
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "success",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 331776,
        "reasoningOutputTokens": 1159,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "ba5675fe6902172a5d10e6751b0f2fcf56eb71ca6245cd0cdd8a922c35ce472c",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:16:36.858Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-02-discovery",
        "repositoryId": "consumer-02",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 71899,
        "responseBytes": 1212,
        "stderrBytes": 0,
        "stdoutHash": "d3b2c279657b04232967be8184915b3d97d2728b6032e7258badecd503a9351d",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 427914,
        "outputTokens": 2421,
        "tokenMethod": "provider",
        "toolCalls": 10,
        "errorCode": "token-budget"
      },
      "contextBytes": 1865,
      "evidenceIds": [
        "artifact:./node_modules/.bin/ak-docs discover --json succeeded; snapshot sourceRevision=0551139b338745da778fffaad15df2a521eca561; contentHash=3b5a3981abca11e57dc9790d0d1f949f5647d59bcf323781079c24d274709dce",
        "artifact:package.json owns monorepo scripts and Doc Bridge integration",
        "artifact:packages/*/package.json define package entrypoints; packages/cli/package.json defines agentskit CLI; packages/mcp/package.json defines agentskit-mcp CLI",
        "artifact:packages/*/src are package implementation boundaries",
        "artifact:apps/docs-next/content/docs/for-agents/index.mdx is canonical agent documentation index and package handoff map",
        "artifact:apps/docs-next/content/docs/for-agents/*.mdx are canonical per-package agent documentation",
        "artifact:doc-bridge.config.json declares agent corpus, human docs owner apps/docs-next, package routing, and ak-docs CLI"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 384256,
        "reasoningOutputTokens": 965,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "d1e6707865f2c854c735b98640a3826cf76a6c0b4685e549fddcac052b1e366d",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:17:44.949Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-04-implementation",
        "repositoryId": "consumer-04",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 68071,
        "responseBytes": 667,
        "stderrBytes": 0,
        "stdoutHash": "a8553430ed0956b44193e9fb42f270cabd80c1164b977966b05ebee2b910b51c",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 468835,
        "outputTokens": 2603,
        "tokenMethod": "provider",
        "toolCalls": 11,
        "errorCode": "token-budget"
      },
      "contextBytes": 2094,
      "evidenceIds": [
        "patch-evidence:content/docs/for-agents.mdx",
        "knowledge-gap:doc-bridge-local-index-query-gate-instructions",
        "proposed-patch:content/docs/for-agents.mdx:add-local-doc-bridge-verification-subsection",
        "verification-plan:ak-docs-check-json",
        "acceptance-check:blocked-command-unavailable",
        "scope:no-files-modified"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 408576,
        "reasoningOutputTokens": 1276,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "49317671b9a9c08ecd4fdb2ff71e72a48e180d86cfd94ab535ab2804b91c854b",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:18:44.157Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-01-documentation",
        "repositoryId": "consumer-01",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 59182,
        "responseBytes": 892,
        "stderrBytes": 0,
        "stdoutHash": "893f64687c2a2acd597df82e73a7738737c6d59683782a45d3f643584e5f3e49",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 353926,
        "outputTokens": 2156,
        "tokenMethod": "provider",
        "toolCalls": 8
      },
      "contextBytes": 2048,
      "evidenceIds": [
        "documentation-evidence:README.md:159 claims eight MCP tools; src/mcp/server.ts:38-151 defines 15 MCP_TOOLS entries",
        "documentation-evidence:README.md:355 repeats the eight-tool claim; docs/mcp.md:55 repeats it",
        "review-limitation:high-confidence direct source/documentation count comparison; mandated audit could not complete because the read-only sandbox denied .doc-bridge/workflow/.lock creation",
        "recommended-next-action:update all eight-tool claims to 15, then rerun ak-docs audit documentation --json"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 286720,
        "reasoningOutputTokens": 941,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "1965187b4aa5765ffbaa637ef4a70c38b14980a662c59125c0b053150f8915ab",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:20:06.746Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-03-implementation",
        "repositoryId": "consumer-03",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 82565,
        "responseBytes": 688,
        "stderrBytes": 0,
        "stdoutHash": "1565e2c5db247b158bc51abc39d939b717baf175cd83906bbc4e1658f8289eb7",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 654409,
        "outputTokens": 3408,
        "tokenMethod": "provider",
        "toolCalls": 13,
        "errorCode": "token-budget"
      },
      "contextBytes": 2085,
      "evidenceIds": [
        "documentation-audit:d74dc3288740869e589fb35b6c3d474fe4cca005bb184e860b465932db630e26",
        "patch-evidence:target-document-not-identifiable-from-available-evidence",
        "verification-plan:ak-docs-check-json-command-not-found"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "incomplete",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 1,
      "reworkCount": 0,
      "measurements": {
        "documentationFindingCount": 1,
        "documentationCompletenessRate": 1,
        "documentationExampleRate": 0.6181818181818182,
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 578048,
        "reasoningOutputTokens": 1040,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "b1a6a72729f917bb9bcc105b14ceba361c0d6271920eabcdc2a5159dc94c6479",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:20:55.609Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-04-architecture",
        "repositoryId": "consumer-04",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 48837,
        "responseBytes": 598,
        "stderrBytes": 0,
        "stdoutHash": "c2fdb562616b83de1b6da012199aaa863691f8a6f63c6b2a7bf3f0ebc2b9f832",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 132595,
        "outputTokens": 944,
        "tokenMethod": "provider",
        "toolCalls": 3
      },
      "contextBytes": 1892,
      "evidenceIds": [
        "architecture-evidence:package.json",
        "architecture-evidence:README.md",
        "architecture-evidence:content/docs/agentskit-chat.md",
        "architecture-evidence:packages/harness/src/adapters/doc-bridge.ts",
        "acceptance:ak-docs-map:EPERM-lock"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 101376,
        "reasoningOutputTokens": 398,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "76fb1ae79cab498c5b0718b78e4d5990e64d3343835552c32301d9c8d6c23be1",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:21:43.078Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-03-implementation",
        "repositoryId": "consumer-03",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 47442,
        "responseBytes": 477,
        "stderrBytes": 0,
        "stdoutHash": "f0fd1a2ce64172ac7c9f6fcfd506495ff636be6f4a4ad9dc9b7ccfcd864a9900",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 206624,
        "outputTokens": 1792,
        "tokenMethod": "provider",
        "toolCalls": 5
      },
      "contextBytes": 2094,
      "evidenceIds": [
        "patch-evidence:docs/examples/deterministic-chat.mdx",
        "verification-plan:ak-docs check --json"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "low",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 156416,
        "reasoningOutputTokens": 932,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "6d54ee49609a7ae1b42588b425e9df26bc596c6b429b3105812c9c316f761c4f",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:22:19.896Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-03-architecture",
        "repositoryId": "consumer-03",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 36794,
        "responseBytes": 952,
        "stderrBytes": 0,
        "stdoutHash": "8c274a07fdf75e3cabbd286564c51f74098dcc8bcba502debeaa761fbfa38764",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 101849,
        "outputTokens": 713,
        "tokenMethod": "provider",
        "toolCalls": 3
      },
      "contextBytes": 1883,
      "evidenceIds": [
        "architecture-evidence:package.json",
        "architecture-evidence:pnpm-workspace.yaml",
        "architecture-evidence:docs/for-agents/architecture.md",
        "architecture-evidence:packages/chat/src/index.ts",
        "architecture-evidence:packages/protocol/src/index.ts",
        "architecture-evidence:packages/react/src/index.tsx",
        "architecture-evidence:packages/vue/src/index.ts",
        "architecture-evidence:packages/angular/src/index.ts",
        "architecture-evidence:packages/devtools/src/index.ts",
        "architecture-attention:packages/cli/src/index.ts",
        "acceptance-check:ak-docs-map-command-not-found"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 75776,
        "reasoningOutputTokens": 158,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "5af31ee3761745f7649b7539c4bf068fd853ee8f1c685f4e13db1a734b9a2faf",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:23:02.762Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-04-architecture",
        "repositoryId": "consumer-04",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 42844,
        "responseBytes": 778,
        "stderrBytes": 0,
        "stdoutHash": "e91e4a342ad4d07a2a9936bec1e4224d827a739b0d736fe1bd4e3411a6bea119",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 94140,
        "outputTokens": 771,
        "tokenMethod": "provider",
        "toolCalls": 3
      },
      "contextBytes": 1883,
      "evidenceIds": [
        "architecture-evidence:package.json",
        "architecture-evidence:pnpm-workspace.yaml",
        "architecture-evidence:README.md",
        "architecture-evidence:app/api/search/route.ts",
        "architecture-evidence:packages/playbook/package.json",
        "architecture-evidence:packages/harness/package.json",
        "attention-point:README-claims-against-observed-workspace-boundaries",
        "blocked:ak-docs-map-command-not-found"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 78848,
        "reasoningOutputTokens": 214,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "05ab4d1f248152c695ccc7fcc5e9680b1aaa0cfe30a8123f0983d6ae0c6c3e75",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:24:37.124Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-01-documentation",
        "repositoryId": "consumer-01",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 94339,
        "responseBytes": 654,
        "stderrBytes": 0,
        "stdoutHash": "a54dcdc9d28226b6fb9d407aa681ee819ab523c6b2ab6f377fb3bcb325e990f8",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 611370,
        "outputTokens": 3289,
        "tokenMethod": "provider",
        "toolCalls": 15,
        "errorCode": "token-budget"
      },
      "contextBytes": 2057,
      "evidenceIds": [
        "documentation-issue:docs/spec/cli.md:8",
        "source-evidence:package.json:7-10",
        "confidence:high",
        "recommended-next-action:correct-only-published-binary-claim-to-account-for-ak-verify",
        "review-limitation:current-audit-command-blocked-by-EPERM-lock-creation",
        "prior-audit-evidence:.codex/verification/runs/1788196344375-21778:documentation-audit-passed"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "cachedInputTokens": 549120,
        "reasoningOutputTokens": 1429,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "b139cca4206d8d24dc85694991e2c6d92a758097f8689d4dbbcd66461ab5073d",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:26:31.828Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-04-implementation",
        "repositoryId": "consumer-04",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 114679,
        "responseBytes": 654,
        "stderrBytes": 0,
        "stdoutHash": "c78e3ed7e2e8060a11fc0e5db331f7031a8be34403c212c166ac661a2cc044ee",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 400023,
        "outputTokens": 2424,
        "tokenMethod": "provider",
        "toolCalls": 10,
        "errorCode": "token-budget"
      },
      "contextBytes": 2093,
      "evidenceIds": [
        "patch-evidence:docs/ecosystem-doc-quality-plan.md:9-18",
        "patch-evidence:README.md:151-159:add-project-bullet-between-Doc-Bridge-and-private-repository",
        "verification-plan:ak-docs-check-json-after-approval",
        "verification-blocked:read-only-workflow-lock"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 342528,
        "reasoningOutputTokens": 925,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "d82a683abdd285bc8fa94138cc1b64ddd8c4fcec50c953f22627d209e4e9835a",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:27:57.745Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-04-implementation",
        "repositoryId": "consumer-04",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 85890,
        "responseBytes": 621,
        "stderrBytes": 0,
        "stdoutHash": "5d8cf9712ede644a77115ea1d37ea84bf9b100cd853de2e436b4ea0fbd7cab85",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 347787,
        "outputTokens": 1861,
        "tokenMethod": "provider",
        "toolCalls": 9
      },
      "contextBytes": 2084,
      "evidenceIds": [
        "patch-evidence:.doc-bridge/workflow/artifacts/reconcile-9777cd3e46fbfecd325ed4caccf5776be28dfcf6d5af61f77378782180f6afeb.json",
        "patch-evidence:packages/playbook/README.md",
        "verification-plan:pnpm exec ak-docs check --json"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "documentationFindingCount": 1,
        "errorRate": 1,
        "cachedInputTokens": 295680,
        "reasoningOutputTokens": 425,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "4f355a82c4718b620d6da83103b72e881c940cbd225016261849368f652743b7",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:28:49.892Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-06-architecture",
        "repositoryId": "consumer-06",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 52123,
        "responseBytes": 1093,
        "stderrBytes": 0,
        "stdoutHash": "5187df30b2d3d85c7ad3af008a8a0d21af86ab4febeb4c413744cb1adf9fabad",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 293891,
        "outputTokens": 2172,
        "tokenMethod": "provider",
        "toolCalls": 6
      },
      "contextBytes": 1884,
      "evidenceIds": [
        "artifact:pnpm-workspace.yaml — workspace packages/apps topology",
        "artifact:docs/architecture/layers.md — L0 os-core → L1 domain → L2 adapters → L3 composition roots; downward dependencies",
        "artifact:docs/for-agents/architecture.md — package.json-derived depends-on relations, including desktop-shell/os-headless/os-cli/os-desktop connections",
        "artifact:README.md — OS owns orchestration/workspace/UX; AgentsKit supplies adapters, tools, RAG, sandbox, and observability primitives",
        "artifact:docs/adr/0064-layered-architecture.md — review boundary: os-sandbox is explicitly mixed, combining a sandbox port with concrete runtimes",
        "blocked:ak-docs-map --json — EPERM while creating temporary file in read-only workspace"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 233728,
        "reasoningOutputTokens": 830,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "87707f0c99d9a9814f1dad998605f9961e5c72b8b47546948417e6170f17fc30",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:29:24.258Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-01-discovery",
        "repositoryId": "consumer-01",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 34345,
        "responseBytes": 637,
        "stderrBytes": 0,
        "stdoutHash": "3ed621d9935f8320752418cd90a9894cb0ebee4889f91b67284f2d3544742872",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 135925,
        "outputTokens": 1342,
        "tokenMethod": "provider",
        "toolCalls": 4
      },
      "contextBytes": 1865,
      "evidenceIds": [
        "entrypoint-evidence",
        "package.json",
        "bin/ak-docs.js",
        "src/index.ts",
        "src/cli/program.ts",
        "src/mcp/server.ts",
        "docs/for-agents.md",
        "docs/agent-corpus/INDEX.md",
        "docs/guides/cli-map.md",
        "docs/guides/mcp-agents.md",
        "node-bin-discover-success",
        "ak-docs-command-not-found"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 90112,
        "reasoningOutputTokens": 688,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "6c4b6772cf0b5223876690fc416d4649f9abc728eadef4a1dee6802739355832",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:30:36.784Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-05-architecture",
        "repositoryId": "consumer-05",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 72500,
        "responseBytes": 1120,
        "stderrBytes": 0,
        "stdoutHash": "85f44d37fb623372a5120e790bced3c41a9823ec7f3b4009e6ceaae180ea2ee0",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 293465,
        "outputTokens": 3101,
        "tokenMethod": "provider",
        "toolCalls": 19
      },
      "contextBytes": 1893,
      "evidenceIds": [
        "docs/architecture.md:registry-source-and-tests->build-registry.mjs->public/r-agent-bundles",
        "docs/architecture.md:independent-validation-evidence->build-registry.mjs",
        "docs/architecture.md:build-registry.mjs->verified-deterministic-artifact",
        "docs/architecture.md:public/r-agent-bundles->external-Registry-Fumadocs-site",
        "docs/architecture.md:deterministic-artifact->AgentsChat-local-answer-plane->canonical-citation-or-trusted-Ask-backend",
        "scripts/lib/deterministic-discovery.mjs:1-12->@agentskit/chat/protocol",
        "boundary-review:ownership-and-integrity-at-the-registry-generated-artifacts-to-external-Fumadocs-and-AgentsChat-boundaries",
        "acceptance-blocked:ak-docs-map-json-could-not-create-.doc-bridge/workflow/.lock-under-read-only-filesystem"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 240640,
        "reasoningOutputTokens": 1115,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "cef9dae368da0e09096906a2b06dc21ff6aa72744874514bab4fd07d4c8829ee",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:31:18.784Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-04-discovery",
        "repositoryId": "consumer-04",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 41978,
        "responseBytes": 602,
        "stderrBytes": 0,
        "stdoutHash": "99629eff3aa517a93d520cb7870a008226ef300b8eae570dca5a122af0745d19",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 114885,
        "outputTokens": 778,
        "tokenMethod": "provider",
        "toolCalls": 4
      },
      "contextBytes": 1864,
      "evidenceIds": [
        "entrypoint-evidence:package.json",
        "entrypoint-evidence:README.md",
        "entrypoint-evidence:AGENTS.md",
        "entrypoint-evidence:docs/for-agents/INDEX.md",
        "entrypoint-evidence:doc-bridge.config.json",
        "discovery-check:ak-docs-command-not-found"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 83712,
        "reasoningOutputTokens": 228,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "bc3a3484061aea253f8c99867e659c3491927208aea372087e40add2a944f108",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:31:53.374Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-06-discovery",
        "repositoryId": "consumer-06",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 34567,
        "responseBytes": 510,
        "stderrBytes": 0,
        "stdoutHash": "abdbca19604d7dcbffd13bb20b5208d637fcf23d5afe9ea3df3288b321d9e8ae",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 90464,
        "outputTokens": 577,
        "tokenMethod": "provider",
        "toolCalls": 2
      },
      "contextBytes": 1855,
      "evidenceIds": [
        "entrypoint-evidence:package.json",
        "entrypoint-evidence:AGENTS.md",
        "entrypoint-evidence:docs/for-agents/INDEX.md"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 3,
        "cachedInputTokens": 58624,
        "reasoningOutputTokens": 236,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "e5c58645770fc7d2bbaec9f47ad792e67b029824d795028354d14d3aeebdf25c",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:37:37.157Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-05-documentation",
        "repositoryId": "consumer-05",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "timed-out",
        "exitCode": null,
        "signal": "SIGTERM",
        "durationMs": 343753,
        "responseBytes": 0,
        "stderrBytes": 0,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "errorCode": "timeout"
      },
      "contextBytes": 2048,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "5d416465e051b126c5552342131bb25c1f5c7753f53395f03b744d607f569e79",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:39:05.108Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-05-discovery",
        "repositoryId": "consumer-05",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 87917,
        "responseBytes": 850,
        "stderrBytes": 0,
        "stdoutHash": "40ec173cb8735b514c0f3ce93bac6e8e44969ba0447c9d3fc9d648e6e7836664",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 466236,
        "outputTokens": 2961,
        "tokenMethod": "provider",
        "toolCalls": 11,
        "errorCode": "token-budget"
      },
      "contextBytes": 1856,
      "evidenceIds": [
        "docs/for-agents/index.md",
        "docs/architecture.md",
        "docs/for-agents/registry-architecture.md",
        "docs/for-agents/registry-discovery.md",
        "docs/for-agents/registry-catalog.md",
        "README.md",
        "CONTRIBUTING.md",
        "package.json",
        "scripts/build-registry.mjs",
        "scripts/build-discovery.mjs",
        "scripts/lib/deterministic-discovery.mjs",
        "public/deterministic/",
        "public/r/",
        "ak-docs discover --json: structured discovery succeeded with local bin PATH; cleanup artifact remained at .codex/verification/latest.json"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 1,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 417536,
        "reasoningOutputTokens": 1244,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "725503570a2c76bd5673806397ce08550ea525e4be32256cebaf319ae51e50f8",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:39:55.330Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-04-architecture",
        "repositoryId": "consumer-04",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 50174,
        "responseBytes": 1020,
        "stderrBytes": 0,
        "stdoutHash": "884cb27fdec2c04ae9751e0b6d1128343547b6696e7459d19e404a794363f7d2",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 180210,
        "outputTokens": 2233,
        "tokenMethod": "provider",
        "toolCalls": 5
      },
      "contextBytes": 1884,
      "evidenceIds": [
        "architecture-evidence: Next.js/Fumadocs host -> content/docs via lib/source.ts",
        "architecture-evidence: lib/discovery.ts imports @agentskit/chat/protocol and @agentskit/core, with backend fallback",
        "architecture-evidence: @agentskit/harness exports verification, runtime, policy, session, metrics, and Doc Bridge adapter APIs",
        "architecture-evidence: @agentskit/playbook provides the zero-dependency gate CLI",
        "architecture-attention: generated Doc Bridge configuration/index is a review boundary between repository documentation and harness context resolution",
        "architecture-check: ak-docs map --json unavailable because ak-docs was not installed"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 129792,
        "reasoningOutputTokens": 1071,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "93363725a4172a25586759366cab49049959fdfef3e64d42a6dc5769a0862b8a",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:40:29.890Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-04-discovery",
        "repositoryId": "consumer-04",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 34532,
        "responseBytes": 660,
        "stderrBytes": 0,
        "stdoutHash": "df66161310c5f35f2161adf91b6619558c280dc692bb76d89172d37fd93169ab",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 72750,
        "outputTokens": 661,
        "tokenMethod": "provider",
        "toolCalls": 2
      },
      "contextBytes": 1855,
      "evidenceIds": [
        "entrypoint-evidence:AGENTS.md",
        "entrypoint-evidence:README.md",
        "entrypoint-evidence:package.json",
        "entrypoint-evidence:content/docs/index.mdx",
        "entrypoint-evidence:content/docs/meta.json",
        "entrypoint-evidence:packages/playbook/package.json",
        "entrypoint-evidence:packages/harness/package.json"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 42496,
        "reasoningOutputTokens": 223,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "70de666add96fecdf058178c1544863f13ec533a73c3ab1d271e615975a7f703",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:41:51.782Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-04-discovery",
        "repositoryId": "consumer-04",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 81868,
        "responseBytes": 780,
        "stderrBytes": 0,
        "stdoutHash": "0c4152c14b9db02dcb2b35896aa1d9eb80c2002abe1bce3f1be03c17efae4764",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 575362,
        "outputTokens": 3202,
        "tokenMethod": "provider",
        "toolCalls": 11,
        "errorCode": "token-budget"
      },
      "contextBytes": 1856,
      "evidenceIds": [
        "entrypoint-evidence",
        "app/page.tsx:188",
        "app/layout.tsx:87",
        "app/docs/[[...slug]]/page.tsx:31",
        "app/raw/[...path]/route.ts:33",
        "app/llms.txt/route.ts:104",
        "app/llms-full.txt/route.ts:38",
        "app/for-agents/page.tsx:3",
        "packages/playbook/bin/agents-playbook.mjs",
        "package.json",
        "AGENTS.md",
        "doc-bridge.config.json",
        "content/docs/index.mdx",
        "content/docs/for-agents.mdx",
        "content/docs/agentskit-chat.md",
        "discovery-check:exit-0"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "success",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 1,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 521984,
        "reasoningOutputTokens": 1300,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "199902a07cd565aa6d58362bca05a14ac7f80234626b5830dfe51d2d5c774c00",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:42:44.397Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-03-documentation",
        "repositoryId": "consumer-03",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 52587,
        "responseBytes": 398,
        "stderrBytes": 0,
        "stdoutHash": "cafbd7882930172b2edb554fbe5f5b43424fe98d7627a4c2b178eb4c07acaec7",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 310697,
        "outputTokens": 2013,
        "tokenMethod": "provider",
        "toolCalls": 8
      },
      "contextBytes": 2057,
      "evidenceIds": [
        "documentation-evidence",
        "review-limitation"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "success",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 255744,
        "reasoningOutputTokens": 752,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "b2c3d505d62944d36149fdfa709598e723c0b59555fa8a55dbe3fde2f88bc0e0",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:43:26.442Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-03-architecture",
        "repositoryId": "consumer-03",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 42001,
        "responseBytes": 548,
        "stderrBytes": 0,
        "stdoutHash": "4078ba8b14df294f4383426595e9af82dd25ced543c0694d3c94bb9e80c2a4fd",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 191926,
        "outputTokens": 1475,
        "tokenMethod": "provider",
        "toolCalls": 5
      },
      "contextBytes": 1884,
      "evidenceIds": [
        "docs/architecture/overview.md",
        "docs/architecture/upstream-adoption.md",
        "pnpm-workspace.yaml",
        "packages/*/package.json",
        "observed-imports",
        "architecture-check:ak-docs-map-command-unavailable"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 155136,
        "reasoningOutputTokens": 688,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "2a56708d32616c0fb2fc9a0580d8c3b54cc9dac09f730b8deea2e84faf9d0f03",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:44:31.721Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-03-documentation",
        "repositoryId": "consumer-03",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 65241,
        "responseBytes": 582,
        "stderrBytes": 0,
        "stdoutHash": "6b65287ed4c091209e7e8a8a5b714fc58a8c13251f69d874b9d1e87937b339fc",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 229318,
        "outputTokens": 1418,
        "tokenMethod": "provider",
        "toolCalls": 7
      },
      "contextBytes": 2056,
      "evidenceIds": [
        "docs/getting-started/index.mdx:13",
        "docs/index.mdx:20",
        "README.md:54",
        "package.json:3",
        "acceptance:ak-docs-audit-unavailable",
        "limitation:semantic-classification-not-human-verified"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 1,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 202752,
        "reasoningOutputTokens": 479,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "bfcd280eede502b48b843b4444d67721962a43753b234e830a7eeeb9e3d54605",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:45:05.295Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-03-architecture",
        "repositoryId": "consumer-03",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 33543,
        "responseBytes": 420,
        "stderrBytes": 0,
        "stdoutHash": "5795b0cb0937e3ebbc56bd5ad9a4950552035842799afb49c954addae219d21c",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 99392,
        "outputTokens": 663,
        "tokenMethod": "provider",
        "toolCalls": 3
      },
      "contextBytes": 1892,
      "evidenceIds": [
        "architecture-evidence"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 64768,
        "reasoningOutputTokens": 226,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "1720a8a2c63cd2f9467efb4dd29d6e5696894558bacf1da69a32fbe417bee6e4",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:46:01.499Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-02-implementation",
        "repositoryId": "consumer-02",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 56173,
        "responseBytes": 386,
        "stderrBytes": 0,
        "stdoutHash": "b6dd47e6382cd012c9994bd5f17681a2f2b2613b048f75c2a9258e06a2805b9e",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 184341,
        "outputTokens": 1142,
        "tokenMethod": "provider",
        "toolCalls": 6
      },
      "contextBytes": 2093,
      "evidenceIds": [
        "verification-plan"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "low",
      "safetyOutcome": "safe",
      "clarificationRequests": 1,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 146944,
        "reasoningOutputTokens": 439,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "597024183d553ebd245d2ae62955a84595df1bd091483c566cabc2ce1ac35069",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:47:27.972Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-01-implementation",
        "repositoryId": "consumer-01",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 86443,
        "responseBytes": 972,
        "stderrBytes": 0,
        "stdoutHash": "c346f593eb7680fd22110dd48779dc9e9a7b31b025ca0df9a1cce1e9ab9770b5",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 296274,
        "outputTokens": 1663,
        "tokenMethod": "provider",
        "toolCalls": 8
      },
      "contextBytes": 2093,
      "evidenceIds": [
        "patch-evidence:docs/spec/config-v1.md:836 incorrectly marks ak-docs chat as planned; src/cli/program.ts:1600-1613 implements the command and src/intelligence/chat.ts:75-95 implements chat execution",
        "proposal:documentation-only;replace \"planned; `intelligence.*`\" with \"`intelligence.*`\" in docs/spec/config-v1.md:836;no-source-changes",
        "verification-plan:after approval run ak-docs check --json and review the one-line diff",
        "verification-blocked:node bin/ak-docs.js check --json failed because the read-only workspace prevented creation of .doc-bridge/workflow/.lock"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 252928,
        "reasoningOutputTokens": 351,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "48f602d2bb202f45c2c18f9496ef229a435e07673393703af17f5a0965522f13",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:47:46.474Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-02-discovery",
        "repositoryId": "consumer-02",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 18470,
        "responseBytes": 364,
        "stderrBytes": 0,
        "stdoutHash": "1a120bbb073f001debcec398f069063af6d3caaa44cb4c2f821d0a0471a736c3",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 44555,
        "outputTokens": 277,
        "tokenMethod": "provider",
        "toolCalls": 1
      },
      "contextBytes": 1864,
      "evidenceIds": [],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "low",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 21248,
        "reasoningOutputTokens": 126,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "8a59b4e8eb239565069c1bc377780d043975cd27467563200c03a3fe437a16ba",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:48:32.274Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-02-documentation",
        "repositoryId": "consumer-02",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 45773,
        "responseBytes": 485,
        "stderrBytes": 0,
        "stdoutHash": "be78231bfd6d8117828e6bb5d333f9a401b674e0e9b9050b5d3d4f1d54d7ac35",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 123486,
        "outputTokens": 915,
        "tokenMethod": "provider",
        "toolCalls": 3
      },
      "contextBytes": 2047,
      "evidenceIds": [
        "evidence-617b0b10b5132040199d1d1c3280fdaf",
        "evidence-7bf1f03eb3f718f3f618a890dc694a67"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 87040,
        "reasoningOutputTokens": 257,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "4a5e9aa1272402c672028981997c41c5edec6fe97ea78145a56092a2428ecb41",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:49:57.111Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-05-implementation",
        "repositoryId": "consumer-05",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 84809,
        "responseBytes": 686,
        "stderrBytes": 0,
        "stdoutHash": "25cb11e8b6a765ffc012c046aca08bc9f42bedcdae1eacbd8d316d596acb9b24",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 275895,
        "outputTokens": 1624,
        "tokenMethod": "provider",
        "toolCalls": 7
      },
      "contextBytes": 2084,
      "evidenceIds": [
        "evidence-a1fd6fa3a7f9cc0884d4cbbec8c615a9",
        "verification-plan:after approval run ak-docs check --json and review that the relevant RELATION_UNDOCUMENTED diagnostic is resolved without new diagnostics; the current read-only run was blocked by EPERM creating .doc-bridge/workflow/.lock"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "success",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "documentationFindingCount": 5117,
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 230656,
        "reasoningOutputTokens": 421,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "d0ae07f544ac141953604bfa5d44da9553cfd58ad0c30ecea73ff89d8ca78cdc",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:51:21.246Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-06-implementation",
        "repositoryId": "consumer-06",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 84104,
        "responseBytes": 512,
        "stderrBytes": 0,
        "stdoutHash": "4804b3b7eaed2793637f268e700ff711a3b48223e77168079efe7b710d398549",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 304526,
        "outputTokens": 2014,
        "tokenMethod": "provider",
        "toolCalls": 7
      },
      "contextBytes": 2094,
      "evidenceIds": [
        "docs/testing/provider-live-certification.md",
        "docs/security/connections-byok-containment.md",
        "acceptance-check:ak-docs-check:blocked-read-only-environment"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "incomplete",
      "evidenceQuality": "low",
      "safetyOutcome": "safe",
      "clarificationRequests": 1,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 280320,
        "reasoningOutputTokens": 909,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "e18ec764390a387e6bda1c72c0c30f5edee758ef064eaa3713649083a55c74db",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:52:00.684Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-04-discovery",
        "repositoryId": "consumer-04",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 39407,
        "responseBytes": 1097,
        "stderrBytes": 0,
        "stdoutHash": "da60ad252dabf56a29f175a9b42e53dd1a998adb532102945d287637207bcfb0",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 162026,
        "outputTokens": 1655,
        "tokenMethod": "provider",
        "toolCalls": 3
      },
      "contextBytes": 1865,
      "evidenceIds": [
        "entrypoint-evidence: app/layout.tsx:87 RootLayout and AskWidget shell; app/page.tsx homepage; app/docs/[[...slug]]/page.tsx documentation route; app/llms.txt/route.ts machine-readable documentation map",
        "ownership-evidence: AGENTS.md:3-9 repository owns Agents Playbook corpus and Next/Fumadocs host; chat state and portable components belong to AgentsKit and AgentsKit Chat; doc-bridge.config.json routing declarations",
        "canonical-docs: content/docs/for-agents.mdx retrieval path; content/docs/discovery.mdx local-first discovery; content/docs/index.mdx corpus index; /llms.txt and /llms-full.txt machine-readable canonical surfaces",
        "discovery-check: pnpm exec ak-docs discover --json exited successfully with structured discovery snapshot"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "success",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 1,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 118528,
        "reasoningOutputTokens": 836,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "63557d50e657901068bb649c0956e1274c8287b23a11078640ce69b64bf8cee0",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:53:32.688Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-02-documentation",
        "repositoryId": "consumer-02",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 91952,
        "responseBytes": 487,
        "stderrBytes": 0,
        "stdoutHash": "ed12451b1c6068763e139d80da523b30aee337abb4c9713d4c75b7d35c4bafb5",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 230961,
        "outputTokens": 1478,
        "tokenMethod": "provider",
        "toolCalls": 8
      },
      "contextBytes": 2056,
      "evidenceIds": [
        "evidence-7082c0a6f2e17a5883c8e9be9f0fbc37",
        "evidence-e56acaff9b9afc2a8696739b857afea6"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 190464,
        "reasoningOutputTokens": 428,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "7c98c76ec778d68d1b5e0086e86f50fed8ba85223786efb715a16529ca996149",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:54:39.155Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-04-documentation",
        "repositoryId": "consumer-04",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 66438,
        "responseBytes": 770,
        "stderrBytes": 0,
        "stdoutHash": "742051ba9a9585e7296277ff4ccd10f5523d1c4db068e41d8b393d177deec930",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 245315,
        "outputTokens": 1435,
        "tokenMethod": "provider",
        "toolCalls": 6
      },
      "contextBytes": 2047,
      "evidenceIds": [
        "packages/harness/package.json:2",
        "content/docs/index.mdx:2",
        "content/docs/index.mdx:7",
        ".doc-bridge/workflow/artifacts/reconcile-9777cd3e46fbfecd325ed4caccf5776be28dfcf6d5af61f77378782180f6afeb.json:79",
        "limitation:cached-reconciliation-source-revision-b1c5affa-differs-from-current-e6079d4e",
        "next-action:add-current-docbridge-coverage-for-package-agentskit-harness-and-rerun-audit"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 197888,
        "reasoningOutputTokens": 470,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "d36ce41497b526fc1e20adac5b97b2d6b0c7c34e628e513daa66592e7f13caf3",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:57:05.389Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-06-documentation",
        "repositoryId": "consumer-06",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 146206,
        "responseBytes": 799,
        "stderrBytes": 0,
        "stdoutHash": "27b1eb75acd854a10ae8f0d1cb021184da985119ca341c493d04ffa545bb5746",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 360856,
        "outputTokens": 3919,
        "tokenMethod": "provider",
        "toolCalls": 22
      },
      "contextBytes": 2057,
      "evidenceIds": [
        "evidence-f57566b072a63ce67f14b5ec0f71fe3d",
        "documentation-evidence:docs/internal/index.generated.json:96 records the current AGENTS.md inventory as 76 package dirs, contradicting README.md:237.",
        "evidence-682662ff61aa5ba3b033d1922a7f2521",
        "recommended-next-action:update README.md:237 to describe 76 active package workspaces and clarify that artifact-only packages/pack-bundle is excluded from workspace counts; regenerate affected indexes."
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "success",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 299776,
        "reasoningOutputTokens": 1444,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "3c44651771798193606a855867eb50fa8d4c931411a0067de46ce456145b9579",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:58:02.062Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-03-discovery",
        "repositoryId": "consumer-03",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 56624,
        "responseBytes": 1583,
        "stderrBytes": 0,
        "stdoutHash": "badb87f3ff990e9e2339edb728e65e89d75660cabd6d0412fb18440857d939f0",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 255688,
        "outputTokens": 2056,
        "tokenMethod": "provider",
        "toolCalls": 7
      },
      "contextBytes": 1865,
      "evidenceIds": [
        "discovery-check:ak-docs discover --json ok=true",
        "entrypoint-evidence:package.json workspace scripts and docs bridge commands",
        "entrypoint-evidence:packages/chat/src/index.ts -> @agentskit/chat; owner docs/for-agents/packages/chat.md",
        "entrypoint-evidence:packages/protocol/src/index.ts -> @agentskit/chat-protocol; owner docs/for-agents/packages/protocol.md",
        "entrypoint-evidence:packages/server/src/index.ts -> @agentskit/chat-server; owner docs/for-agents/packages/server.md",
        "entrypoint-evidence:packages/{react,vue,svelte,solid,angular,react-native,ink}/src/index.* -> native renderer packages; owners docs/for-agents/packages/*.md",
        "entrypoint-evidence:packages/cli/src/index.ts and bin -> @agentskit/chat-cli; owner docs/for-agents/packages/cli.md",
        "entrypoint-evidence:apps/example-shared/src/index.ts -> framework-neutral shared definition; owner docs/for-agents/apps/example-shared.md",
        "entrypoint-evidence:apps/docs/lib/{chat-definition,ask-handler,knowledge}.ts -> Fumadocs host seams; owner docs/for-agents/apps/docs.md",
        "canonical-documentation:docs/for-agents/index.md ownership index; docs/architecture/overview.md boundaries; docs/architecture/upstream-adoption.md upstream mapping; docs/ root canonical corpus"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "success",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 1,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 178688,
        "reasoningOutputTokens": 837,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "31dc6549fe65cf5300561569554fa3d0bbaf9e72f0de34eca7afe1a6c1fd690b",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T18:58:56.424Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-05-documentation",
        "repositoryId": "consumer-05",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 54305,
        "responseBytes": 612,
        "stderrBytes": 0,
        "stdoutHash": "07c82d09bd380c9fe623c9b1e07983484d707fdd7999eab482b1bb5af96daab3",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 138858,
        "outputTokens": 1016,
        "tokenMethod": "provider",
        "toolCalls": 5
      },
      "contextBytes": 2047,
      "evidenceIds": [
        "documentation-audit:1788121223061-35463",
        "68ccc09226589f00a515b3b024e2f981aade87109648baec39b732c00592cc66",
        "review-limitation:DOCUMENTATION_SEMANTICS_NOT_ANALYZED"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 1,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "documentationExampleRate": 0.16145833333333334,
        "cachedInputTokens": 118272,
        "reasoningOutputTokens": 355,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "83cb61bf8077eea45e109a6eaaa67460c6b7282e4e1d6d18863a71b4268143d2",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T19:00:11.030Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-02-implementation",
        "repositoryId": "consumer-02",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 74435,
        "responseBytes": 551,
        "stderrBytes": 0,
        "stdoutHash": "d8fe0772eabb8084d874367b7c2352256884776cf0960cf3f99b53e7534b72f2",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 648671,
        "outputTokens": 2575,
        "tokenMethod": "provider",
        "toolCalls": 11,
        "errorCode": "token-budget"
      },
      "contextBytes": 2094,
      "evidenceIds": [
        "patch-evidence:apps/docs-next/content/docs/reference/packages/runtime.mdx",
        "patch-evidence:packages/runtime/src/index.ts",
        "verification-plan:ak-docs-check-json-blocked-by-read-only-environment"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 573440,
        "reasoningOutputTokens": 877,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "53238bc95b6b17b6073a9a4b88971d4275238d2051e2703386dbda3e076d19de",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T19:01:09.795Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-06-documentation",
        "repositoryId": "consumer-06",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 58700,
        "responseBytes": 813,
        "stderrBytes": 0,
        "stdoutHash": "753cd38450039d9a240bd2c22c7a9a288bad24937ec2a697bc374b76617071ba",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 135488,
        "outputTokens": 1080,
        "tokenMethod": "provider",
        "toolCalls": 4
      },
      "contextBytes": 2056,
      "evidenceIds": [
        "evidence-19f322dbc90749607d39a6434e2de729",
        "corroboration:docs/internal/assistant-confidence-demo-2026-07-09.md:28 independently records the same P2 discrepancy and recommended wording change.",
        "review-limitation:ak-docs audit documentation --json could not run because ak-docs was not installed or available on PATH; classification is based on deterministic repository inspection, not the required audit harness."
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 120576,
        "reasoningOutputTokens": 277,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "2420ba71a0e1a2787429e05f2f32f7d18afaaf4d9e3e36a1fc51d6cb40d54026",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T19:02:19.980Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-02-implementation",
        "repositoryId": "consumer-02",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 69814,
        "responseBytes": 582,
        "stderrBytes": 0,
        "stdoutHash": "127b30ffc2d9e8e26a359fb70ef6be74b7c780145ca2ce8a9b17755a1f46570c",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 522786,
        "outputTokens": 2450,
        "tokenMethod": "provider",
        "toolCalls": 10,
        "errorCode": "token-budget"
      },
      "contextBytes": 2085,
      "evidenceIds": [
        "patch-evidence:current-audit-reports-structureGapCount-0",
        "patch-evidence:current-audit-reports-only-not-analyzed-semantic-limitation",
        "verification-plan:ak-docs-check-json-command-unavailable"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 444928,
        "reasoningOutputTokens": 1143,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "f2cec9e0911c7c90af89e1b5b94b5c2ddb8d2fd7cf0b6e2a4679ef9974d0cf1f",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T19:03:14.576Z",
      "runId": "phase-8-ab-baseline-recovery-01",
      "planHash": "42d96e1152014318eadd2e0790259efc0ea96b138a1dcc4ac60c182ec357215f",
      "task": {
        "taskId": "consumer-03-documentation",
        "repositoryId": "consumer-03",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 54561,
        "responseBytes": 443,
        "stderrBytes": 0,
        "stdoutHash": "e69b4527ba98e851d4c6740e144a96ca755ba0ec190fb91dbcbfb2a4331cd6a8",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 248216,
        "outputTokens": 2289,
        "tokenMethod": "provider",
        "toolCalls": 6
      },
      "contextBytes": 2048,
      "evidenceIds": [
        "documentation-evidence",
        "review-limitation"
      ],
      "round": "ab-baseline-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "documentationFindingCount": 1,
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 175360,
        "reasoningOutputTokens": 1146,
        "stderrBytes": 0
      },
      "adjudication": {
        "status": "pending"
      },
      "contentHash": "56b2d2a056ba125f12d198414aa548da69f22425c84aa5a777aa29494d7651b7",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T19:27:00.711Z",
      "runId": "phase-9-ab-adjudicated-cost-01",
      "planHash": "7d03ca649d4d4d23cc9c7cb7237aba6e51697d04d2d8ac1912101ccb439b08b8",
      "task": {
        "taskId": "consumer-05-discovery",
        "repositoryId": "consumer-05",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 110746,
        "responseBytes": 597,
        "stderrBytes": 0,
        "stdoutHash": "40cc454606c9a0ebdb7cb83db2041822d1e64f8553f587483866ff924f497989",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 583826,
        "outputTokens": 4006,
        "tokenMethod": "provider",
        "toolCalls": 11,
        "errorCode": "token-budget"
      },
      "contextBytes": 1865,
      "evidenceIds": [
        "entrypoint-evidence",
        "package.json",
        "docs/for-agents/index.md",
        "docs/for-agents/registry-architecture.md",
        "docs/for-agents/registry-discovery.md",
        "docs/architecture.md",
        "scripts/build-discovery.mjs",
        "scripts/lib/deterministic-discovery.mjs"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "success",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 1,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 519936,
        "reasoningOutputTokens": 1731,
        "stderrBytes": 0,
        "providerTokenCostUnits": 587832
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "blocked",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "0fba9b24d5feca95f3cdb021afc54647d6472fd79d643840d283ce121eec90d9",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T19:29:15.801Z",
      "runId": "phase-9-ab-adjudicated-cost-01",
      "planHash": "7d03ca649d4d4d23cc9c7cb7237aba6e51697d04d2d8ac1912101ccb439b08b8",
      "task": {
        "taskId": "consumer-03-implementation",
        "repositoryId": "consumer-03",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 135049,
        "responseBytes": 578,
        "stderrBytes": 0,
        "stdoutHash": "8ba61964afd3712e6e4562e8d8c288d4bb721cc13265bf8a974595c3c2062442",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 687823,
        "outputTokens": 2531,
        "tokenMethod": "provider",
        "toolCalls": 16,
        "errorCode": "token-budget"
      },
      "contextBytes": 2084,
      "evidenceIds": [
        "patch-evidence:docs/for-agents/apps/docs.md:document-apps-docs-demo-knowledge-and-demo-ask-boundaries",
        "verification-plan:ak-docs-check-json-blocked-by-read-only-workflow-lock"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 616448,
        "reasoningOutputTokens": 660,
        "stderrBytes": 0,
        "providerTokenCostUnits": 690354
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "blocked",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "04fe1ff4823b97c98efcf821bd1b33367820626ed9c86a821db6568b661881b5",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T19:30:07.355Z",
      "runId": "phase-9-ab-adjudicated-cost-01",
      "planHash": "7d03ca649d4d4d23cc9c7cb7237aba6e51697d04d2d8ac1912101ccb439b08b8",
      "task": {
        "taskId": "consumer-06-implementation",
        "repositoryId": "consumer-06",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 51512,
        "responseBytes": 449,
        "stderrBytes": 0,
        "stdoutHash": "b7b6166ba12cb751f77d3c5022ab123c27a2698ac22e3e664ef82513ba8f181d",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 324730,
        "outputTokens": 1836,
        "tokenMethod": "provider",
        "toolCalls": 9
      },
      "contextBytes": 2085,
      "evidenceIds": [
        "docs/testing/provider-live-certification.md",
        "docs/security/connections-byok-containment.md"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 293376,
        "reasoningOutputTokens": 907,
        "stderrBytes": 0,
        "providerTokenCostUnits": 326566
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "926ef39c19d05dc34dfd505ba3a19e34d161721df08f02fa3faa727de222bcc3",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T19:30:49.979Z",
      "runId": "phase-9-ab-adjudicated-cost-01",
      "planHash": "7d03ca649d4d4d23cc9c7cb7237aba6e51697d04d2d8ac1912101ccb439b08b8",
      "task": {
        "taskId": "consumer-01-discovery",
        "repositoryId": "consumer-01",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 42569,
        "responseBytes": 639,
        "stderrBytes": 0,
        "stdoutHash": "2a9f4781ee26223510640596c86fcc930f96dadff773bb8f420d3a97e46490d2",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 189948,
        "outputTokens": 1478,
        "tokenMethod": "provider",
        "toolCalls": 5
      },
      "contextBytes": 1856,
      "evidenceIds": [
        "entrypoint:package.json:bin.ak-docs",
        "entrypoint:bin/ak-docs.js",
        "ownership:doc-bridge.config.json:routing.options.ownership",
        "canonical-doc:docs/index.md",
        "canonical-doc:docs/for-agents.md",
        "agent-corpus:docs/agent-corpus/OVERVIEW.md",
        "discovery-check:node-bin-ak-docs-discover-json"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 148992,
        "reasoningOutputTokens": 783,
        "stderrBytes": 0,
        "providerTokenCostUnits": 191426
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "e08a999390278f825527d60181a8d59bc7c9a6c9519a52ce10d5cf18ed1e2087",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T19:31:21.476Z",
      "runId": "phase-9-ab-adjudicated-cost-01",
      "planHash": "7d03ca649d4d4d23cc9c7cb7237aba6e51697d04d2d8ac1912101ccb439b08b8",
      "task": {
        "taskId": "consumer-01-implementation",
        "repositoryId": "consumer-01",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 31464,
        "responseBytes": 972,
        "stderrBytes": 0,
        "stdoutHash": "d3332eb305ff09f62b8c51944eef0dcb03adf450cc333d54eb89d414ade57e89",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 171799,
        "outputTokens": 1118,
        "tokenMethod": "provider",
        "toolCalls": 4
      },
      "contextBytes": 2094,
      "evidenceIds": [
        "patch-evidence:docs/spec/config-v1.md:836 incorrectly marks ak-docs chat as planned; src/cli/program.ts:1600-1613 implements the command and src/intelligence/chat.ts:75-95 implements chat execution",
        "proposal:documentation-only;replace \"planned; `intelligence.*`\" with \"`intelligence.*`\" in docs/spec/config-v1.md:836;no-source-changes",
        "verification-plan:after approval run ak-docs check --json and review the one-line diff",
        "verification-blocked:node bin/ak-docs.js check --json failed because the read-only workspace prevented creation of .doc-bridge/workflow/.lock"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 126720,
        "reasoningOutputTokens": 413,
        "stderrBytes": 0,
        "providerTokenCostUnits": 172917
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "97c1a1181dc70242aec27de56e0a0a5ca95ce277b9b81fc6f1d6d9e853d75659",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T19:32:58.989Z",
      "runId": "phase-9-ab-adjudicated-cost-01",
      "planHash": "7d03ca649d4d4d23cc9c7cb7237aba6e51697d04d2d8ac1912101ccb439b08b8",
      "task": {
        "taskId": "consumer-04-documentation",
        "repositoryId": "consumer-04",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 97471,
        "responseBytes": 795,
        "stderrBytes": 0,
        "stdoutHash": "47b52331c78f0205529b41b1cd7012574245211b68b48f59439dc167ef3e1cdc",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 631872,
        "outputTokens": 4281,
        "tokenMethod": "provider",
        "toolCalls": 29,
        "errorCode": "token-budget"
      },
      "contextBytes": 2048,
      "evidenceIds": [
        "documentation-evidence:content/docs/scripts/index.md:49 claims 12 gate reference implementations; source find content/docs/scripts/*.mjs lists 13, including check-tokens.example.mjs",
        "documentation-evidence:README.md:36 and content/docs/discovery.mdx:33 independently claim 13 scripts; scripts/compute-stats.mjs:37 derives the count from the same source directory",
        "evidence-e0949ec7dd7301cbb2840148bfaf6bd3"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 537856,
        "reasoningOutputTokens": 1123,
        "stderrBytes": 0,
        "providerTokenCostUnits": 636153
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "blocked",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "f5e0b53e320eaa3949a3b134856f4a0ca3c7b4556ffddd484df0b5c6310b8349",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T19:33:57.495Z",
      "runId": "phase-9-ab-adjudicated-cost-01",
      "planHash": "7d03ca649d4d4d23cc9c7cb7237aba6e51697d04d2d8ac1912101ccb439b08b8",
      "task": {
        "taskId": "consumer-01-implementation",
        "repositoryId": "consumer-01",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 58474,
        "responseBytes": 992,
        "stderrBytes": 0,
        "stdoutHash": "c95ef29a32c90cb408ab8ee4ca67b9562041ac2101bee961a0853c1656e5c70a",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 318136,
        "outputTokens": 2196,
        "tokenMethod": "provider",
        "toolCalls": 8
      },
      "contextBytes": 2085,
      "evidenceIds": [
        "patch-evidence:docs/spec/config-v1.md:836; stale planned qualifier conflicts with implemented ak-docs chat in src/cli/program.ts:1598-1613 and src/intelligence/chat.ts:75-95",
        "proposal:documentation-only;replace \"| `ak-docs chat` | planned; `intelligence.*` |\" with \"| `ak-docs chat` | `intelligence.*` |\";no-source-changes",
        "convention:preserve-existing-CLI-mapping-table-format",
        "verification-plan:after-approval-run-node-bin/ak-docs.js-check---json-and-review-the-one-line-diff",
        "acceptance-check:ak-docs-check---json-blocked-by-read-only-filesystem-EPERM-on-.doc-bridge/workflow/.lock"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 260096,
        "reasoningOutputTokens": 939,
        "stderrBytes": 0,
        "providerTokenCostUnits": 320332
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "1c4a26eee81361e46ad959cfdd412dc9d0763a41e98bb420341e5f4a7748fcd9",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T19:34:32.955Z",
      "runId": "phase-9-ab-adjudicated-cost-01",
      "planHash": "7d03ca649d4d4d23cc9c7cb7237aba6e51697d04d2d8ac1912101ccb439b08b8",
      "task": {
        "taskId": "consumer-03-discovery",
        "repositoryId": "consumer-03",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 35425,
        "responseBytes": 388,
        "stderrBytes": 0,
        "stdoutHash": "7d13d8c4736761f947184b07a3c47e68a57b542c670aab2701e1f4d4c48aaae1",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 95511,
        "outputTokens": 599,
        "tokenMethod": "provider",
        "toolCalls": 3
      },
      "contextBytes": 1864,
      "evidenceIds": [
        "entrypoint-evidence"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 64768,
        "reasoningOutputTokens": 213,
        "stderrBytes": 0,
        "providerTokenCostUnits": 96110
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "0907c780f9f0000da93180e649e9e0f401c386f80f8dc5a3471e8789b796e505",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T19:34:56.398Z",
      "runId": "phase-9-ab-adjudicated-cost-01",
      "planHash": "7d03ca649d4d4d23cc9c7cb7237aba6e51697d04d2d8ac1912101ccb439b08b8",
      "task": {
        "taskId": "consumer-06-discovery",
        "repositoryId": "consumer-06",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 23414,
        "responseBytes": 364,
        "stderrBytes": 0,
        "stdoutHash": "5f460416dc1a4c4c3c64b9f555d0142ffdc0cdd3f7c6919cab13fb7b8ab21015",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 52539,
        "outputTokens": 295,
        "tokenMethod": "provider",
        "toolCalls": 1
      },
      "contextBytes": 1864,
      "evidenceIds": [],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "low",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 25344,
        "reasoningOutputTokens": 143,
        "stderrBytes": 0,
        "providerTokenCostUnits": 52834
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "incomplete",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "317b116c9762e4217a110f6bb263a155df0d1cd996545de2330164b76a431fbf",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T19:36:31.954Z",
      "runId": "phase-9-ab-adjudicated-cost-01",
      "planHash": "7d03ca649d4d4d23cc9c7cb7237aba6e51697d04d2d8ac1912101ccb439b08b8",
      "task": {
        "taskId": "consumer-05-architecture",
        "repositoryId": "consumer-05",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 95521,
        "responseBytes": 951,
        "stderrBytes": 0,
        "stdoutHash": "54d3f26699dcd91052573c538290ba68c41a6c06c1c7bcae6a6926b36cc8aacf",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 706488,
        "outputTokens": 3751,
        "tokenMethod": "provider",
        "toolCalls": 14,
        "errorCode": "token-budget"
      },
      "contextBytes": 1884,
      "evidenceIds": [
        "docs/architecture.md: registry/* -> scripts/build-registry.mjs -> public/r bundles and deterministic artifact -> Fumadocs site/AgentsChat",
        "docs/architecture.md: boundary is Registry-owned source/artifacts versus AgentsKit Chat protocol and external Fumadocs application",
        "package.json: pinned @agentskit/chat 0.4.0 and @agentskit/doc-bridge 1.7.45 dependencies",
        "doc-bridge-normalize-artifact: observed package dependency relations and sourceRevision 1bd12daea4cfa59cd2e319b45d661f7e301d4733dfe4ba42b967b54ebc491fea",
        "ak-docs map --json: blocked by EPERM creating .doc-bridge/workflow/.lock"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 636416,
        "reasoningOutputTokens": 1069,
        "stderrBytes": 0,
        "providerTokenCostUnits": 710239
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "blocked",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "1388bff1cc208e4eb9c2dc20a07dc850e941f883f95289c0b28fb0491113dfa3",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T19:37:12.175Z",
      "runId": "phase-9-ab-adjudicated-cost-01",
      "planHash": "7d03ca649d4d4d23cc9c7cb7237aba6e51697d04d2d8ac1912101ccb439b08b8",
      "task": {
        "taskId": "consumer-06-discovery",
        "repositoryId": "consumer-06",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 40188,
        "responseBytes": 839,
        "stderrBytes": 0,
        "stdoutHash": "3374ed0c3d742a37fa3ff97bd51636f33689de0b27f36224e5e81d336b6d1224",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 226812,
        "outputTokens": 1668,
        "tokenMethod": "provider",
        "toolCalls": 5
      },
      "contextBytes": 1856,
      "evidenceIds": [
        "entrypoint-evidence",
        "README.md",
        "AGENTS.md",
        "docs/for-agents/INDEX.md",
        "docs/internal/README.md",
        "package.json",
        "apps/desktop/src/main.tsx",
        "apps/console/src/main.tsx",
        "apps/web/app/layout.tsx",
        "apps/cloud/src/server.ts",
        "apps/admin/app/layout.tsx",
        "apps/license-service/src/index.ts",
        "packages/os-cli/src/index.ts",
        "packages/os-core/src/index.ts",
        "packages/os-headless/src/index.ts",
        "packages/os-desktop/src/index.ts",
        "packages/desktop-shell/src/index.ts",
        "discovery-check-failed"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 184832,
        "reasoningOutputTokens": 670,
        "stderrBytes": 0,
        "providerTokenCostUnits": 228480
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "64f67cf3f1a7cbf0382859e39a781686bc212845b1256ef679bee43d61a7e9a0",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T19:38:28.381Z",
      "runId": "phase-9-ab-adjudicated-cost-01",
      "planHash": "7d03ca649d4d4d23cc9c7cb7237aba6e51697d04d2d8ac1912101ccb439b08b8",
      "task": {
        "taskId": "consumer-06-implementation",
        "repositoryId": "consumer-06",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 76170,
        "responseBytes": 611,
        "stderrBytes": 0,
        "stdoutHash": "1b2180c2ba33577d8b7dbe5513b43312b99d29e4f321aadf5b457cdda23a965d",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 251140,
        "outputTokens": 1550,
        "tokenMethod": "provider",
        "toolCalls": 8
      },
      "contextBytes": 2084,
      "evidenceIds": [
        "patch-evidence:.changeset/5538-live-provider-certification.md:6",
        "patch-target:docs/testing/provider-live-certification.md:1",
        "supporting-evidence:packages/os-integrations/src/integration-product-metadata.ts:399",
        "verification-plan:ak-docs check --json"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "success",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksTotal": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 199168,
        "reasoningOutputTokens": 605,
        "stderrBytes": 0,
        "providerTokenCostUnits": 252690
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "blocked",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "5ee130f4bf7e495815ad86844c98992a16b417b20555c77491df5fa85b1d3b41",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T19:39:37.209Z",
      "runId": "phase-9-ab-adjudicated-cost-01",
      "planHash": "7d03ca649d4d4d23cc9c7cb7237aba6e51697d04d2d8ac1912101ccb439b08b8",
      "task": {
        "taskId": "consumer-05-documentation",
        "repositoryId": "consumer-05",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 68795,
        "responseBytes": 790,
        "stderrBytes": 0,
        "stdoutHash": "ace8981ff7160ff4628368e78913971a507b848d8255868b47583524d8705428",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 161688,
        "outputTokens": 1310,
        "tokenMethod": "provider",
        "toolCalls": 5
      },
      "contextBytes": 2056,
      "evidenceIds": [
        "evidence-5e0cf2425ebca4762584abc683529cae",
        "review-limitation:.codex/verification/runs/1788120908034-26987/run.json states natural-language semantic judgments are not deterministically analyzed and require agent or human review",
        "acceptance-limitation:ak-docs was unavailable on PATH; configured local audit could not acquire .doc-bridge/workflow/.lock because the filesystem is read-only"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 1,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 121600,
        "reasoningOutputTokens": 433,
        "stderrBytes": 0,
        "providerTokenCostUnits": 162998
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "37fa7a554788bc4f56060364129556e66017f33ffee400e70e05d02ca4ccf269",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T19:40:32.960Z",
      "runId": "phase-9-ab-adjudicated-cost-01",
      "planHash": "7d03ca649d4d4d23cc9c7cb7237aba6e51697d04d2d8ac1912101ccb439b08b8",
      "task": {
        "taskId": "consumer-02-architecture",
        "repositoryId": "consumer-02",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 55719,
        "responseBytes": 601,
        "stderrBytes": 0,
        "stdoutHash": "4ce03617f19610c26e83638106637a16c547edb3834ffd45d7cf4cc09ce4870f",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 174413,
        "outputTokens": 943,
        "tokenMethod": "provider",
        "toolCalls": 6
      },
      "contextBytes": 1883,
      "evidenceIds": [
        "architecture-evidence:package.json",
        "architecture-evidence:pnpm-workspace.yaml",
        "architecture-evidence:doc-bridge.config.json",
        "architecture-evidence:.doc-bridge/index.json",
        "architecture-check:blocked-by-read-only-filesystem-lock"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 1,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 111360,
        "reasoningOutputTokens": 256,
        "stderrBytes": 0,
        "providerTokenCostUnits": 175356
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "997db7058d730e14fc8a6f8b3ed797d3a4149e7d71430e08b7094e0204367b2a",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T19:41:50.710Z",
      "runId": "phase-9-ab-adjudicated-cost-01",
      "planHash": "7d03ca649d4d4d23cc9c7cb7237aba6e51697d04d2d8ac1912101ccb439b08b8",
      "task": {
        "taskId": "consumer-05-documentation",
        "repositoryId": "consumer-05",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 77714,
        "responseBytes": 529,
        "stderrBytes": 0,
        "stdoutHash": "5b9cfd6922e9b67cda5837d7a84ab48237a622f0e9750a131ccfd3da8731c63d",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 462275,
        "outputTokens": 2552,
        "tokenMethod": "provider",
        "toolCalls": 11,
        "errorCode": "token-budget"
      },
      "contextBytes": 2057,
      "evidenceIds": [
        "documentation-evidence:docs/architecture.md:27-40|package.json:36-40",
        "review-limitation:semantic-version-claim-requires-human-confirmation"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 408576,
        "reasoningOutputTokens": 1040,
        "stderrBytes": 0,
        "providerTokenCostUnits": 464827
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "blocked",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "33ae5717b4d01ed95418c7081ba1b11d969bb52a25b776476da17b673f8c4424",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T19:43:18.491Z",
      "runId": "phase-9-ab-adjudicated-cost-01",
      "planHash": "7d03ca649d4d4d23cc9c7cb7237aba6e51697d04d2d8ac1912101ccb439b08b8",
      "task": {
        "taskId": "consumer-06-architecture",
        "repositoryId": "consumer-06",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 87746,
        "responseBytes": 685,
        "stderrBytes": 0,
        "stdoutHash": "5c0b149aef575fb8009934cf103058986590deb98ab1972f08cc4dd035d25154",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 467992,
        "outputTokens": 1979,
        "tokenMethod": "provider",
        "toolCalls": 10,
        "errorCode": "token-budget"
      },
      "contextBytes": 1893,
      "evidenceIds": [
        "architecture-evidence",
        "docs/for-agents/INDEX.md",
        "docs/for-agents/packages/os-core.md",
        "docs/for-agents/packages/os-runtime.md",
        "docs/for-agents/packages/os-headless.md",
        "docs/for-agents/packages/desktop-sidecar-bridge.md",
        "packages/os-core/package.json",
        "packages/os-contracts/package.json",
        "packages/os-desktop/package.json"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 414976,
        "reasoningOutputTokens": 712,
        "stderrBytes": 0,
        "providerTokenCostUnits": 469971
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "blocked",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "ea3d0a24621f83a7ba0ef5486e9ec8794c8d55a56a05690fdf413b28e00213fd",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T19:44:53.971Z",
      "runId": "phase-9-ab-adjudicated-cost-01",
      "planHash": "7d03ca649d4d4d23cc9c7cb7237aba6e51697d04d2d8ac1912101ccb439b08b8",
      "task": {
        "taskId": "consumer-01-implementation",
        "repositoryId": "consumer-01",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 95443,
        "responseBytes": 654,
        "stderrBytes": 0,
        "stdoutHash": "958cb6c5f33ba98d00c8801e05481de5abdbb80ccd89b19b06ac2491e0c92c8d",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 451916,
        "outputTokens": 1827,
        "tokenMethod": "provider",
        "toolCalls": 11,
        "errorCode": "token-budget"
      },
      "contextBytes": 2084,
      "evidenceIds": [
        "patch-evidence-blocked:docs/study/observation-ledger-v1.json records that no verified knowledge gap or safe target document was established for this task",
        "verification-plan:ak-docs check --json; execution blocked because the read-only repository denied creation of the workflow lock"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 391424,
        "reasoningOutputTokens": 491,
        "stderrBytes": 0,
        "providerTokenCostUnits": 453743
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "blocked",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "219d99e92b5ee87c3ec91ab409f2d6190c89767da0d9b4eedf1a1d62db6406b9",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T19:45:48.535Z",
      "runId": "phase-9-ab-adjudicated-cost-01",
      "planHash": "7d03ca649d4d4d23cc9c7cb7237aba6e51697d04d2d8ac1912101ccb439b08b8",
      "task": {
        "taskId": "consumer-01-architecture",
        "repositoryId": "consumer-01",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 54532,
        "responseBytes": 605,
        "stderrBytes": 0,
        "stdoutHash": "f2496765eb65e10778bf34eaeb4cf033b59dcca62cc4129790d1dad1a3129a67",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 346427,
        "outputTokens": 1910,
        "tokenMethod": "provider",
        "toolCalls": 8
      },
      "contextBytes": 1893,
      "evidenceIds": [
        "architecture-evidence:discovery-snapshot:50ba14dc3c6280ae68bff5f22612d7aa32ce8c3fb32db3791f23817b619fa583",
        "doc-bridge.config.json:ownership-map",
        "src/agents/registry-adapter.ts:140",
        "src/intelligence/peers.ts:50",
        "coverage:dynamic-imports:partial"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 280576,
        "reasoningOutputTokens": 713,
        "stderrBytes": 0,
        "providerTokenCostUnits": 348337
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "a2702dd22b378420d881f8306adea78c8e9006e6ce9b60249ae12d9028c6f87f",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T19:47:46.109Z",
      "runId": "phase-9-ab-adjudicated-cost-01",
      "planHash": "7d03ca649d4d4d23cc9c7cb7237aba6e51697d04d2d8ac1912101ccb439b08b8",
      "task": {
        "taskId": "consumer-06-discovery",
        "repositoryId": "consumer-06",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 117541,
        "responseBytes": 1317,
        "stderrBytes": 0,
        "stdoutHash": "191fbabc97ce38fa4205c3bd146cbc12bb8949d07fafb19f6856300dff292641",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 704396,
        "outputTokens": 3205,
        "tokenMethod": "provider",
        "toolCalls": 13,
        "errorCode": "token-budget"
      },
      "contextBytes": 1865,
      "evidenceIds": [
        "apps/console/src/main.tsx | owner: apps/console | docs: docs/for-agents/apps/console.md",
        "apps/desktop | owner: apps/desktop plus desktop-* packages | docs: docs/for-agents/apps/desktop.md and docs/for-agents/INDEX.md",
        "apps/cloud/src/server.ts and apps/cloud/bin/server.mjs | owner: apps/cloud | docs: docs/for-agents/apps/cloud.md",
        "apps/admin | owner: apps/admin | docs: docs/for-agents/apps/admin.md",
        "apps/license-service/bin/server.mjs | owner: apps/license-service | docs: docs/for-agents/apps/license-service.md",
        "apps/web | owner: apps/web | docs: docs/for-agents/apps/web.md",
        "packages/os-cli/src and packages/os-cli/bin | owner: os-cli | docs: docs/for-agents/packages/os-cli.md",
        "packages/os-headless/bin/sidecar.ts | owner: os-headless | docs: docs/for-agents/packages/os-headless.md",
        "Canonical routing: AGENTS.md and docs/for-agents/INDEX.md",
        "Structured discovery succeeded via direct ak-docs CLI; pnpm wrapper was blocked by read-only EPERM"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 622336,
        "reasoningOutputTokens": 1320,
        "stderrBytes": 0,
        "providerTokenCostUnits": 707601
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "blocked",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "233d9da948a72d306ea09d2e9f413f12eea68fabde274a014c0f16a6b32495da",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T19:48:28.587Z",
      "runId": "phase-9-ab-adjudicated-cost-01",
      "planHash": "7d03ca649d4d4d23cc9c7cb7237aba6e51697d04d2d8ac1912101ccb439b08b8",
      "task": {
        "taskId": "consumer-01-discovery",
        "repositoryId": "consumer-01",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 42448,
        "responseBytes": 535,
        "stderrBytes": 0,
        "stdoutHash": "78bb0c12d0f6a88188d66930374606a499449a23d4488a346efafac21f29f36a",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 99125,
        "outputTokens": 764,
        "tokenMethod": "provider",
        "toolCalls": 3
      },
      "contextBytes": 1864,
      "evidenceIds": [
        "package.json",
        "bin/ak-docs.js",
        "src/index.ts",
        "src/cli/program.ts",
        "doc-bridge.config.json",
        "README.md",
        "docs/index.md",
        "docs/agent-corpus/OVERVIEW.md",
        "GOVERNANCE.md"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 81920,
        "reasoningOutputTokens": 243,
        "stderrBytes": 0,
        "providerTokenCostUnits": 99889
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "b7b0b7f53684d1a29608459d9649993eb589d0ce75641eae41a8814b5b5e3539",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T19:49:12.717Z",
      "runId": "phase-9-ab-adjudicated-cost-01",
      "planHash": "7d03ca649d4d4d23cc9c7cb7237aba6e51697d04d2d8ac1912101ccb439b08b8",
      "task": {
        "taskId": "consumer-01-architecture",
        "repositoryId": "consumer-01",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 44094,
        "responseBytes": 1218,
        "stderrBytes": 0,
        "stdoutHash": "21183d27bfc1833d9899c1004d1b129645e58c172a076e5c5ff0319833edadc4",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 184817,
        "outputTokens": 1684,
        "tokenMethod": "provider",
        "toolCalls": 5
      },
      "contextBytes": 1884,
      "evidenceIds": [
        "architecture:src/discovery/repository.ts=>observes packages/modules/documents and emits evidence-backed contains/depends-on/imports/re-exports relations",
        "architecture:src/index-builder/build-index.ts=>combines agent corpus, discovered packages, human docs, handoffs, lookup, and generated index",
        "architecture:src/query/query.ts=>consumes the generated index for package/ownership/intent/change/search handoffs",
        "architecture:src/mcp/server.ts=>exposes read-only document, diagnostics, relations, and workflow-state interfaces",
        "attention-point:discovery-to-architecture-boundary=>dynamic imports, unresolved runtime wiring, and generated code are explicitly partial or not-analyzed in discovery coverage",
        "blocked:ak-docs-map=>ak-docs unavailable on PATH; node bin/ak-docs.js map --json could not create .doc-bridge/workflow/.lock under read-only permissions"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 153088,
        "reasoningOutputTokens": 714,
        "stderrBytes": 0,
        "providerTokenCostUnits": 186501
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "7e1f2fb1ba49fc7016c23ad7411caf0b9fc1cf137a8d5115c86d9d2731c2ac7c",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T19:49:47.863Z",
      "runId": "phase-9-ab-adjudicated-cost-01",
      "planHash": "7d03ca649d4d4d23cc9c7cb7237aba6e51697d04d2d8ac1912101ccb439b08b8",
      "task": {
        "taskId": "consumer-04-architecture",
        "repositoryId": "consumer-04",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 35114,
        "responseBytes": 379,
        "stderrBytes": 0,
        "stdoutHash": "4442f9c8ea3ce0d618da252c809dfb26ee79dd6c7483d9e791659e65fe81d0ee",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 162097,
        "outputTokens": 1274,
        "tokenMethod": "provider",
        "toolCalls": 5
      },
      "contextBytes": 1893,
      "evidenceIds": [
        "architecture-evidence"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 1,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 122368,
        "reasoningOutputTokens": 627,
        "stderrBytes": 0,
        "providerTokenCostUnits": 163371
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "8a9ce17ba3c1ad2900840c4fb0caf1b4b8b9e2f007518389db75adbcc4d03bb3",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T19:50:24.153Z",
      "runId": "phase-9-ab-adjudicated-cost-01",
      "planHash": "7d03ca649d4d4d23cc9c7cb7237aba6e51697d04d2d8ac1912101ccb439b08b8",
      "task": {
        "taskId": "consumer-01-discovery",
        "repositoryId": "consumer-01",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 36259,
        "responseBytes": 448,
        "stderrBytes": 0,
        "stdoutHash": "5c3fd03c12a356e050c4749bbe5ea3dea8c11c35684a4b7ecc752e5b998adb9e",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 91181,
        "outputTokens": 661,
        "tokenMethod": "provider",
        "toolCalls": 3
      },
      "contextBytes": 1855,
      "evidenceIds": [
        "package.json",
        "bin/ak-docs.js",
        "README.md",
        "docs/index.md",
        "apps/docs/AGENTS.md"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 61696,
        "reasoningOutputTokens": 231,
        "stderrBytes": 0,
        "providerTokenCostUnits": 91842
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "67d3cc19f16af3e2abd8a161851d4f44a82d294337278958875047aca0d5ae60",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T19:52:03.105Z",
      "runId": "phase-9-ab-adjudicated-cost-01",
      "planHash": "7d03ca649d4d4d23cc9c7cb7237aba6e51697d04d2d8ac1912101ccb439b08b8",
      "task": {
        "taskId": "consumer-01-architecture",
        "repositoryId": "consumer-01",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 98923,
        "responseBytes": 684,
        "stderrBytes": 0,
        "stdoutHash": "3bd85e3293fb55aba900a08e7aacac13590f8349b5178b29652b2d68da9df9c4",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 353557,
        "outputTokens": 2002,
        "tokenMethod": "provider",
        "toolCalls": 9
      },
      "contextBytes": 1892,
      "evidenceIds": [
        "architecture-evidence:.doc-bridge/workflow/artifacts/collect-068e5c916af22dff7ff824884ded6cf9d58388ed8479f92c9a69e489d770b76d.json",
        "workspace-evidence:package.json",
        "documentation-claim:README.md",
        "boundary-review:src/agents/registry-adapter.ts:140"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "architectureEntityCount": 369,
        "architectureRelationCount": 1147,
        "cachedInputTokens": 300800,
        "reasoningOutputTokens": 530,
        "stderrBytes": 0,
        "providerTokenCostUnits": 355559
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "f47e08ff2554e3e97ec279b9b8071bca3a93bbf1c15c88da829c83cd1742a2a7",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T19:52:56.165Z",
      "runId": "phase-9-ab-adjudicated-cost-01",
      "planHash": "7d03ca649d4d4d23cc9c7cb7237aba6e51697d04d2d8ac1912101ccb439b08b8",
      "task": {
        "taskId": "consumer-02-architecture",
        "repositoryId": "consumer-02",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 53028,
        "responseBytes": 408,
        "stderrBytes": 0,
        "stdoutHash": "e274b68916675677bba55ffd378b6bf1b844bb42271c712cf665f87145d3db97",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 318534,
        "outputTokens": 1808,
        "tokenMethod": "provider",
        "toolCalls": 9
      },
      "contextBytes": 1893,
      "evidenceIds": [
        "architecture-evidence",
        "architecture-check-blocked"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 267008,
        "reasoningOutputTokens": 554,
        "stderrBytes": 0,
        "providerTokenCostUnits": 320342
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "6c831098c8eb01a9f45496c3ebe022e24aaa5e205b2ff5ba62b5efdf522f1530",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T19:53:19.408Z",
      "runId": "phase-9-ab-adjudicated-cost-01",
      "planHash": "7d03ca649d4d4d23cc9c7cb7237aba6e51697d04d2d8ac1912101ccb439b08b8",
      "task": {
        "taskId": "consumer-06-implementation",
        "repositoryId": "consumer-06",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 23212,
        "responseBytes": 350,
        "stderrBytes": 0,
        "stdoutHash": "42c0dd88856f43800d5fe564cf00ef540f952e07c3f4816f34292017cb27798e",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 53008,
        "outputTokens": 463,
        "tokenMethod": "provider",
        "toolCalls": 1
      },
      "contextBytes": 2093,
      "evidenceIds": [],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "low",
      "safetyOutcome": "safe",
      "clarificationRequests": 1,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 25344,
        "reasoningOutputTokens": 264,
        "stderrBytes": 0,
        "providerTokenCostUnits": 53471
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "incomplete",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "1e59eae3a9ce9f09ec47e1f3387e2fb331216c62e2a7918cad99fc25a2c5146f",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T19:53:56.365Z",
      "runId": "phase-9-ab-adjudicated-cost-01",
      "planHash": "7d03ca649d4d4d23cc9c7cb7237aba6e51697d04d2d8ac1912101ccb439b08b8",
      "task": {
        "taskId": "consumer-05-implementation",
        "repositoryId": "consumer-05",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 36922,
        "responseBytes": 705,
        "stderrBytes": 0,
        "stdoutHash": "e5646ee8a2f524cf9cdebbc7fefd8e1d169fcab69f27f15811e35f8ea651c97b",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 86494,
        "outputTokens": 753,
        "tokenMethod": "provider",
        "toolCalls": 2
      },
      "contextBytes": 2093,
      "evidenceIds": [
        "patch-evidence:docs/for-agents/index.md:architecture-and-ownership-link-bypasses-dedicated-registry-architecture-handoff",
        "proposal:replace-docs/architecture.md-link-with-registry-architecture.md-only",
        "verification-plan:ak-docs-check-json-exit-0",
        "verification-result:blocked-by-read-only-filesystem-EPERM"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 51712,
        "reasoningOutputTokens": 296,
        "stderrBytes": 0,
        "providerTokenCostUnits": 87247
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "29533d64c4d8b507724efd566f7451a67b5397da8a21681bcdfff81641b6a24c",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T19:54:25.992Z",
      "runId": "phase-9-ab-adjudicated-cost-01",
      "planHash": "7d03ca649d4d4d23cc9c7cb7237aba6e51697d04d2d8ac1912101ccb439b08b8",
      "task": {
        "taskId": "consumer-05-architecture",
        "repositoryId": "consumer-05",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 29597,
        "responseBytes": 613,
        "stderrBytes": 0,
        "stdoutHash": "60f6a41f063be464771730642279b3fe5b5f3b46102b583a61de0c91105dc5d5",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 86145,
        "outputTokens": 573,
        "tokenMethod": "provider",
        "toolCalls": 3
      },
      "contextBytes": 1892,
      "evidenceIds": [
        "architecture-evidence:package.json",
        "architecture-evidence:pnpm-workspace.yaml",
        "architecture-evidence:scripts/build-registry.mjs",
        "architecture-evidence:scripts/lib/deterministic-discovery.mjs",
        "architecture-check:ak-docs-map-command-not-found"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 61696,
        "reasoningOutputTokens": 162,
        "stderrBytes": 0,
        "providerTokenCostUnits": 86718
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "cdfbf3fe135108de87b4c35294286357b5138e7cb448969abfcc2cbee4ba051e",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T19:55:18.724Z",
      "runId": "phase-9-ab-adjudicated-cost-01",
      "planHash": "7d03ca649d4d4d23cc9c7cb7237aba6e51697d04d2d8ac1912101ccb439b08b8",
      "task": {
        "taskId": "consumer-05-architecture",
        "repositoryId": "consumer-05",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 52698,
        "responseBytes": 733,
        "stderrBytes": 0,
        "stdoutHash": "76223c14c185b068051716582eabbb2c704aa74e9b951466c795e9fa9f0f2ec9",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 178353,
        "outputTokens": 1012,
        "tokenMethod": "provider",
        "toolCalls": 6
      },
      "contextBytes": 1883,
      "evidenceIds": [
        "architecture-evidence:docs/architecture.md",
        "architecture-evidence:docs/for-agents/index.md",
        "architecture-evidence:package.json",
        "architecture-evidence:scripts/build-registry.mjs",
        "architecture-evidence:scripts/lib/deterministic-discovery.mjs",
        "architecture-attention-point:@agentskit/chat/protocol-boundary",
        "architecture-check:blocked-by-read-only-filesystem"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 112384,
        "reasoningOutputTokens": 242,
        "stderrBytes": 0,
        "providerTokenCostUnits": 179365
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "35fc51bef02220d5602c51d225dc461374323934a2de47f317d4189504eee241",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T19:56:11.886Z",
      "runId": "phase-9-ab-adjudicated-cost-01",
      "planHash": "7d03ca649d4d4d23cc9c7cb7237aba6e51697d04d2d8ac1912101ccb439b08b8",
      "task": {
        "taskId": "consumer-02-discovery",
        "repositoryId": "consumer-02",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 53130,
        "responseBytes": 575,
        "stderrBytes": 0,
        "stdoutHash": "85d49281fda42bf19141c0a6f8db09ed5934ba06b6be0fc13197adecd767ab3d",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 164140,
        "outputTokens": 1016,
        "tokenMethod": "provider",
        "toolCalls": 5
      },
      "contextBytes": 1855,
      "evidenceIds": [
        "discovery-snapshot:3b5a3981abca11e57dc9790d0d1f949f5647d59bcf323781079c24d274709dce",
        "artifact:package.json",
        "artifact:packages/*/package.json",
        "artifact:.github/CODEOWNERS",
        "artifact:AGENTS.md",
        "artifact:apps/docs-next"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "success",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 1,
      "measurements": {
        "acceptanceChecksPassed": 1,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 121600,
        "reasoningOutputTokens": 365,
        "stderrBytes": 0,
        "providerTokenCostUnits": 165156
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "16e6ab718efebc6fcbd1a7848bf9fb6a4f07490d6c2b10add07712a62ea3c3d8",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T19:56:54.402Z",
      "runId": "phase-9-ab-adjudicated-cost-01",
      "planHash": "7d03ca649d4d4d23cc9c7cb7237aba6e51697d04d2d8ac1912101ccb439b08b8",
      "task": {
        "taskId": "consumer-02-architecture",
        "repositoryId": "consumer-02",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 42462,
        "responseBytes": 365,
        "stderrBytes": 0,
        "stdoutHash": "2d5447b77b03ebc9cf8557e9e53db78884d3bf22477e7ecae86acc9d2f487458",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 113064,
        "outputTokens": 570,
        "tokenMethod": "provider",
        "toolCalls": 4
      },
      "contextBytes": 1892,
      "evidenceIds": [],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "low",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 1,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 87040,
        "reasoningOutputTokens": 150,
        "stderrBytes": 0,
        "providerTokenCostUnits": 113634
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "incomplete",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "9bf05a049961054ba762c2d76f2ccaad1d473e3699784109a04198a4907a2ef3",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T19:58:08.143Z",
      "runId": "phase-9-ab-adjudicated-cost-01",
      "planHash": "7d03ca649d4d4d23cc9c7cb7237aba6e51697d04d2d8ac1912101ccb439b08b8",
      "task": {
        "taskId": "consumer-03-implementation",
        "repositoryId": "consumer-03",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 73709,
        "responseBytes": 574,
        "stderrBytes": 0,
        "stdoutHash": "5e537103f3baee69d5e6ece6b9340356f63c4f12756bd513b2858607da908fe8",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 253468,
        "outputTokens": 1416,
        "tokenMethod": "provider",
        "toolCalls": 7
      },
      "contextBytes": 2093,
      "evidenceIds": [
        "evidence-8b787c88523bd2c0b07d03a5eed32ef3",
        "verification-plan:After approval, review the documentation-only diff and run ak-docs check --json from the repository root; require exit code 0."
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "success",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 212992,
        "reasoningOutputTokens": 385,
        "stderrBytes": 0,
        "providerTokenCostUnits": 254884
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "35ffb1c463a89693bb71f342770610ef8ab9434a5bcf7f750f98ecfc512d8b3d",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T19:59:26.110Z",
      "runId": "phase-9-ab-adjudicated-cost-01",
      "planHash": "7d03ca649d4d4d23cc9c7cb7237aba6e51697d04d2d8ac1912101ccb439b08b8",
      "task": {
        "taskId": "consumer-04-implementation",
        "repositoryId": "consumer-04",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 77930,
        "responseBytes": 555,
        "stderrBytes": 0,
        "stdoutHash": "22d02b15a7db5a8d77b28d52d20ef084afc236f546df4522a9ab7f6c97b77b93",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 386055,
        "outputTokens": 2584,
        "tokenMethod": "provider",
        "toolCalls": 11
      },
      "contextBytes": 2085,
      "evidenceIds": [
        "doc-bridge-handoff",
        "documentation-audit"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 1,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "documentationExampleRate": 0.5697674418604651,
        "documentationFreshnessRate": 1,
        "documentationCompletenessRate": 1,
        "cachedInputTokens": 334080,
        "reasoningOutputTokens": 1237,
        "stderrBytes": 0,
        "providerTokenCostUnits": 388639
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "60aa14448b61082fd180b572fc1db8ead10356d7c19e93bac61e2151631999d1",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:00:21.525Z",
      "runId": "phase-9-ab-adjudicated-cost-01",
      "planHash": "7d03ca649d4d4d23cc9c7cb7237aba6e51697d04d2d8ac1912101ccb439b08b8",
      "task": {
        "taskId": "consumer-02-architecture",
        "repositoryId": "consumer-02",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 55382,
        "responseBytes": 472,
        "stderrBytes": 0,
        "stdoutHash": "36a32c9b8a88f9c06c74d60d3c1f1a111879531b8a32d6808a8c943606cd648a",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 347770,
        "outputTokens": 2062,
        "tokenMethod": "provider",
        "toolCalls": 7
      },
      "contextBytes": 1884,
      "evidenceIds": [
        "architecture-evidence",
        "workspace-metadata",
        "core-imports",
        "composition-rules",
        "acceptance-check-blocked"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "not-applicable",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 287488,
        "reasoningOutputTokens": 625,
        "stderrBytes": 0,
        "providerTokenCostUnits": 349832
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "023f1d7d82c858903620a463aadfb5f67eb0c5231f70ae646de5c267945ab3f3",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:00:54.507Z",
      "runId": "phase-9-ab-adjudicated-cost-01",
      "planHash": "7d03ca649d4d4d23cc9c7cb7237aba6e51697d04d2d8ac1912101ccb439b08b8",
      "task": {
        "taskId": "consumer-03-discovery",
        "repositoryId": "consumer-03",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 32944,
        "responseBytes": 388,
        "stderrBytes": 0,
        "stdoutHash": "2305aabc06adaee8c83c352ecd0f3e6783034539bdfc1524d6bcebda98012a84",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 92947,
        "outputTokens": 603,
        "tokenMethod": "provider",
        "toolCalls": 3
      },
      "contextBytes": 1855,
      "evidenceIds": [
        "entrypoint-evidence"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 62720,
        "reasoningOutputTokens": 200,
        "stderrBytes": 0,
        "providerTokenCostUnits": 93550
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "41b6ba48406da01e5247756083e12cc88440515dc7a0d040511ba18b4823361e",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:02:34.142Z",
      "runId": "phase-9-ab-adjudicated-cost-01",
      "planHash": "7d03ca649d4d4d23cc9c7cb7237aba6e51697d04d2d8ac1912101ccb439b08b8",
      "task": {
        "taskId": "consumer-02-documentation",
        "repositoryId": "consumer-02",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 99598,
        "responseBytes": 1011,
        "stderrBytes": 0,
        "stdoutHash": "196cd214e5a28b8147d484d90701f27020cfef316c77c39c9408a5e111ef6796",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 879576,
        "outputTokens": 3331,
        "tokenMethod": "provider",
        "toolCalls": 15,
        "errorCode": "token-budget"
      },
      "contextBytes": 2057,
      "evidenceIds": [
        "documentation-evidence:README.md:43 claims 140 providers, 5000+ models, 21 skills, and 69 recipes",
        "documentation-evidence:ecosystem-claims.json:61-117 records 184 providers, 6297 models, 22 skills, and 71 recipes",
        "documentation-evidence:scripts/compute-stats.mjs:87-114 defines the current derivation",
        "documentation-evidence:apps/docs-next/lib/ecosystem-stats.snapshot.json:5-13 matches the newer counts",
        "review-limitation:semantic classification is based on committed ledger, derivation script, and generated snapshot; ak-docs audit documentation --json was blocked by EPERM while creating .doc-bridge/workflow/.lock"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "documentationFindingCount": 1,
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 800512,
        "reasoningOutputTokens": 1186,
        "stderrBytes": 0,
        "providerTokenCostUnits": 882907
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "blocked",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "5e9c7934963a53ad1ad98a3c127b58f882ce7474516f6cb35b174322bd814831",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:04:15.273Z",
      "runId": "phase-9-ab-adjudicated-cost-01",
      "planHash": "7d03ca649d4d4d23cc9c7cb7237aba6e51697d04d2d8ac1912101ccb439b08b8",
      "task": {
        "taskId": "consumer-06-architecture",
        "repositoryId": "consumer-06",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 100978,
        "responseBytes": 486,
        "stderrBytes": 0,
        "stdoutHash": "5e8185a190c475ed2bc8b63703029f0f6b216f1717251dec9f1e79bd09f6f11d",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 195615,
        "outputTokens": 985,
        "tokenMethod": "provider",
        "toolCalls": 5
      },
      "contextBytes": 1892,
      "evidenceIds": [
        "package.json:327-330",
        "docs/rfc/0056-doc-bridge-ecosystem-dogfood.md:20",
        "ak-docs-map:EPERM-.doc-bridge/workflow/.lock"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "low",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 2,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 163328,
        "reasoningOutputTokens": 333,
        "stderrBytes": 0,
        "providerTokenCostUnits": 196600
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "6c357c21defa736e1fd2ef5f6572bbfa593439e7f718fbc79e92bf26e4412285",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:07:05.734Z",
      "runId": "phase-9-ab-adjudicated-cost-01",
      "planHash": "7d03ca649d4d4d23cc9c7cb7237aba6e51697d04d2d8ac1912101ccb439b08b8",
      "task": {
        "taskId": "consumer-02-implementation",
        "repositoryId": "consumer-02",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 170389,
        "responseBytes": 493,
        "stderrBytes": 0,
        "stdoutHash": "9d5e6238399015edfb5997f7ee46ce6174f8ed92199604d95ee1b3e0e259fb7d",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 522464,
        "outputTokens": 2852,
        "tokenMethod": "provider",
        "toolCalls": 13,
        "errorCode": "token-budget"
      },
      "contextBytes": 2084,
      "evidenceIds": [
        "evidence-9925541610b16c07b09a1118532a1638",
        "verification-plan:ak-docs check --json"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "success",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "documentationFindingCount": 1,
        "documentationExampleRate": 0.4296148738379814,
        "documentationFreshnessRate": 1,
        "cachedInputTokens": 479488,
        "reasoningOutputTokens": 952,
        "stderrBytes": 0,
        "providerTokenCostUnits": 525316
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "blocked",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "0a6e97e3e3771f33fa20a4f9e52495e617fcd3c7d27b618ec233ea1b3c5f4669",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:11:15.989Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-05-discovery",
        "repositoryId": "consumer-05",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 65878,
        "responseBytes": 461,
        "stderrBytes": 0,
        "stdoutHash": "34b05a1009b46b0d23a6ac65af45b18213ba2f2674b88fe2d5e295a4b7147cc7",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 236104,
        "outputTokens": 2191,
        "tokenMethod": "provider",
        "toolCalls": 6
      },
      "contextBytes": 1865,
      "evidenceIds": [
        "entrypoint-evidence",
        "package.json:scripts",
        "docs/for-agents/index.md",
        "docs/architecture.md",
        "AGENTS.md"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "success",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 1,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 20,
        "reasoningOutputTokens": 1129,
        "stderrBytes": 0,
        "providerTokenCostUnits": 238295,
        "adjudicatorLatencyMs": 52,
        "adjudicatorInputTokens": 120,
        "adjudicatorOutputTokens": 30,
        "adjudicatorTokenCostUnits": 150,
        "adjudicatorCostUsd": 0.00034
      },
      "adjudication": {
        "status": "automated",
        "actor": "independent-reviewer",
        "method": "independent-rubric-v1",
        "outcome": "partial",
        "confidence": 0.8,
        "reasonCodes": [
          "smoke-rubric"
        ],
        "configurationHash": "7ea4b4496d23c15f93527120927ec46105d8fc46c66114e0c710426f5f417ea8",
        "reason": "Independent adjudicator evaluated the anonymized bounded candidate record against the task rubric."
      },
      "contentHash": "307b82e64bf05c6ad5d483f625405603af17f99f0314f3504d4c3c31687421df",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:13:20.236Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-03-implementation",
        "repositoryId": "consumer-03",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 124193,
        "responseBytes": 569,
        "stderrBytes": 0,
        "stdoutHash": "9de06fed895ad0ce66563f39950d6d2502dfdfe9646608ed24abc0915c32c511",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 563453,
        "outputTokens": 2633,
        "tokenMethod": "provider",
        "toolCalls": 13,
        "errorCode": "token-budget"
      },
      "contextBytes": 2084,
      "evidenceIds": [
        "patch-evidence:.doc-bridge/report.html:packageStatus.missing=1;target=docs/for-agents/index.md",
        "verification-plan:ak-docs check --json"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "documentationFindingCount": 1,
        "documentationCompletenessRate": 0.9523809524,
        "cachedInputTokens": 509440,
        "reasoningOutputTokens": 920,
        "stderrBytes": 0,
        "providerTokenCostUnits": 566086
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "blocked",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "129d0526f843c8fe9e668f6cedf2e299a044e9176042b2c967900ff7e1bb4089",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:14:31.729Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-06-implementation",
        "repositoryId": "consumer-06",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 71450,
        "responseBytes": 666,
        "stderrBytes": 0,
        "stdoutHash": "969e100de0df1179a44c40f97d13326c2803a9df1a9bd4caaa4eafd5c5c20b23",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 666231,
        "outputTokens": 2804,
        "tokenMethod": "provider",
        "toolCalls": 11,
        "errorCode": "token-budget"
      },
      "contextBytes": 2085,
      "evidenceIds": [
        "patch-evidence:docs/security/connections-byok-containment.md and docs/testing/provider-live-certification.md distinguish general catalog certification from the MVP allowlist; existing uncommitted diff contains the minimal documentation correction",
        "verification-plan:run ak-docs check --json after approval"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 575744,
        "reasoningOutputTokens": 1266,
        "stderrBytes": 0,
        "providerTokenCostUnits": 669035
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "blocked",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "f9f69df21e95f87cf85cd0262c314862b9f111d6ae9869fbe134ca5a16ac1c81",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:15:31.343Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-01-discovery",
        "repositoryId": "consumer-01",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 59557,
        "responseBytes": 597,
        "stderrBytes": 0,
        "stdoutHash": "1fc83208bd9cd6d9ba471d19205c4854b857981b5e67f27202fc0199b99f317d",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 308144,
        "outputTokens": 2231,
        "tokenMethod": "provider",
        "toolCalls": 13
      },
      "contextBytes": 1856,
      "evidenceIds": [
        "entrypoint-evidence",
        "package.json:7-16",
        "bin/ak-docs.js:1-5",
        "src/index.ts",
        "src/mcp/server.ts",
        "doc-bridge.config.json:24-97",
        "docs/for-agents.md:8-22",
        "docs/guides/cli-map.md:8-21",
        "docs/index.md:10-37",
        "docs/agent-corpus/OVERVIEW.md:1-9"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 248832,
        "reasoningOutputTokens": 673,
        "stderrBytes": 0,
        "providerTokenCostUnits": 310375
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "3c6161e5cb01fa5bf45afc9d04a9396dfdf92ddcabc02fb45a12380af4976b36",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:16:11.943Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-01-implementation",
        "repositoryId": "consumer-01",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 40559,
        "responseBytes": 743,
        "stderrBytes": 0,
        "stdoutHash": "cbb0000d7af9d834ff7ef0647e9303b8e7c2fb02389cb24255b685b7b3dccead",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 273188,
        "outputTokens": 1304,
        "tokenMethod": "provider",
        "toolCalls": 6
      },
      "contextBytes": 2094,
      "evidenceIds": [
        "patch-evidence:docs/studies/v1-readiness-audit.md:29-31",
        "gap-evidence:docs/STABILITY.md:34-61",
        "proposal:mark-the-audit-finding-as-historical-and-align-its-package-count/tier-claims-with-the-current-stability-map;documentation-only;no-source-changes",
        "verification-plan:ak-docs-check-json-after-approval",
        "blocked:ak-docs-command-unavailable;direct-run-blocked-by-read-only-filesystem"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 228608,
        "reasoningOutputTokens": 357,
        "stderrBytes": 0,
        "providerTokenCostUnits": 274492
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "69f05d8bcc72226732892160e282c4147233ae98d63c535a1bcb6e415e62b4cf",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:17:19.608Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-04-documentation",
        "repositoryId": "consumer-04",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 67627,
        "responseBytes": 630,
        "stderrBytes": 0,
        "stdoutHash": "429e2458e3cdf3d0f53bb51ccbc76618ac2358f2c972ac7f76c80412ec909a12",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 270569,
        "outputTokens": 2781,
        "tokenMethod": "provider",
        "toolCalls": 15
      },
      "contextBytes": 2048,
      "evidenceIds": [
        "evidence-277e5cba0702c966c27e4c01cdc2c5c2",
        "review-limitation:semantic judgment is based on direct source/document comparison; ak-docs audit could not run because ak-docs is unavailable and bridge writes are blocked by read-only filesystem."
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 230400,
        "reasoningOutputTokens": 1098,
        "stderrBytes": 0,
        "providerTokenCostUnits": 273350
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "a122c3a1ffac6dc078a67f5d261be1d93ea7d549ac0468a0424132a112b228e2",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:18:09.484Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-01-implementation",
        "repositoryId": "consumer-01",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 49832,
        "responseBytes": 967,
        "stderrBytes": 0,
        "stdoutHash": "ee354e69628e492443cbfc338897f00a962e59bfae81e2d83ce535aed284c63f",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 435412,
        "outputTokens": 1749,
        "tokenMethod": "provider",
        "toolCalls": 9,
        "errorCode": "token-budget"
      },
      "contextBytes": 2085,
      "evidenceIds": [
        "patch-evidence:docs/spec/config-v1.md:836; stale planned qualifier conflicts with implemented ak-docs chat in src/cli/program.ts:1598-1613 and src/intelligence/chat.ts:75-95",
        "proposal:documentation-only;replace the stale planned qualifier in docs/spec/config-v1.md:836;no-source-changes",
        "convention:preserve-existing-CLI-mapping-table-format",
        "verification-plan:after-approval-run-ak-docs-check---json-and-review-the-one-line-diff",
        "acceptance-check-blocked:ak-docs-command-not-found;node-bin/ak-docs.js-check---json-failed-EPERM-creating-.doc-bridge/workflow/.lock"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 384512,
        "reasoningOutputTokens": 523,
        "stderrBytes": 0,
        "providerTokenCostUnits": 437161
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "blocked",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "97384a59a810f6f781c6010cc15433a5205929225344921947c0c3da12799559",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:18:38.084Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-03-discovery",
        "repositoryId": "consumer-03",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 28560,
        "responseBytes": 603,
        "stderrBytes": 0,
        "stdoutHash": "4b76c1432483519f6914280035194c03c93bed28614d0bf637076be4fb49dee2",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 71922,
        "outputTokens": 592,
        "tokenMethod": "provider",
        "toolCalls": 2
      },
      "contextBytes": 1864,
      "evidenceIds": [
        "entrypoint-evidence:package.json",
        "entrypoint-evidence:README.md",
        "entrypoint-evidence:AGENTS.md",
        "entrypoint-evidence:docs/meta.json",
        "entrypoint-evidence:docs/for-agents/index.md",
        "entrypoint-evidence:docs/for-agents/architecture.md"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 42496,
        "reasoningOutputTokens": 197,
        "stderrBytes": 0,
        "providerTokenCostUnits": 72514
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "a06b48cc750f1aa3492c6bd777006f1d1ec18a32ebb1a488931cdde606b84146",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:19:08.740Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-06-discovery",
        "repositoryId": "consumer-06",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 30622,
        "responseBytes": 418,
        "stderrBytes": 0,
        "stdoutHash": "6025b16942eba7f944edc21936108abf7e8b70ba6fd864a1c7f16bb2eeea5bb1",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 80048,
        "outputTokens": 498,
        "tokenMethod": "provider",
        "toolCalls": 2
      },
      "contextBytes": 1864,
      "evidenceIds": [
        "entrypoint-evidence"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 2,
        "cachedInputTokens": 60672,
        "reasoningOutputTokens": 223,
        "stderrBytes": 0,
        "providerTokenCostUnits": 80546
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "34df848f5cd9ab03e7244c4014bb35032ec737d6bc4a50519b2f0c0585ffeb09",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:20:15.785Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-05-architecture",
        "repositoryId": "consumer-05",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 67007,
        "responseBytes": 732,
        "stderrBytes": 0,
        "stdoutHash": "24fc25e7233cd2e83435824da9bdb0341fe3b20db6b305e38453f53a1c6a94ec",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 356146,
        "outputTokens": 2493,
        "tokenMethod": "provider",
        "toolCalls": 10
      },
      "contextBytes": 1884,
      "evidenceIds": [
        "architecture-evidence:docs/architecture.md",
        "architecture-evidence:package.json",
        "architecture-evidence:doc-bridge.config.json",
        "architecture-evidence:scripts/build-registry.mjs",
        "architecture-evidence:scripts/lib/deterministic-discovery.mjs",
        "boundary-review:Registry-to-AgentsKit-Chat-protocol",
        "acceptance-check:ak-docs-map-blocked-command-not-found-and-EPERM"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "not-applicable",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 300544,
        "reasoningOutputTokens": 1070,
        "stderrBytes": 0,
        "providerTokenCostUnits": 358639
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "ab80ea104f4403ec13eaebebaa25b809c0baa663e176a0d1e1def51dada6cb83",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:20:47.621Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-06-discovery",
        "repositoryId": "consumer-06",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 31798,
        "responseBytes": 869,
        "stderrBytes": 0,
        "stdoutHash": "db58e081dc4d87bd4a04cd69a5cd47369c0c5e6bb984fb11d62f03773377e0c9",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 126411,
        "outputTokens": 1182,
        "tokenMethod": "provider",
        "toolCalls": 2
      },
      "contextBytes": 1856,
      "evidenceIds": [
        "entrypoint-evidence",
        "artifact:README.md",
        "artifact:package.json",
        "artifact:AGENTS.md",
        "artifact:docs/for-agents/INDEX.md",
        "artifact:docs/internal/README.md",
        "artifact:apps/console/src/main.tsx",
        "artifact:apps/desktop/src/main.tsx",
        "artifact:apps/cloud/src/index.ts",
        "artifact:apps/admin/package.json",
        "artifact:apps/web/README.md",
        "artifact:apps/license-service/src/index.ts",
        "artifact:packages/os-cli/src/index.ts",
        "artifact:.codex/verification.json",
        "acceptance-check:discovery-check:blocked-command-not-found"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 81152,
        "reasoningOutputTokens": 517,
        "stderrBytes": 0,
        "providerTokenCostUnits": 127593
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "0bbc42a94d3fc3a3187275510609c7d536df88ca40c78a2bcaef0238f9e1100d",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:21:37.667Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-06-implementation",
        "repositoryId": "consumer-06",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 50005,
        "responseBytes": 790,
        "stderrBytes": 0,
        "stdoutHash": "20a0ce4a7a5d8e623e396e028a1b3f58f9ce70773864cf75c3eb4991d32c1a2d",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 134086,
        "outputTokens": 1069,
        "tokenMethod": "provider",
        "toolCalls": 4
      },
      "contextBytes": 2084,
      "evidenceIds": [
        "evidence-065debe43894dfdaaf5ae64560798762",
        "verification-plan:ak-docs check --json; additionally review the documentation diff to confirm only the provider-certification wording changes and existing terminology is preserved",
        "runner-blocker:Doc Bridge search could not execute because the read-only environment denied creation of its temporary file, so the required acceptance check was not established"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 119552,
        "reasoningOutputTokens": 452,
        "stderrBytes": 0,
        "providerTokenCostUnits": 135155
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "7e924044d24d057b80926387323264d4280fa785fa71b8509dd568321ae1937e",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:23:02.149Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-05-documentation",
        "repositoryId": "consumer-05",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 84440,
        "responseBytes": 653,
        "stderrBytes": 0,
        "stdoutHash": "a480be8e95aecf53273fd07ef2326b6931bff26808ed43d2c7aa565af35e627b",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 226351,
        "outputTokens": 1676,
        "tokenMethod": "provider",
        "toolCalls": 7
      },
      "contextBytes": 2056,
      "evidenceIds": [
        "reconciliation:RELATION_UNDOCUMENTED:008635f212a848069814456bee2b774a:graph-undocumented-relation",
        "registry/product-nps-analyzer/agent.ts:4",
        "registry/product-nps-analyzer/README.md:1",
        "review-limitation:audit-could-not-be-refreshed-read-only-lock"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 1,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 5117,
        "cachedInputTokens": 196608,
        "reasoningOutputTokens": 486,
        "stderrBytes": 0,
        "providerTokenCostUnits": 228027
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "fa86eb509e5f4a37be974df3e09a0be00fccad79af6db1ba048335887a4a5062",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:24:00.563Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-02-architecture",
        "repositoryId": "consumer-02",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 58374,
        "responseBytes": 713,
        "stderrBytes": 0,
        "stdoutHash": "6b7cd211c2aa4d67444b52e587b504ae9506589f3f352a348f1b4d849032878e",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 219608,
        "outputTokens": 936,
        "tokenMethod": "provider",
        "toolCalls": 6
      },
      "contextBytes": 1883,
      "evidenceIds": [
        "architecture-evidence:apps/docs-next/content/docs/get-started/architecture-at-a-glance.mdx#the-stack-in-one-view",
        "architecture-evidence:package.json#scripts",
        "attention-point:documented-direction-core-to-consumers-deserves-import-level-review",
        "acceptance-blocked:ak-docs-map-json-EPERM-doc-bridge-workflow-lock"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 1,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 192768,
        "reasoningOutputTokens": 175,
        "stderrBytes": 0,
        "providerTokenCostUnits": 220544
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "e014ad4dd9b5e75c5d306b64f9d3ead51023cec5f338d36e04d4c77b8a7a2f23",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:25:27.326Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-05-documentation",
        "repositoryId": "consumer-05",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 86723,
        "responseBytes": 1064,
        "stderrBytes": 0,
        "stdoutHash": "648038e1f39b5ee389728c7531a13dec4d9521d1ce2e460e131f818b750c9494",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 463484,
        "outputTokens": 3211,
        "tokenMethod": "provider",
        "toolCalls": 13,
        "errorCode": "token-budget"
      },
      "contextBytes": 2057,
      "evidenceIds": [
        "documentation-evidence: README.md claims 346 validated agents; public/r/index.json reports validated=346, so this claim is current, not stale.",
        "documentation-evidence: README.md repeats the same 346/333 claims in multiple sections, creating redundant documentation that should be consolidated or generated.",
        "review-limitation: audit reports DOCUMENTATION_SEMANTICS_NOT_ANALYZED with low confidence; natural-language redundancy and contradiction detection require human review.",
        "acceptance-check: ak-docs audit documentation --json was blocked by EPERM creating .doc-bridge/workflow/.lock; no current executable audit result."
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "documentationFindingCount": 1,
        "documentationExampleRate": 0.16145833333333334,
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 418304,
        "reasoningOutputTokens": 1421,
        "stderrBytes": 0,
        "providerTokenCostUnits": 466695
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "blocked",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "af030088a5f38883535131f55ae1f5d2fa8d5cf6a5b6f407698b2cc6e6d043d7",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:26:15.581Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-06-architecture",
        "repositoryId": "consumer-06",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 48212,
        "responseBytes": 379,
        "stderrBytes": 0,
        "stdoutHash": "89026e1369a54354c2e90edfe48471a363d2ec140da6b92c214310ab0fabf610",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 281449,
        "outputTokens": 1883,
        "tokenMethod": "provider",
        "toolCalls": 6
      },
      "contextBytes": 1893,
      "evidenceIds": [
        "architecture-evidence"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 224768,
        "reasoningOutputTokens": 757,
        "stderrBytes": 0,
        "providerTokenCostUnits": 283332
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "db62c3ddbddc218f53945a0ff1441a99ea39665c3351085a4b841bed5a5ec4b6",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:27:06.797Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-01-implementation",
        "repositoryId": "consumer-01",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 51103,
        "responseBytes": 889,
        "stderrBytes": 0,
        "stdoutHash": "06f6422045364c5e936a9b8bb027cba36f51db09ee40819f8331a506efcb22e8",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 143491,
        "outputTokens": 1076,
        "tokenMethod": "provider",
        "toolCalls": 4
      },
      "contextBytes": 2084,
      "evidenceIds": [
        "patch-evidence:docs/spec/config-v1.md:836 — the CLI mapping incorrectly marks `ak-docs chat` as planned",
        "implementation-evidence:src/cli/program.ts:20,149,895,1644 and src/intelligence/chat.ts:75 — the chat command and execution path exist",
        "proposed-patch:docs/spec/config-v1.md:836 — replace `planned; intelligence.*` with `intelligence.*`, preserving the existing table convention",
        "verification-plan:after approval run `ak-docs check --json` and review the single-row documentation diff against the cited implementation"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "success",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksTotal": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 115456,
        "reasoningOutputTokens": 405,
        "stderrBytes": 0,
        "providerTokenCostUnits": 144567
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "blocked",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "5dfbd9d54db5771adc53ebe8a5a0c1c53ac8bced1d6afb9c98f9d62239930595",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:28:20.548Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-01-architecture",
        "repositoryId": "consumer-01",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 73714,
        "responseBytes": 1155,
        "stderrBytes": 0,
        "stdoutHash": "d807f4b3cefada1944e3e7224838161716a638acf3b3ac1d7ab43efe7f9af067",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 586815,
        "outputTokens": 3083,
        "tokenMethod": "provider",
        "toolCalls": 11,
        "errorCode": "token-budget"
      },
      "contextBytes": 1893,
      "evidenceIds": [
        "architecture-evidence: stale normalized artifact .doc-bridge/workflow/artifacts/normalize-d375d63d...; observed 1 package, 199 modules, 86 documents, and 1147 relations",
        "architecture-evidence: CLI -> index-builder, query, gates, discovery, MCP, intelligence, memory, study, and report; index-builder -> config, lib, schemas; study -> index-builder",
        "architecture-evidence: review optional-runtime boundary at src/agents/registry-adapter.ts:140 and src/intelligence/peers.ts:50; non-literal dynamic imports are explicitly unresolved",
        "architecture-evidence: docs/for-agents.md and docs/agent-corpus/doc-bridge.md claim deterministic ownership routing through CLI, index, MCP, gates, and doctor",
        "architecture-check: node bin/ak-docs.js map --json blocked by EPERM creating .doc-bridge/workflow/.lock"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "low",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 509952,
        "reasoningOutputTokens": 988,
        "stderrBytes": 0,
        "providerTokenCostUnits": 589898
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "blocked",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "47fabb5859ad49ed65469c2628e32ccf2c1a0154c9bd68950c02b66e5f71cd3b",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:31:20.593Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-06-discovery",
        "repositoryId": "consumer-06",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "timed-out",
        "exitCode": null,
        "signal": "SIGTERM",
        "durationMs": 180004,
        "responseBytes": 0,
        "stderrBytes": 0,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "errorCode": "timeout"
      },
      "contextBytes": 1865,
      "evidenceIds": [],
      "round": "ab-adjudicated-cost-2026-08-31",
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "blocked",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "c8853b61569294f55eaa4fc411964e5a292e7f4802edc10be5356b6d9406000c",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:31:44.135Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-01-discovery",
        "repositoryId": "consumer-01",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 23497,
        "responseBytes": 364,
        "stderrBytes": 0,
        "stdoutHash": "03cc491e158c8f2b1a8c7d8c9799c952edf18eafe82dab63cacb989ef10c1501",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 63247,
        "outputTokens": 421,
        "tokenMethod": "provider",
        "toolCalls": 2
      },
      "contextBytes": 1864,
      "evidenceIds": [],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "low",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 40448,
        "reasoningOutputTokens": 137,
        "stderrBytes": 0,
        "providerTokenCostUnits": 63668
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "incomplete",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "6627a5f94a16c9717c701c95d05093407846c696907adf354c69484f0dbfe048",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:32:23.659Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-01-architecture",
        "repositoryId": "consumer-01",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 39488,
        "responseBytes": 858,
        "stderrBytes": 0,
        "stdoutHash": "b5f02297d7692c9bcc9c5b4e51657e38f6bca148276ad62c6b8cf7d8a3a53574",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 193067,
        "outputTokens": 1413,
        "tokenMethod": "provider",
        "toolCalls": 5
      },
      "contextBytes": 1884,
      "evidenceIds": [
        "architecture-evidence",
        "src/cli/program.ts:orchestrates-config-discovery-index-query-gates-mcp-study",
        "src/index-builder/build-index.ts:discovery-to-index-pipeline",
        "src/mcp/server.ts:read-only-mcp-surface-over-query-gates-memory-workflow",
        "src/discovery/repository.ts:workspace-and-import-discovery",
        "src/query/query.ts:handoff-query-boundary",
        "src/intelligence/rag.ts:optional-peer-integration-boundary",
        "attention-point:src/cli/program.ts-is-a-high-coupling-orchestration-boundary-deserving-review"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 158208,
        "reasoningOutputTokens": 594,
        "stderrBytes": 0,
        "providerTokenCostUnits": 194480
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "4b8955bea5aba8aa60af464d3602b737010d6d4d94f02d25224869bda6889b26",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:33:11.664Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-04-architecture",
        "repositoryId": "consumer-04",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 47953,
        "responseBytes": 400,
        "stderrBytes": 0,
        "stdoutHash": "dea5a1fdf7d7bc70a61b042ae35ba5cb2363b89061425a2d94b659ab41238241",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 213351,
        "outputTokens": 1704,
        "tokenMethod": "provider",
        "toolCalls": 6
      },
      "contextBytes": 1893,
      "evidenceIds": [
        "architecture-evidence",
        "architecture-check"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 177408,
        "reasoningOutputTokens": 715,
        "stderrBytes": 0,
        "providerTokenCostUnits": 215055
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "d6537726e7d2a712ff96ade4abce1d190b980091fb9f5cff6b0c264fb382a7ac",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:33:35.901Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-01-discovery",
        "repositoryId": "consumer-01",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 24202,
        "responseBytes": 428,
        "stderrBytes": 0,
        "stdoutHash": "9c7a60ac73ad9e4d63bd27cd0c238e24a959cd967be1b57a6ccd9959cecfd4c4",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 42054,
        "outputTokens": 459,
        "tokenMethod": "provider",
        "toolCalls": 1
      },
      "contextBytes": 1855,
      "evidenceIds": [
        "package.json",
        "README.md",
        "docs/index.md",
        "apps/docs/AGENTS.md"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "low",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 20224,
        "reasoningOutputTokens": 188,
        "stderrBytes": 0,
        "providerTokenCostUnits": 42513
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "3d6ffb15eac7da44c2e2902435914a6210b9e40bafe4e960ae09054bb697c46c",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:34:30.680Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-01-architecture",
        "repositoryId": "consumer-01",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 54748,
        "responseBytes": 821,
        "stderrBytes": 0,
        "stdoutHash": "7a99f0a82df7f78267b68e0dc5b259e3998107f22d3fd75cfb2c9cab1aad6ad8",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 140067,
        "outputTokens": 1053,
        "tokenMethod": "provider",
        "toolCalls": 4
      },
      "contextBytes": 1892,
      "evidenceIds": [
        "architecture-evidence:package.json",
        "architecture-evidence:pnpm-workspace.yaml",
        "architecture-evidence:src/cli/program.ts",
        "architecture-evidence:src/index-builder/build-index.ts",
        "architecture-evidence:src/query/query.ts",
        "architecture-evidence:src/mcp/server.ts",
        "architecture-attention-point:mcp-server-cross-component-coupling",
        "acceptance-evidence:ak-docs-map-command-not-found",
        "acceptance-evidence:node-bin-map-blocked-by-read-only-filesystem"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 111360,
        "reasoningOutputTokens": 312,
        "stderrBytes": 0,
        "providerTokenCostUnits": 141120
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "276bd155316078b691e2bc61a9e350190abe347da1d91ce0390772ab7d774482",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:35:34.217Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-02-architecture",
        "repositoryId": "consumer-02",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 63502,
        "responseBytes": 401,
        "stderrBytes": 0,
        "stdoutHash": "47d6ab72a8ce2187f5d02dcf0ffd1977951ee8f2f346f7bf165d48d3602bf5ab",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 416149,
        "outputTokens": 2232,
        "tokenMethod": "provider",
        "toolCalls": 11,
        "errorCode": "token-budget"
      },
      "contextBytes": 1893,
      "evidenceIds": [
        "architecture-evidence",
        "architecture-check"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 373760,
        "reasoningOutputTokens": 770,
        "stderrBytes": 0,
        "providerTokenCostUnits": 418381
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "blocked",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "be0cf69f922d6a6986de2b4168efe26d8b0ba087521a3b1c8086c266cf072be2",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:36:28.008Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-06-implementation",
        "repositoryId": "consumer-06",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 53752,
        "responseBytes": 487,
        "stderrBytes": 0,
        "stdoutHash": "29299354927a4ee1c9c55c88e8919dcfad83ba01be7405ea1ecab9d9a8878f14",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 175251,
        "outputTokens": 1069,
        "tokenMethod": "provider",
        "toolCalls": 7
      },
      "contextBytes": 2093,
      "evidenceIds": [
        "patch-evidence:docs/rfc/0056-doc-bridge-ecosystem-dogfood.md",
        "verification-plan:ak-docs-check-json"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 164352,
        "reasoningOutputTokens": 363,
        "stderrBytes": 0,
        "providerTokenCostUnits": 176320
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "9fcb259d9fa80cbbb5670bfbbbaa9bb1dbfc9f019f0096ec881b4842715db279",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:36:59.356Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-05-implementation",
        "repositoryId": "consumer-05",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 31313,
        "responseBytes": 618,
        "stderrBytes": 0,
        "stdoutHash": "d2b61f59041b00e617c8e051d62c3512c2aee36dd6f0c59c1e8eb85569e8722c",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 89451,
        "outputTokens": 657,
        "tokenMethod": "provider",
        "toolCalls": 2
      },
      "contextBytes": 2093,
      "evidenceIds": [
        "patch-evidence:docs/for-agents/registry-discovery.md:document-exact-generated-artifacts-and-draft-exclusion-from-scripts/lib/deterministic-discovery.mjs",
        "verification-plan:ak-docs check --json:blocked-command-not-found"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 51712,
        "reasoningOutputTokens": 258,
        "stderrBytes": 0,
        "providerTokenCostUnits": 90108
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "bedbf616e1fb107219f3a48d54e4677ba46ae4c80abed836952dbded8dd13a1d",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:37:35.443Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-05-architecture",
        "repositoryId": "consumer-05",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 36044,
        "responseBytes": 494,
        "stderrBytes": 0,
        "stdoutHash": "c4596628682d8d48a18969d3f5a3e8673cf8d1bffd13c2753e02f480ba01f5c5",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 84454,
        "outputTokens": 566,
        "tokenMethod": "provider",
        "toolCalls": 3
      },
      "contextBytes": 1892,
      "evidenceIds": [
        "architecture-evidence:package.json:34-37",
        "architecture-evidence:pnpm-workspace.yaml:4",
        "architecture-check:ak-docs-map-eperm"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 60416,
        "reasoningOutputTokens": 184,
        "stderrBytes": 0,
        "providerTokenCostUnits": 85020
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "ad560bf16726fb3cbe0985c547ca461f1a927882d2cce507fb66ba4bcb079915",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:38:12.363Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-05-architecture",
        "repositoryId": "consumer-05",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 36884,
        "responseBytes": 506,
        "stderrBytes": 0,
        "stdoutHash": "f9d9125c2c4334f7f8e2d8a0290890f05aff630e223f7da3146fecd09eda5179",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 88475,
        "outputTokens": 645,
        "tokenMethod": "provider",
        "toolCalls": 3
      },
      "contextBytes": 1883,
      "evidenceIds": [
        "package.json:17",
        "package.json:43-50",
        "pnpm-workspace.yaml:4",
        "docs/for-agents/index.md:32",
        "ak-docs-map:EPERM-.doc-bridge/workflow/.lock"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 62720,
        "reasoningOutputTokens": 245,
        "stderrBytes": 0,
        "providerTokenCostUnits": 89120
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "f951f17540c525ec04265ea1ef0935ab9ae09ddf8be7a948bdf7f82ad47c56e7",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:38:34.890Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-02-discovery",
        "repositoryId": "consumer-02",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 22493,
        "responseBytes": 364,
        "stderrBytes": 0,
        "stdoutHash": "3b17d9f5bf16f2a7c213c0af5fbdc2bd2b82a78f43d9c0c3a036c819b9fff567",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 66743,
        "outputTokens": 401,
        "tokenMethod": "provider",
        "toolCalls": 2
      },
      "contextBytes": 1855,
      "evidenceIds": [],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "low",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 63744,
        "reasoningOutputTokens": 179,
        "stderrBytes": 0,
        "providerTokenCostUnits": 67144
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "incomplete",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "cc1301eebc81352f1713051e70664847800d36442732ad25df8c175483a6c427",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:39:28.511Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-02-architecture",
        "repositoryId": "consumer-02",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 53584,
        "responseBytes": 478,
        "stderrBytes": 0,
        "stdoutHash": "0eebd513f07ca044e69c2b7b22e100243d183355cc6220dc209645aaa6721116",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 169665,
        "outputTokens": 874,
        "tokenMethod": "provider",
        "toolCalls": 6
      },
      "contextBytes": 1892,
      "evidenceIds": [
        "architecture-check:ak-docs-map-json:command-not-found",
        "architecture-check:local-binary:filesystem-lock-denied"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "low",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 136704,
        "reasoningOutputTokens": 230,
        "stderrBytes": 0,
        "providerTokenCostUnits": 170539
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "0b15728859d3e919ee5440a9cb207d19b2b495cbf139b4bdabe34c1e384594d2",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:40:36.868Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-03-implementation",
        "repositoryId": "consumer-03",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 68316,
        "responseBytes": 420,
        "stderrBytes": 0,
        "stdoutHash": "0dffa43e1cce6472f3d2926f10bd19fa46f11ef8203a58a91be7cff2cfc196a8",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 278829,
        "outputTokens": 1450,
        "tokenMethod": "provider",
        "toolCalls": 8
      },
      "contextBytes": 2093,
      "evidenceIds": [
        "missing-preceding-gap-evidence",
        "verification-command-unavailable"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "low",
      "safetyOutcome": "safe",
      "clarificationRequests": 1,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 235520,
        "reasoningOutputTokens": 348,
        "stderrBytes": 0,
        "providerTokenCostUnits": 280279
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "02908475266f045c6b4f61f94c47bbb5a1d5bf6d66a9447507e0c5991d48da5a",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:41:57.780Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-04-implementation",
        "repositoryId": "consumer-04",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 80872,
        "responseBytes": 526,
        "stderrBytes": 0,
        "stdoutHash": "7830d758c1768fa96d9d33c4b91c2b13b304d3c0c3b2600191e1d95c65b06a40",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 490003,
        "outputTokens": 3057,
        "tokenMethod": "provider",
        "toolCalls": 11,
        "errorCode": "token-budget"
      },
      "contextBytes": 2085,
      "evidenceIds": [
        "patch-evidence:content/docs/index.mdx:6-8;add docbridge.covers package:@agentskit/playbook",
        "verification-plan:ak-docs check --json;blocked by read-only workflow lock"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 441344,
        "reasoningOutputTokens": 1206,
        "stderrBytes": 0,
        "providerTokenCostUnits": 493060
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "blocked",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "62ecc78757fc1be3cd1aa6137d18bbf389e96fbb9646c9eded16d026e6ec35cd",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:42:42.992Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-02-architecture",
        "repositoryId": "consumer-02",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 45171,
        "responseBytes": 1298,
        "stderrBytes": 0,
        "stdoutHash": "2d33c3800f3718b2c14045e86b76b22f5180583a69ce10680bd0a031fa846be4",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 192491,
        "outputTokens": 1817,
        "tokenMethod": "provider",
        "toolCalls": 5
      },
      "contextBytes": 1884,
      "evidenceIds": [
        "evidence-8f5860d28e8e6c5025449fa996cc3789",
        "architecture-evidence: @agentskit/runtime composes adapters, tools, skills and memory; packages/runtime/CONVENTIONS.md defines these as separate boundaries and packages/cli/package.json imports runtime, adapters, tools, templates, memory, rag and skills.",
        "architecture-evidence: @agentskit/rag composes a Retriever from an embedder supplied by @agentskit/adapters and a vector store supplied by @agentskit/memory; documented in packages/rag/CONVENTIONS.md and apps/docs-next/lib/rag/* imports.",
        "architecture-evidence: React, Ink, Vue, Svelte, Solid, Angular and React Native bindings consume core chat contracts; observed in package metadata, framework package READMEs and imports.",
        "evidence-dc6f9dc6961b15ed165d2b0b84779f97",
        "acceptance-check: pnpm exec ak-docs map --json was attempted but blocked by EPERM creating repository-local temporary file _tmp_56126_9e41148b286a456ec2e88a341db1e6e2"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 155136,
        "reasoningOutputTokens": 647,
        "stderrBytes": 0,
        "providerTokenCostUnits": 194308
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "ecc858aec8f66d75021a2eaa33264eb98147595758669501bb29f93f369dce97",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:43:13.956Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-03-discovery",
        "repositoryId": "consumer-03",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 30925,
        "responseBytes": 388,
        "stderrBytes": 0,
        "stdoutHash": "bf103b48ef858176752ff52a3e5725f94fb22f7cef24ead638246578716c9101",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 95655,
        "outputTokens": 618,
        "tokenMethod": "provider",
        "toolCalls": 3
      },
      "contextBytes": 1855,
      "evidenceIds": [
        "entrypoint-evidence"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 64768,
        "reasoningOutputTokens": 206,
        "stderrBytes": 0,
        "providerTokenCostUnits": 96273
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "9cc7e33c46054c67bf67e8b39629ad1e7a05dde4de0aeecf6aa1fd385bfa38cb",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:44:13.915Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-02-documentation",
        "repositoryId": "consumer-02",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 59916,
        "responseBytes": 921,
        "stderrBytes": 0,
        "stdoutHash": "b700c9476950207ce0a00f65cb14449b310b04abcdbd8b8c0cb34b0f769e1a9a",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 316614,
        "outputTokens": 2229,
        "tokenMethod": "provider",
        "toolCalls": 7
      },
      "contextBytes": 2057,
      "evidenceIds": [
        "documentation-evidence:packages/cli/src/rules/windsurf.ts:8 claims ~19 packages; scripts/compute-stats.mjs derives 22 published packages; ecosystem-claims.json records 22; apps/docs-next/content/docs/reference/packages/overview.mdx:95 agrees with 22",
        "review-limitation:confidence medium; semantic audit could not complete because the read-only workspace rejected creation of .doc-bridge/workflow/.lock (EPERM); recommended next action is replace the hard-coded ~19 claim with the canonical derived count",
        "audit-error:ak-docs audit documentation --json exited 2"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 268288,
        "reasoningOutputTokens": 1099,
        "stderrBytes": 0,
        "providerTokenCostUnits": 318843
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "f681ea90635e1d4f1549a93d2942c07d8d156c7cebeabfec659685ddf8066fdc",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:45:22.566Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-06-architecture",
        "repositoryId": "consumer-06",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 68601,
        "responseBytes": 366,
        "stderrBytes": 0,
        "stdoutHash": "584b092b0223376fde9be2da1eaea61871e7a639d651eda893d37657343843ce",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 163989,
        "outputTokens": 686,
        "tokenMethod": "provider",
        "toolCalls": 4
      },
      "contextBytes": 1892,
      "evidenceIds": [],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "low",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 1,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 148992,
        "reasoningOutputTokens": 223,
        "stderrBytes": 0,
        "providerTokenCostUnits": 164675
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "incomplete",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "2f90ad41d90626f911103a20baa179584755c9d0125654eafc1eb2e16c08b6a9",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:47:05.800Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-02-implementation",
        "repositoryId": "consumer-02",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 103182,
        "responseBytes": 649,
        "stderrBytes": 0,
        "stdoutHash": "dd8941e35d145fd619f3ca1be38102ac86dbc4abaeddf0c0e80bc9af24f1e0a1",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 308259,
        "outputTokens": 2033,
        "tokenMethod": "provider",
        "toolCalls": 9
      },
      "contextBytes": 2084,
      "evidenceIds": [
        "evidence-5febe3c6559b3b821d119204565fb090",
        "verification-plan:After approval, run `ak-docs check --json` and require a successful exit for the repository contract; also review the diff to confirm only apps/docs-next/content/docs/for-agents/doc-bridge.mdx changed."
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "success",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 252672,
        "reasoningOutputTokens": 767,
        "stderrBytes": 0,
        "providerTokenCostUnits": 310292
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "38d3e8c9cd0ea5ba09d1a1d46074a7a6306f9ada904e652e528e36bf7c79f6c7",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:49:29.208Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-06-documentation",
        "repositoryId": "consumer-06",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 143189,
        "responseBytes": 641,
        "stderrBytes": 0,
        "stdoutHash": "43bc8d082ef8c3304484fce8ff8a03e4fe7fe13dbfe7f0b3d0b67e92ea7c7b7d",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 331512,
        "outputTokens": 1761,
        "tokenMethod": "provider",
        "toolCalls": 8
      },
      "contextBytes": 2047,
      "evidenceIds": [
        "evidence-7f2d801aab371aa8c7cdcbc7c29ed5fa",
        "evidence-35547cd11b670175f464169e257071b6",
        "acceptance-check-blocked:ak-docs audit documentation --json could not complete because the read-only environment denied creation of .doc-bridge/workflow/.lock (EPERM)."
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "documentationFindingCount": 1,
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 274432,
        "reasoningOutputTokens": 464,
        "stderrBytes": 0,
        "providerTokenCostUnits": 333273
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "58c1a6aa21b36317c4c11d88c51dd7313fb443bea4f9ca5749a6f8b05f6f036c",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:50:47.527Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-01-documentation",
        "repositoryId": "consumer-01",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 78275,
        "responseBytes": 790,
        "stderrBytes": 0,
        "stdoutHash": "903cbbfaa8bfc4ee8295fa8b65a6ad968367bd4531a1913e37acb496d3ce0365",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 175903,
        "outputTokens": 1562,
        "tokenMethod": "provider",
        "toolCalls": 6
      },
      "contextBytes": 2056,
      "evidenceIds": [
        "documentation-evidence:stale-version-claim:README.md:344 claims v1.4.0 stable; package.json:3 declares 1.7.45; recommended action: update or derive the README version claim from the release source of truth",
        "review-limitation:high-confidence direct textual contradiction; full documentation audit could not run because the read-only environment denied creation of .doc-bridge/workflow/.lock"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 158976,
        "reasoningOutputTokens": 560,
        "stderrBytes": 0,
        "providerTokenCostUnits": 177465
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "35918ca3bc22d0a2bd359f775b82efdf8f0f4b2873fc29f7206939f5bdf4f1d7",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:53:47.711Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-06-documentation",
        "repositoryId": "consumer-06",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "timed-out",
        "exitCode": null,
        "signal": "SIGTERM",
        "durationMs": 180038,
        "responseBytes": 0,
        "stderrBytes": 0,
        "stdoutHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "errorCode": "timeout"
      },
      "contextBytes": 2048,
      "evidenceIds": [],
      "round": "ab-adjudicated-cost-2026-08-31",
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "blocked",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "e10925f24e8cefc6cefc21fedc337ad5615dc91746015f0e38c7ad0bfd373106",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:55:10.655Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-03-discovery",
        "repositoryId": "consumer-03",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 82830,
        "responseBytes": 1797,
        "stderrBytes": 0,
        "stdoutHash": "9f4b375dc27eabf568f380d5f036e9e18df970549304b1446fc13fa79d56300f",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 477647,
        "outputTokens": 2826,
        "tokenMethod": "provider",
        "toolCalls": 9,
        "errorCode": "token-budget"
      },
      "contextBytes": 1856,
      "evidenceIds": [
        "entrypoint-evidence:packages/chat/src/index.ts:516-539:ChatDefinition-and-defineChat",
        "entrypoint-evidence:packages/chat/src/index.ts:674:createChatSession",
        "entrypoint-evidence:packages/server/src/index.ts:12-52:ChatHandler-and-createChatHandler",
        "entrypoint-evidence:packages/server/src/ask-service.ts:129:createAskServiceHandler",
        "entrypoint-evidence:packages/cli/src/bin.ts:8-46:CLI-entrypoint",
        "entrypoint-evidence:packages/cli/src/index.ts:176:initChatProject",
        "entrypoint-evidence:packages/react/src/index.tsx:323:AgentChat",
        "entrypoint-evidence:packages/react-native/src/index.tsx:256:AgentChatNative",
        "entrypoint-evidence:packages/ink/src/index.tsx:293:AgentChat",
        "entrypoint-evidence:apps/docs/app/page.tsx:38:docs-home-entrypoint",
        "entrypoint-evidence:apps/example-shared/src/index.ts:33-52:shared-definition",
        "ownership-evidence:docs/for-agents/index.md:5-30:canonical-ownership-index",
        "ownership-evidence:docs/for-agents/architecture.md:110-120:architecture-canonical-docs",
        "ownership-evidence:docs/for-agents/packages/chat.md:3-20:chat-package-boundary",
        "ownership-evidence:docs/architecture/overview.md:1-120:system-boundaries-and-modules",
        "ownership-evidence:docs/architecture/upstream-adoption.md:1-40:upstream-adoption-boundary",
        "discovery-evidence:ak-docs-discover-json:structured-snapshot-emitted-with-local-bin",
        "acceptance-evidence:discovery-check:direct-command-not-found-and-pnpm-exec-blocked-by-read-only-tempfile"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 408064,
        "reasoningOutputTokens": 1116,
        "stderrBytes": 0,
        "providerTokenCostUnits": 480473
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "blocked",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "5d80da22fc1391eba055c82a3f358c2ebbe1a96acd3daf1198c152a4ec9a6e8e",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:56:45.357Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-02-discovery",
        "repositoryId": "consumer-02",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 94602,
        "responseBytes": 885,
        "stderrBytes": 0,
        "stdoutHash": "1dc31630c1ba4bbf43e09e6a9994f0892df2d0122e389f0b021b2c193018aad0",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 265255,
        "outputTokens": 2143,
        "tokenMethod": "provider",
        "toolCalls": 6
      },
      "contextBytes": 1856,
      "evidenceIds": [
        "AGENTS.md:53-70",
        "AGENTS.md:87-95",
        "package.json:1-43",
        "pnpm-workspace.yaml:1-3",
        "apps/docs-next/README.md:65",
        "packages/core/package.json:2",
        "packages/core/src/index.ts:1",
        "packages/cli/package.json:2",
        "packages/cli/src/bin.ts:1",
        "packages/mcp/package.json:2",
        "packages/mcp/src/cli.ts:1",
        "apps/example-edge/package.json:2",
        "apps/example-edge/src/worker.ts:1",
        "discover:sourceRevision=0551139b338745da778fffaad15df2a521eca561",
        "discover:configurationHash=8b01bbcbc686cb6050a8602de90d6c451131bf4aea9038f0b3a93f110bbda224"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "not-applicable",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 1,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 220416,
        "reasoningOutputTokens": 971,
        "stderrBytes": 0,
        "providerTokenCostUnits": 267398
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "5f86f2d3f8a4b5be31e90f3499716ca308f3d230788cfdd81b4aa6474761126b",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:57:11.951Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-05-discovery",
        "repositoryId": "consumer-05",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 26051,
        "responseBytes": 364,
        "stderrBytes": 0,
        "stdoutHash": "f0035e8d4910269d3d3146d653a5a0d88812c284579263137fdba118f3581539",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 42098,
        "outputTokens": 330,
        "tokenMethod": "provider",
        "toolCalls": 1
      },
      "contextBytes": 1855,
      "evidenceIds": [],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "low",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 20224,
        "reasoningOutputTokens": 177,
        "stderrBytes": 0,
        "providerTokenCostUnits": 42428
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "incomplete",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "e9d42cd314aa0b983a4690d2adcceaf4707831e7d178228e6ed99fd01528067f",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:58:16.797Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-03-architecture",
        "repositoryId": "consumer-03",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 64743,
        "responseBytes": 689,
        "stderrBytes": 0,
        "stdoutHash": "ee14fda900c0ae4c54b0951dea7dccf2b443d38fc6e4ec1119324395c831e7c8",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 343734,
        "outputTokens": 2642,
        "tokenMethod": "provider",
        "toolCalls": 7
      },
      "contextBytes": 1893,
      "evidenceIds": [
        "architecture-evidence",
        "docs/architecture/overview.md",
        "docs/for-agents/architecture.md",
        "docs/architecture/adrs/0002-upstream-first-no-reimplementation.md",
        "doc-bridge.config.json",
        "packages/chat/src/index.ts",
        "packages/protocol/src/index.ts",
        "packages/server/src/index.ts",
        "packages/react/src/index.tsx",
        ".doc-bridge/report.html"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 270336,
        "reasoningOutputTokens": 1390,
        "stderrBytes": 0,
        "providerTokenCostUnits": 346376
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "8486da5c8623b508b32f2fd9fda99c7749ca5c41c793c9351d8777575f836179",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:58:50.180Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-01-architecture",
        "repositoryId": "consumer-01",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 33346,
        "responseBytes": 578,
        "stderrBytes": 0,
        "stdoutHash": "ad58240c7b7b34be58e8056aabd40d831237791bcc426163aa8bcf555dce8899",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 69059,
        "outputTokens": 612,
        "tokenMethod": "provider",
        "toolCalls": 2
      },
      "contextBytes": 1883,
      "evidenceIds": [
        "architecture-evidence:package.json",
        "architecture-evidence:pnpm-workspace.yaml",
        "architecture-evidence:src/index.ts",
        "architecture-evidence:src/cli/program.ts",
        "architecture-evidence:docs/agent-corpus/OVERVIEW.md"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "low",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 43520,
        "reasoningOutputTokens": 230,
        "stderrBytes": 0,
        "providerTokenCostUnits": 69671
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "f52cceb8119b5dd25f62c7221dc196874a502b161961389371247dc6034a8bd9",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T20:59:59.101Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-01-documentation",
        "repositoryId": "consumer-01",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 68886,
        "responseBytes": 603,
        "stderrBytes": 0,
        "stdoutHash": "b9ba2e3508b36d3c2cc8dfad6db2ce9d15a2f6ac73bd7983290c542b1d426e5b",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 171773,
        "outputTokens": 1421,
        "tokenMethod": "provider",
        "toolCalls": 6
      },
      "contextBytes": 2047,
      "evidenceIds": [
        "evidence-f042363412289e9d6d7e4f2babb213cf",
        "evidence-3aa1b1e78e3bff7f03e839844ef25d18",
        "artifact:docs/study/documentation-audit-round-2026-08-31.json#consumer-01"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 1,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "documentationExampleRate": 0.4296148738,
        "cachedInputTokens": 111360,
        "reasoningOutputTokens": 554,
        "stderrBytes": 0,
        "providerTokenCostUnits": 173194
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "714aa84ba8c5cad18126f4be8f72344caf9a7786c6d9d7f71c972f4978d1d340",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:01:02.174Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-05-implementation",
        "repositoryId": "consumer-05",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 63031,
        "responseBytes": 515,
        "stderrBytes": 0,
        "stdoutHash": "72f4dd7cf669374533e8ecbd70ce77e1a5c65ed3a5ce2a187264998434833122",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 294907,
        "outputTokens": 2390,
        "tokenMethod": "provider",
        "toolCalls": 8
      },
      "contextBytes": 2094,
      "evidenceIds": [
        "patch-evidence:docs/for-agents/registry-discovery.md:checks-omit-npm-run-docs:bridge:index",
        "verification-plan:ak-docs-check---json:blocked-by-workflow-lock"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 262912,
        "reasoningOutputTokens": 1118,
        "stderrBytes": 0,
        "providerTokenCostUnits": 297297
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "ac6992034567d8388f727f2ebb22943e0aa107dcc821161d423bcc3081bad0d5",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:02:13.480Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-05-implementation",
        "repositoryId": "consumer-05",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 70916,
        "responseBytes": 775,
        "stderrBytes": 0,
        "stdoutHash": "5423b51d9aafb796b8abe89b700e40e2a4a6d27013b6c4af82d6b36c9c4af326",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 410528,
        "outputTokens": 2620,
        "tokenMethod": "provider",
        "toolCalls": 9,
        "errorCode": "token-budget"
      },
      "contextBytes": 2085,
      "evidenceIds": [
        "patch-evidence:registry/research/README.md:documented-configuration-options-do-not-cover-memory-retriever-delegates-onConfirm-observers-maxSteps-present-in-registry/research/agent.ts",
        "verification-plan:after-approval-run-ak-docs-check---json-and-review-registry/research/README.md-against-registry/research/agent.ts",
        "acceptance-check:ak-docs-check---json-not-run-ak-docs-command-unavailable"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "documentationFindingCount": 1,
        "documentationExampleRate": 0.16145833333333334,
        "cachedInputTokens": 363008,
        "reasoningOutputTokens": 1286,
        "stderrBytes": 0,
        "providerTokenCostUnits": 413148
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "blocked",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "8b52126e1f573ebaec7f9efc72476034ca0e4474b90a1743282432847d1972d2",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:02:48.063Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-04-documentation",
        "repositoryId": "consumer-04",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 34529,
        "responseBytes": 807,
        "stderrBytes": 0,
        "stdoutHash": "34d3b1054d1f86d3051791acd6ff12e11cfa407ca0a4f1cd4437e60f0660a1b8",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 135706,
        "outputTokens": 1214,
        "tokenMethod": "provider",
        "toolCalls": 3
      },
      "contextBytes": 2057,
      "evidenceIds": [
        "evidence-c330436b9efbb186c30acd6dea384e10",
        "review-limitation: semantic confidence is medium because source/package inspection confirms the command is absent, but the required ak-docs audit could not run: ak-docs was unavailable in the environment; no documentation edits were made.",
        "recommended-next-action: replace the stale command with the repository's current equivalent or add the missing script, then rerun ak-docs audit documentation --json."
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 92160,
        "reasoningOutputTokens": 465,
        "stderrBytes": 0,
        "providerTokenCostUnits": 136920
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "8aa746998e11d139303694667fdcdef77845a5d985058d5f1b88d8494cfd050b",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:04:33.001Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-06-architecture",
        "repositoryId": "consumer-06",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 104393,
        "responseBytes": 574,
        "stderrBytes": 0,
        "stdoutHash": "a2cdfbbfb78291606ba0a5838128f74c308111e465a337cdd2e6c6aa5b884d01",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 195718,
        "outputTokens": 915,
        "tokenMethod": "provider",
        "toolCalls": 5
      },
      "contextBytes": 1883,
      "evidenceIds": [
        "artifact:package.json",
        "artifact:pnpm-workspace.yaml",
        "artifact:docs/internal/README.md",
        "artifact:docs/for-agents/INDEX.md",
        "command:ak-docs-map-json:exit-127",
        "command:node-modules-ak-docs-map-json:exit-2-eperm-lock"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 163328,
        "reasoningOutputTokens": 264,
        "stderrBytes": 0,
        "providerTokenCostUnits": 196633
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "81659a162bc6f45343b92cbf72337fc6aa9e9111bba494b6a7db955adc37fac2",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:05:00.182Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-05-discovery",
        "repositoryId": "consumer-05",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 27057,
        "responseBytes": 388,
        "stderrBytes": 0,
        "stdoutHash": "98276ee93cf4e084048d900600dbb35d81e6d21076e8d951cf523d1cf87482d7",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 65300,
        "outputTokens": 516,
        "tokenMethod": "provider",
        "toolCalls": 2
      },
      "contextBytes": 1864,
      "evidenceIds": [
        "entrypoint-evidence"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 56576,
        "reasoningOutputTokens": 167,
        "stderrBytes": 0,
        "providerTokenCostUnits": 65816
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "b1af0fd9f28f0d8e8f4b9260bbbbfb9df5a0397ddd967d94acc51e3a92c80ceb",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:06:04.573Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-04-documentation",
        "repositoryId": "consumer-04",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 64355,
        "responseBytes": 1058,
        "stderrBytes": 0,
        "stdoutHash": "1ff62a2faa8b0e840043dc2fefca59a7808603bf2d72ab8b4922f0b128604872",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 179892,
        "outputTokens": 1323,
        "tokenMethod": "provider",
        "toolCalls": 6
      },
      "contextBytes": 2056,
      "evidenceIds": [
        "evidence-a12353e803628852bcf38f4a4ebb0751",
        "documentation-evidence:current-source:scripts/readme-standard.test.mjs:25 derives the guide count independently, while public/deterministic/knowledge.json supplies local-answer data",
        "review-limitation:high-confidence textual mismatch; not classified as numeric staleness because the displayed values were not disproved",
        "recommended-next-action:clarify README.md by mapping each displayed count to its actual source artifact",
        "acceptance-check:blocked:ak-docs audit documentation --json is unsupported by installed @agentskit/doc-bridge 1.7.45 and emitted CLI help instead of measurable findings and limitations"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 1,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 145920,
        "reasoningOutputTokens": 455,
        "stderrBytes": 0,
        "providerTokenCostUnits": 181215
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "a19633241816e7815c6fa2437cdddd0179cab7ae65b89168d68c216a090b0d20",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:06:42.732Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-03-documentation",
        "repositoryId": "consumer-03",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 38121,
        "responseBytes": 867,
        "stderrBytes": 0,
        "stdoutHash": "c49650493fc06f99e8d210d7e6811688dd9fe249ed632018f1fba4023b0c372a",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 103764,
        "outputTokens": 757,
        "tokenMethod": "provider",
        "toolCalls": 3
      },
      "contextBytes": 2047,
      "evidenceIds": [
        "documentation-evidence:stale-version-claim:docs/for-agents/packages/svelte.md:3 claims @agentskit/svelte@0.4.3; packages/svelte/package.json:39,43 declares ^0.4.4; recommended-next-action=update the handoff claim to the current supported dependency version",
        "review-limitation:high-confidence repository comparison, but ak-docs audit documentation --json was unavailable (command not found), so generated-index consistency and the required audit could not be verified"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 80896,
        "reasoningOutputTokens": 248,
        "stderrBytes": 0,
        "providerTokenCostUnits": 104521
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "e1b19d318dc14d5bb5fa041dc6be5579611593b9494ac755865eebbed6e5f420",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:07:43.174Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-02-documentation",
        "repositoryId": "consumer-02",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 60405,
        "responseBytes": 746,
        "stderrBytes": 0,
        "stdoutHash": "72b3aaae74ae8a0614fa5e0d38a78ee1509b069e7d145f3203f488edbc92c203",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 291289,
        "outputTokens": 2619,
        "tokenMethod": "provider",
        "toolCalls": 7
      },
      "contextBytes": 2048,
      "evidenceIds": [
        "documentation-evidence:apps/docs-next/README.md:59",
        "documentation-evidence:apps/docs-next/content/docs/get-started/concepts:1-10",
        "documentation-evidence:.doc-bridge/index.json:generatedAt-2026-08-30",
        "review-limitation:ak-docs-audit-unavailable-and-verification-contract-missing",
        "recommended-next-action:update-stale-docs-next-readme-and-regenerate-index"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 1,
      "reworkCount": 0,
      "measurements": {
        "documentationFindingCount": 1,
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 250880,
        "reasoningOutputTokens": 1250,
        "stderrBytes": 0,
        "providerTokenCostUnits": 293908
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "04cebd916788d29fbaf8199c575423f467c46b7db949024c83e952ced4c883fb",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:08:19.644Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-02-discovery",
        "repositoryId": "consumer-02",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 36430,
        "responseBytes": 541,
        "stderrBytes": 0,
        "stdoutHash": "7e9ba9477919adf177e006e3e567c87403936a045eceb60ce51da0ba6c2435af",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 168880,
        "outputTokens": 1299,
        "tokenMethod": "provider",
        "toolCalls": 5
      },
      "contextBytes": 1865,
      "evidenceIds": [
        "entrypoint-evidence",
        "README.md",
        "doc-bridge.config.json",
        "apps/docs-next/README.md",
        "apps/docs-next/content/docs/for-agents/index.mdx",
        "packages/*/package.json",
        "packages/*/README.md"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 127744,
        "reasoningOutputTokens": 643,
        "stderrBytes": 0,
        "providerTokenCostUnits": 170179
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "2a19d0117a04ef280b0685be7b58282d0a501f7525956750c06f184dcfe6169d",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:09:29.984Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-04-implementation",
        "repositoryId": "consumer-04",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 70303,
        "responseBytes": 652,
        "stderrBytes": 0,
        "stdoutHash": "fa3f1275d5b6f8547fba6f8ee6dead3a5c095b7b4791c4c0e506920100ce8f44",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 479428,
        "outputTokens": 2445,
        "tokenMethod": "provider",
        "toolCalls": 12,
        "errorCode": "token-budget"
      },
      "contextBytes": 2094,
      "evidenceIds": [
        "patch-evidence: stale Doc Bridge reconciliation reports 2 undocumented packages but does not identify their package IDs or knowledge gap; ownership query routes playbook documentation to content/docs/index.mdx",
        "verification-plan: run ak-docs check --json after approval",
        "acceptance-check: ak-docs check --json failed because ak-docs is unavailable"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "low",
      "safetyOutcome": "safe",
      "clarificationRequests": 1,
      "reworkCount": 0,
      "measurements": {
        "cachedInputTokens": 409600,
        "reasoningOutputTokens": 893,
        "stderrBytes": 0,
        "providerTokenCostUnits": 481873
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "blocked",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "0d66f30308d49ca28094dcfb77d9f65d1d001d43173d8c0a629a17a95184e734",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:11:27.916Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-01-documentation",
        "repositoryId": "consumer-01",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 117892,
        "responseBytes": 939,
        "stderrBytes": 0,
        "stdoutHash": "da7227b68de9e1dad5e8170663fe8ab9b2eb65feeefb162ce33e2feb4834edf7",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 901212,
        "outputTokens": 3809,
        "tokenMethod": "provider",
        "toolCalls": 15,
        "errorCode": "token-budget"
      },
      "contextBytes": 2048,
      "evidenceIds": [
        "stale-claim:README.md:119,258 documents AgentsKit-io/doc-bridge@v1.4.0 while action.yml:20-23 defines package-version default 1.7.45; confidence=high; next-action=update documented action tag or explicitly explain intentional pin",
        "generated-index:.doc-bridge/index.json:3-5 was generated 2026-08-29 and may not reflect current modified source/docs",
        "review-limitation:semantic documentation correctness remains human-review judgment; required audit command could not execute because ak-docs was unavailable and local CLI attempted a prohibited lock-directory write"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 824064,
        "reasoningOutputTokens": 1545,
        "stderrBytes": 0,
        "providerTokenCostUnits": 905021
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "blocked",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "26d8c84b9cd72cd9e088b0090674c33df6d08ebbe927eaaddf19b627a6827c5a",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:12:28.207Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-03-implementation",
        "repositoryId": "consumer-03",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 60248,
        "responseBytes": 338,
        "stderrBytes": 0,
        "stdoutHash": "079ef8bff26b561f361c334d2b1c77addedd2583a1b2230da9d8564b4f7a64f7",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 553428,
        "outputTokens": 2194,
        "tokenMethod": "provider",
        "toolCalls": 11,
        "errorCode": "token-budget"
      },
      "contextBytes": 2085,
      "evidenceIds": [
        "patch-evidence",
        "verification-plan"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "success",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "cachedInputTokens": 495616,
        "reasoningOutputTokens": 656,
        "stderrBytes": 0,
        "providerTokenCostUnits": 555622
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "blocked",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "562790138da8ab08c312e938097cfb1021106e16c10fa8c701f594c9e1ce5738",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:13:24.342Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-04-architecture",
        "repositoryId": "consumer-04",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 56089,
        "responseBytes": 688,
        "stderrBytes": 0,
        "stdoutHash": "ada691fa132e5eae63c76d2758c8b17433e11cf8490e874e0fc695aa8ee42628",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 159662,
        "outputTokens": 1045,
        "tokenMethod": "provider",
        "toolCalls": 5
      },
      "contextBytes": 1892,
      "evidenceIds": [
        "architecture-evidence:package.json",
        "architecture-evidence:pnpm-workspace.yaml",
        "architecture-evidence:next.config.ts",
        "architecture-evidence:app/layout.tsx",
        "architecture-evidence:components/ask-widget.tsx",
        "architecture-evidence:lib/discovery.ts",
        "architecture-check:ak-docs-map-EPERM"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 125696,
        "reasoningOutputTokens": 389,
        "stderrBytes": 0,
        "providerTokenCostUnits": 160707
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "5beed89b2a810459f05c71331c9b72f3805f3657d37971b31119a596eb52c7a6",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:14:17.330Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-03-implementation",
        "repositoryId": "consumer-03",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 52949,
        "responseBytes": 577,
        "stderrBytes": 0,
        "stdoutHash": "fb223521cbca8bd109ca07b51a107569b78079e1d443a28d97c1815f461cd717",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 364714,
        "outputTokens": 2228,
        "tokenMethod": "provider",
        "toolCalls": 8
      },
      "contextBytes": 2094,
      "evidenceIds": [
        "patch-evidence:.doc-bridge/workflow/artifacts/report-ff5e69fbcfef7041594065be162c4fb4bdb12656f739e1c2cef116767cfade0c:findings-empty",
        "verification-plan:ak-docs-check-json:command-not-found"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 1,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "documentationFindingCount": 0,
        "cachedInputTokens": 287744,
        "reasoningOutputTokens": 788,
        "stderrBytes": 0,
        "providerTokenCostUnits": 366942
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "10f81a3f0f8d97df26954d65e5865943f663ff8f62c385baf16825e0eb1bd7ad",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:14:54.122Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-03-architecture",
        "repositoryId": "consumer-03",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 36747,
        "responseBytes": 737,
        "stderrBytes": 0,
        "stdoutHash": "d642e6c1030a6f8724d47cf3a91d0f734002ad521a7f90acff165caf56269aa6",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 98540,
        "outputTokens": 718,
        "tokenMethod": "provider",
        "toolCalls": 3
      },
      "contextBytes": 1883,
      "evidenceIds": [
        "architecture-evidence:pnpm-workspace.yaml",
        "architecture-evidence:docs/architecture/overview.md",
        "architecture-evidence:packages/*/package.json",
        "architecture-evidence:packages/*/src imports",
        "attention-boundary:@agentskit/chat-to-@agentskit/chat-protocol dependency-direction",
        "acceptance-evidence:ak-docs-map-command-not-found"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "not-applicable",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 63744,
        "reasoningOutputTokens": 188,
        "stderrBytes": 0,
        "providerTokenCostUnits": 99258
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "705306e40772be1e638275a8aa67af28b807d2a543bb4e8afe8acf0b0ae04a37",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:15:44.578Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-04-architecture",
        "repositoryId": "consumer-04",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 50414,
        "responseBytes": 1406,
        "stderrBytes": 0,
        "stdoutHash": "b3a0be84a56ab46edd06398e3bf33edebbbb81fabae287ed72b6b029bbb5037e",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 100654,
        "outputTokens": 961,
        "tokenMethod": "provider",
        "toolCalls": 3
      },
      "contextBytes": 1883,
      "evidenceIds": [
        "package.json: workspace root connects Next.js/Fumadocs host, AgentsKit dependencies, @agentskit/harness workspace dependency, and doc-bridge scripts",
        "pnpm-workspace.yaml: packages/* defines the workspace boundary",
        "source.config.ts: Fumadocs MDX consumes content/docs",
        "lib/source.ts: generated MDX source flows through fumadocs-core loader",
        "app/docs/[[...slug]]/page.tsx: docs route consumes lib/source and Fumadocs UI",
        "app/api/search/route.ts: search endpoint consumes the same lib/source",
        "app/layout.tsx: host composes Fumadocs provider with AskWidget and shared ecosystem UI",
        "lib/discovery.ts: discovery adapter depends directionally on @agentskit/core and @agentskit/chat",
        "packages/playbook/package.json: independent zero-dependency quality-gate CLI package",
        "packages/harness/package.json: independent verification harness package exposed as ak-harness and ak-verify",
        "evidence-1b5a490ad11ff397eba6bf7f42332f2c",
        "acceptance-check: ak-docs map --json exited 127 because ak-docs was not available"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 64768,
        "reasoningOutputTokens": 215,
        "stderrBytes": 0,
        "providerTokenCostUnits": 101615
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "9977ceca6f89c027dfcbb07f44d3bb23caf1d98e10a63e8b8e6a22a3fe19e76d",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:16:51.615Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-01-documentation",
        "repositoryId": "consumer-01",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 66991,
        "responseBytes": 767,
        "stderrBytes": 0,
        "stdoutHash": "82d7c8fe9ffe1bacdaa0759a4c4ed16b1440c1619228057db83f1943ad0669cc",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 350132,
        "outputTokens": 2180,
        "tokenMethod": "provider",
        "toolCalls": 10
      },
      "contextBytes": 2057,
      "evidenceIds": [
        "evidence-c2e0ea7b8f353e8d058f3c0e814c69d7",
        "review-limitation: Direct source and generated-manifest comparison gives high confidence, but ak-docs audit documentation --json was blocked because the read-only workspace could not create .doc-bridge/workflow/.lock.",
        "recommended-next-action: Confirm the intended public bundle surface, then update the README count and rerun the documentation audit."
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 307456,
        "reasoningOutputTokens": 818,
        "stderrBytes": 0,
        "providerTokenCostUnits": 352312
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "ac79986398b2ee194485ea299a724f2e87180e3618ab00d3924284ea2ee55a7c",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:18:15.274Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-04-implementation",
        "repositoryId": "consumer-04",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 83600,
        "responseBytes": 666,
        "stderrBytes": 0,
        "stdoutHash": "4806bb6119b4b552c221cfc7d8ef50cf9ca0d20135fe6b0d747bf8fbac390511",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 221423,
        "outputTokens": 1702,
        "tokenMethod": "provider",
        "toolCalls": 7
      },
      "contextBytes": 2093,
      "evidenceIds": [
        "patch-evidence:content/docs/index.mdx:for-agents-section-omits-deterministic-doc-bridge-routing-workflow",
        "patch-evidence:doc-bridge-query:ownership-playbook:edit-root-content/docs/index.mdx",
        "verification-plan:ak-docs-check-json:blocked-by-read-only-lock-creation"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 193536,
        "reasoningOutputTokens": 652,
        "stderrBytes": 0,
        "providerTokenCostUnits": 223125
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "1a2528aad06734c006d8cb77ae0ecee33de72da367edb8337d0a63bb0578261f",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:19:41.738Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-04-implementation",
        "repositoryId": "consumer-04",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 86406,
        "responseBytes": 558,
        "stderrBytes": 0,
        "stdoutHash": "9d81faa1ba475f6b22f30458beea1378974db89085cccf8cd01c0ad7bf13094d",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 245733,
        "outputTokens": 1621,
        "tokenMethod": "provider",
        "toolCalls": 8
      },
      "contextBytes": 2084,
      "evidenceIds": [
        "patch-evidence:content/docs/index.mdx:frontmatter-covers-package:@agentskit/playbook",
        "verification-plan:ak-docs-check-json:blocked-by-read-only-lock-creation"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 200704,
        "reasoningOutputTokens": 678,
        "stderrBytes": 0,
        "providerTokenCostUnits": 247354
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "920d9ebaa0cf96e18bd9b1a42410272c9b61a408e8c7caf35b05c35a4f16cbb2",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:21:41.607Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-06-architecture",
        "repositoryId": "consumer-06",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 119140,
        "responseBytes": 897,
        "stderrBytes": 0,
        "stdoutHash": "c444a1fa6cf83e0a53c166a6919842e9895a160cd701228770c478748d6517b7",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 475662,
        "outputTokens": 2653,
        "tokenMethod": "provider",
        "toolCalls": 10,
        "errorCode": "token-budget"
      },
      "contextBytes": 1884,
      "evidenceIds": [
        "docs/architecture/layers.md:L14-L51",
        "docs/for-agents/architecture.md:docbridge.relations",
        "docs/enterprise-final/architecture-overview.md:L67-L166",
        "pnpm-workspace.yaml:packages-and-apps",
        "packages/os-headless/package.json:L79-L104",
        "packages/os-runtime/package.json:L51-L52",
        "packages/os-flow/package.json:L40",
        "apps/console/package.json:L17-L18",
        "apps/desktop/package.json:L31-L43",
        "packages/os-sandbox/package.json:L40",
        "attention:os-sandbox-mixed-port-and-runtime-boundary",
        "architecture:os-core->domain->bindings->composition-roots"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 426240,
        "reasoningOutputTokens": 1235,
        "stderrBytes": 0,
        "providerTokenCostUnits": 478315
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "blocked",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "17d77dbb3fb6c9f5c463b8cba6d925977eb22ca54938e35908b85397b73e0d68",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:22:24.323Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-01-discovery",
        "repositoryId": "consumer-01",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 42656,
        "responseBytes": 1085,
        "stderrBytes": 0,
        "stdoutHash": "21cad30dfd71857a5197b91f820d9dd5f0a578e1b67c3ebc2121641afacf9b47",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 166690,
        "outputTokens": 1536,
        "tokenMethod": "provider",
        "toolCalls": 5
      },
      "contextBytes": 1865,
      "evidenceIds": [
        "package.json:bin.ak-docs=bin/ak-docs.js;main=dist/index.js;author=AgentsKit",
        "bin/ak-docs.js:CLI entrypoint delegates to dist/cli/program.js",
        "src/index.ts:public library exports discoverRepository and core APIs",
        "doc-bridge.config.json:routing ownership boundaries for doc-bridge, cli, index, query, mcp, gates, conformance, doctor, memory, and chat",
        "docs/index.md:canonical documentation map",
        "docs/guides/index-and-query.md:canonical ownership and indexing workflow",
        "docs/spec/cli.md:canonical CLI reference",
        "docs/mcp.md:canonical MCP documentation",
        "docs/agent-corpus/OVERVIEW.md:canonical agent documentation index",
        "discovery-snapshot:ok=true;contentHash=58d5239f0de4d2a27cc4530b56e0e7c2d334fc5272ae4ded4f4d6f795da7de3"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 135680,
        "reasoningOutputTokens": 594,
        "stderrBytes": 0,
        "providerTokenCostUnits": 168226
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "5c2d84a182be654d3a57e41e43d71d5d3bd0ebe32a94b1530373e1c0085774e7",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:23:28.645Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-05-architecture",
        "repositoryId": "consumer-05",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 64268,
        "responseBytes": 660,
        "stderrBytes": 0,
        "stdoutHash": "89006f93bc92f00150a7c70b1665ea9f0047b28fcad2fe317debc0a85a4b67d9",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 313060,
        "outputTokens": 2474,
        "tokenMethod": "provider",
        "toolCalls": 11
      },
      "contextBytes": 1893,
      "evidenceIds": [
        "architecture-evidence:docs/architecture.md",
        "architecture-evidence:docs/for-agents/index.md",
        "architecture-evidence:package.json",
        "architecture-evidence:.doc-bridge/workflow/artifacts/normalize-015404de24fb9d344a33e4eefdf6cb9e2ae1136213ccd427bad8df1ecb8581a.json",
        "architecture-check:ak-docs-map:EPERM"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 247808,
        "reasoningOutputTokens": 985,
        "stderrBytes": 0,
        "providerTokenCostUnits": 315534
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "206c62ec351a244c9c530cb7d37c0f3de309a97330b2c8d08c870bc58593c53a",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:24:10.838Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-04-discovery",
        "repositoryId": "consumer-04",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 41910,
        "responseBytes": 388,
        "stderrBytes": 0,
        "stdoutHash": "0be42989f0b0f67bb7941ee242715d5e5e9dd9c538b4f4833cc7a2693a1d1c46",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 70005,
        "outputTokens": 530,
        "tokenMethod": "provider",
        "toolCalls": 2
      },
      "contextBytes": 1864,
      "evidenceIds": [
        "entrypoint-evidence"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 50432,
        "reasoningOutputTokens": 146,
        "stderrBytes": 0,
        "providerTokenCostUnits": 70535
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "51c75e318b4f08290e15c1741f1488ef44949438312ff0cb206df5848da661c1",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:24:49.697Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-06-discovery",
        "repositoryId": "consumer-06",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 38581,
        "responseBytes": 629,
        "stderrBytes": 0,
        "stdoutHash": "7a85bc1caa65496c8ad587c54d40eaa4a8c3c22dc3d0b66a1018ccafabeed76e",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 119906,
        "outputTokens": 694,
        "tokenMethod": "provider",
        "toolCalls": 3
      },
      "contextBytes": 1855,
      "evidenceIds": [
        "entrypoint-evidence:package.json",
        "entrypoint-evidence:pnpm-workspace.yaml",
        "entrypoint-evidence:README.md",
        "entrypoint-evidence:AGENTS.md",
        "entrypoint-evidence:docs/for-agents/INDEX.md",
        "discovery-check:ak-docs-command-not-found"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 5,
        "cachedInputTokens": 79104,
        "reasoningOutputTokens": 218,
        "stderrBytes": 0,
        "providerTokenCostUnits": 120600
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "b83b3e5a1b5befe9b5d35c9b36e18e7f8215c1543a859dd231bc4ca03beba063",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:26:41.749Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-05-documentation",
        "repositoryId": "consumer-05",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 111630,
        "responseBytes": 479,
        "stderrBytes": 0,
        "stdoutHash": "a0c51378d21c4ac79aafe08cea461f8464840566393b0f38363c1bc7b0f5f5b6",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 729403,
        "outputTokens": 4236,
        "tokenMethod": "provider",
        "toolCalls": 13,
        "errorCode": "token-budget"
      },
      "contextBytes": 2048,
      "evidenceIds": [
        "documentation-evidence",
        "review-limitation"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "documentationFindingCount": 1,
        "documentationExampleRate": 0.16145833333333334,
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 639744,
        "reasoningOutputTokens": 1562,
        "stderrBytes": 0,
        "providerTokenCostUnits": 733639
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "blocked",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "d949b02980b014ee936dfa2204798ad98e4f47616d1b6574918a9cca1fd8701b",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:28:06.184Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-05-discovery",
        "repositoryId": "consumer-05",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 84359,
        "responseBytes": 406,
        "stderrBytes": 0,
        "stdoutHash": "71af92147ca6d0182a4d25f31971f4b270b5a718046ac9b4fbfabba9b633db4e",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 373205,
        "outputTokens": 2229,
        "tokenMethod": "provider",
        "toolCalls": 10
      },
      "contextBytes": 1856,
      "evidenceIds": [
        "entrypoint-evidence",
        "npx-ak-docs-discover-json"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 336128,
        "reasoningOutputTokens": 927,
        "stderrBytes": 0,
        "providerTokenCostUnits": 375434
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "4cc3dc4b4453eae3f71d95e358f93a3df86f7074be1e51b8c7160dfa1fd6086d",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:30:04.062Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-04-architecture",
        "repositoryId": "consumer-04",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 117826,
        "responseBytes": 1205,
        "stderrBytes": 0,
        "stdoutHash": "60002c53ad1d2fcae8c68d66d7fceec7b719635673d2ede08d223cd28d2913e8",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 865904,
        "outputTokens": 4327,
        "tokenMethod": "provider",
        "toolCalls": 15,
        "errorCode": "token-budget"
      },
      "contextBytes": 1884,
      "evidenceIds": [
        "architecture:agents-playbook-root->Fumadocs/docs-and-API-routes->deterministic-artifact->Ask-Playbook",
        "architecture:packages/harness->portable-verification-library-and-ak-harness-cli",
        "architecture:packages/playbook->zero-dependency-agents-playbook-cli-and-quality-gates",
        "relation:components/ask-widget.tsx->@agentskit/chat,@agentskit/chat/react,@agentskit/core,@agentskit/react,lib/discovery.ts",
        "relation:lib/source.ts->.source/index.ts->content/docs via Fumadocs",
        "attention:review-components/ask-widget.tsx-as-the-host/upstream-chat-boundary; docs require @agentskit/chat ownership of state/streaming/memory/cancellation and forbid reimplementation",
        "artifact:.doc-bridge/workflow/manifest.json sourceRevision=e6079d4ea8f730012ddb406f14f597112f5dcf5e entities=374 relations=637",
        "acceptance:ak-docs-map-json-blocked-by-read-only-lock/EPERM"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 780032,
        "reasoningOutputTokens": 1370,
        "stderrBytes": 0,
        "providerTokenCostUnits": 870231
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "blocked",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "091e60d5a12ffc80b57790aa0150e00b2f009a665192330f19ab598d19a4d514",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:30:33.143Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-04-discovery",
        "repositoryId": "consumer-04",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 29030,
        "responseBytes": 386,
        "stderrBytes": 0,
        "stdoutHash": "118d368b6fba243b9e916e1ed9abd0abd5f22a281f5e352e174cf60326d6a289",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 73922,
        "outputTokens": 603,
        "tokenMethod": "provider",
        "toolCalls": 2
      },
      "contextBytes": 1855,
      "evidenceIds": [
        "entrypoint-evidence"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 62720,
        "reasoningOutputTokens": 177,
        "stderrBytes": 0,
        "providerTokenCostUnits": 74525
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "0150f0d11395490fb4395e40205ac63fe3955267335571031a9b5ff15f123a70",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:31:31.740Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-04-discovery",
        "repositoryId": "consumer-04",
        "category": "discovery",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 58550,
        "responseBytes": 694,
        "stderrBytes": 0,
        "stdoutHash": "be4b0e921eeb6f47d84094d7d569c5276f0a4003e41fe8493b879fc3b000d13e",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 215651,
        "outputTokens": 2290,
        "tokenMethod": "provider",
        "toolCalls": 5
      },
      "contextBytes": 1856,
      "evidenceIds": [
        "entrypoint-evidence",
        "AGENTS.md",
        "README.md",
        "doc-bridge.config.json",
        "docs/for-agents/INDEX.md",
        "content/docs/index.mdx",
        "content/docs/for-agents.mdx",
        "app/docs/[[...slug]]/page.tsx",
        "app/raw/[...path]/route.ts",
        "app/api/search/route.ts",
        "components/ask-widget.tsx",
        "lib/discovery.ts",
        "lib/source.ts",
        "source.config.ts",
        "app/layout.tsx"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 159488,
        "reasoningOutputTokens": 886,
        "stderrBytes": 0,
        "providerTokenCostUnits": 217941
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "604a4bf94c71b497e374dedae2fa0c30102bf14fb02414003fc50daa9804efd0",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:32:35.436Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-03-documentation",
        "repositoryId": "consumer-03",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 63650,
        "responseBytes": 802,
        "stderrBytes": 0,
        "stdoutHash": "70cee98da59acf963bb2ab2fa903a1011c53792302a3aed53928bb4fad685bff",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 328626,
        "outputTokens": 2517,
        "tokenMethod": "provider",
        "toolCalls": 8
      },
      "contextBytes": 2057,
      "evidenceIds": [
        "documentation-evidence:docs/guides/install-and-run.mdx:11 claims @agentskit/chat-cli@0.4.0; package.json:3 and docs/releases/migration-to-0.3.md:7-8 establish current stable 0.4.1; classify stale; recommend update to 0.4.1",
        "review-limitation:semantic judgment is high-confidence from direct source comparison, but the required audit could not run because ak-docs is unavailable and configured execution hit EPERM"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "documentationFindingCount": 1,
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 283392,
        "reasoningOutputTokens": 1034,
        "stderrBytes": 0,
        "providerTokenCostUnits": 331143
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "646a6fb45617236720fa22527ad11041e704aa2e2d6751abfeb38b7ac2bf03fa",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:33:15.856Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-03-architecture",
        "repositoryId": "consumer-03",
        "category": "architecture",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 40372,
        "responseBytes": 822,
        "stderrBytes": 0,
        "stdoutHash": "f3f8fdde18ca76051ca68ce557e80a91e202cb7f8aad45a3352f6885dde18241",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 198522,
        "outputTokens": 1562,
        "tokenMethod": "provider",
        "toolCalls": 5
      },
      "contextBytes": 1884,
      "evidenceIds": [
        "docs/architecture/overview.md",
        "doc-bridge.config.json",
        "packages/chat/package.json",
        "packages/server/src/index.ts",
        "packages/server/src/ask-service.ts",
        "packages/react/src/index.tsx",
        "packages/vue/src/index.ts",
        "packages/svelte/src/index.ts",
        "packages/solid/src/index.tsx",
        "packages/angular/src/index.ts",
        "packages/react-native/src/index.tsx",
        "packages/ink/src/index.tsx",
        "boundary:server-ask-injected-retriever-generator",
        "ak-docs-map:unavailable-EPERM"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "medium",
      "safetyOutcome": "not-applicable",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 162304,
        "reasoningOutputTokens": 689,
        "stderrBytes": 0,
        "providerTokenCostUnits": 200084
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "ff1225c1ce826cd8b51716840dee52bc743f66d5098a11ec8c3fa55b8700489c",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:34:16.027Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-03-documentation",
        "repositoryId": "consumer-03",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 60131,
        "responseBytes": 682,
        "stderrBytes": 0,
        "stdoutHash": "6631a997702911179f80fe9ab802d1d8ed9f8920378e1b3e9c992da82c10abaa",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 163396,
        "outputTokens": 1245,
        "tokenMethod": "provider",
        "toolCalls": 5
      },
      "contextBytes": 2056,
      "evidenceIds": [
        "documentation-evidence:stale-contradictory-svelte-version:docs/architecture/upstream-adoption.md:44 claims @agentskit/svelte 0.3.1; packages/svelte/package.json:39-43 requires ^0.4.4; generated .doc-bridge/index.json:184-185 repeats 0.4.3",
        "evidence-37a62a7470217231835de9598324a6c4"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 118528,
        "reasoningOutputTokens": 445,
        "stderrBytes": 0,
        "providerTokenCostUnits": 164641
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "a006121b669f397fea457b296d2ba91a87bce9621d9974d17362551cfe2f8e1b",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:34:52.853Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-03-architecture",
        "repositoryId": "consumer-03",
        "category": "architecture",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "medium"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 36783,
        "responseBytes": 797,
        "stderrBytes": 0,
        "stdoutHash": "43b76171f0168685df90bb4f359ba8036c3a62f694983506e1aeca69fb641997",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 98386,
        "outputTokens": 709,
        "tokenMethod": "provider",
        "toolCalls": 3
      },
      "contextBytes": 1892,
      "evidenceIds": [
        "architecture-evidence:docs/for-agents/architecture.md:static-relations",
        "architecture-evidence:docs/architecture/overview.md:containers-and-reference-turn-pipeline",
        "architecture-evidence:packages/*/package.json:workspace-and-peer-dependency-direction",
        "attention-point:packages/chat/package.json-framework-peer-dependency-boundary-deserves-review",
        "blocked-acceptance:ak-docs-map-command-not-found"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 65792,
        "reasoningOutputTokens": 189,
        "stderrBytes": 0,
        "providerTokenCostUnits": 99095
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "84927e240e962f646beeb35057f921a82f090a9455d2030ff4b9db6dd9adf4e4",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:36:12.667Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-02-implementation",
        "repositoryId": "consumer-02",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 79766,
        "responseBytes": 789,
        "stderrBytes": 0,
        "stdoutHash": "1046d9765cf3932e07811be3ec50bd519731aac3e23703d47041a1f4efb4556b",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 257015,
        "outputTokens": 1593,
        "tokenMethod": "provider",
        "toolCalls": 8
      },
      "contextBytes": 2093,
      "evidenceIds": [
        "patch-evidence:apps/docs-next/content/docs/for-agents/doc-bridge.mdx:16 uses nonexistent pnpm script docs:bridge:query; package.json defines no such script; propose replacing it with pnpm exec ak-docs query package <package-id> --agent",
        "verification-plan:after approval run ak-docs check --json and confirm successful exit; current read-only run was blocked by EPERM creating .doc-bridge/workflow/.lock"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 215040,
        "reasoningOutputTokens": 443,
        "stderrBytes": 0,
        "providerTokenCostUnits": 258608
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "e9605fa3286d678f65c843f8f7938704dd9690f1e818625ba17c982e113a3d49",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:37:12.713Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-01-implementation",
        "repositoryId": "consumer-01",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 59996,
        "responseBytes": 989,
        "stderrBytes": 0,
        "stdoutHash": "4dc554387798337ac27865782d957f6f0fbcfe74323e3fa7a05c0efe3d848766",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 215647,
        "outputTokens": 1277,
        "tokenMethod": "provider",
        "toolCalls": 6
      },
      "contextBytes": 2093,
      "evidenceIds": [
        "patch-evidence:docs/spec/config-v1.md:836 incorrectly marks ak-docs chat as planned; src/cli/program.ts:1600-1613 and src/intelligence/chat.ts:75-95 show it is implemented",
        "proposal:documentation-only;replace \"planned; `intelligence.*`\" with \"`intelligence.*`\" in docs/spec/config-v1.md:836;preserve-table-conventions;no-source-changes",
        "verification-plan:after approval run ak-docs check --json and review the one-line documentation diff",
        "verification-blocked:acceptance command could not complete in the read-only workspace because it attempted to create .doc-bridge/workflow/.lock"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 183552,
        "reasoningOutputTokens": 224,
        "stderrBytes": 0,
        "providerTokenCostUnits": 216924
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "e517934f4f6907b17d2d53927db2b2c7d2688b243ad384b9dff61582be42d63a",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:37:38.018Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-02-discovery",
        "repositoryId": "consumer-02",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "easy"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 25174,
        "responseBytes": 385,
        "stderrBytes": 0,
        "stdoutHash": "75da997eef9cb5854560d530956fb77e8c84235756d0b40579392724a265db55",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 68450,
        "outputTokens": 472,
        "tokenMethod": "provider",
        "toolCalls": 1
      },
      "contextBytes": 1864,
      "evidenceIds": [
        "entrypoint-evidence"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "low",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 1,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "cachedInputTokens": 59648,
        "reasoningOutputTokens": 168,
        "stderrBytes": 0,
        "providerTokenCostUnits": 68922
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "c26e38a6c80af6957527cfe4b4f41489c46295797499e7cab31a42866794b0dc",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:38:51.695Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-02-documentation",
        "repositoryId": "consumer-02",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 73629,
        "responseBytes": 703,
        "stderrBytes": 0,
        "stdoutHash": "ef0104b02909923cf3c96c4132a876a8327a51f2feeb653720ff1d73ec29f7b5",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 216928,
        "outputTokens": 1518,
        "tokenMethod": "provider",
        "toolCalls": 7
      },
      "contextBytes": 2047,
      "evidenceIds": [
        "evidence-017cd94aa8cd4152945506db6de11c7e",
        "review-limitation:high-confidence repository-only comparison; external publication state was not verified.",
        "acceptance-check:blocked:ak-docs audit documentation --json is unsupported by the installed CLI and emitted help text rather than measurable findings."
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 170240,
        "reasoningOutputTokens": 553,
        "stderrBytes": 0,
        "providerTokenCostUnits": 218446
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "e590f1259afd30e8f973e3e361b56125098b04e89a607b3026c3f566a6b31e16",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:40:37.088Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-05-implementation",
        "repositoryId": "consumer-05",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 105346,
        "responseBytes": 670,
        "stderrBytes": 0,
        "stdoutHash": "aeb29a7687522a06599d29381b308754ef4fe9ae5e6b0b620bae37604b143890",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 252283,
        "outputTokens": 2043,
        "tokenMethod": "provider",
        "toolCalls": 9
      },
      "contextBytes": 2084,
      "evidenceIds": [
        "patch-evidence:docs/for-agents/index.md:add-link-to-docs/for-agents/registry-discovery.md",
        "patch-evidence:docs/for-agents/registry-discovery.md",
        "doc-bridge-query:registry-discovery:exit-0",
        "verification-plan:ak-docs-check-json",
        "verification-blocker:EPERM-.doc-bridge/workflow/.lock"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 216832,
        "reasoningOutputTokens": 697,
        "stderrBytes": 0,
        "providerTokenCostUnits": 254326
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "d2627de748e9a5950d6a8c573c3af8a8e53be9a2a446b7a9a675f0e2b3f2d055",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:41:52.645Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-06-implementation",
        "repositoryId": "consumer-06",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "budget-exceeded",
        "exitCode": 0,
        "signal": null,
        "durationMs": 75502,
        "responseBytes": 559,
        "stderrBytes": 0,
        "stdoutHash": "1627c91b013272e0cadcdff8a9f9e49444c5519328824d78564aa9cc8050d820",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 681605,
        "outputTokens": 2663,
        "tokenMethod": "provider",
        "toolCalls": 11,
        "errorCode": "token-budget"
      },
      "contextBytes": 2094,
      "evidenceIds": [
        "patch-evidence:docs/spec/config-v1.md:836-stale-planned-claim",
        "patch-evidence:src/intelligence/chat.ts:75-runChatOnce-implementation",
        "verification-plan:pnpm exec ak-docs check --json"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "searchHitRate": 1,
        "cachedInputTokens": 611072,
        "reasoningOutputTokens": 700,
        "stderrBytes": 0,
        "providerTokenCostUnits": 684268
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "blocked",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "6bd0acb118c5eccae4c6162518f32ba0e337d11b63737e8b790d947db5c5ac13",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:42:52.034Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-04-discovery",
        "repositoryId": "consumer-04",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 59277,
        "responseBytes": 756,
        "stderrBytes": 0,
        "stdoutHash": "182b784733f86b41d8af21dad58d7362e68123df207e5d34ceb400d2a19e8d86",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 302490,
        "outputTokens": 2001,
        "tokenMethod": "provider",
        "toolCalls": 8
      },
      "contextBytes": 1865,
      "evidenceIds": [
        "entrypoint-evidence:README.md,AGENTS.md,package.json,source.config.ts,next.config.ts,app/page.tsx,app/docs/layout.tsx,app/docs/[[...slug]]/page.tsx,app/raw/[...path]/route.ts,app/llms.txt/route.ts,app/llms-full.txt/route.ts",
        "canonical-docs:doc-bridge.config.json,content/docs/index.mdx,content/docs/for-agents.mdx,docs/for-agents/INDEX.md",
        "discovery-check:blocked-by-EPERM-and-ak-docs-unavailable"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 256768,
        "reasoningOutputTokens": 674,
        "stderrBytes": 0,
        "providerTokenCostUnits": 304491
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "59ae9b902c31da29cadd61a85072fd4d932dd1809f54e1b5e47c2e8b28a33669",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:44:06.868Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-02-documentation",
        "repositoryId": "consumer-02",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 74718,
        "responseBytes": 701,
        "stderrBytes": 0,
        "stdoutHash": "642b5cd48bb6053e1f9418f4e48338230e2595c3b9b5fc1154e66ad3b4bd0e0d",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 230522,
        "outputTokens": 1397,
        "tokenMethod": "provider",
        "toolCalls": 8
      },
      "contextBytes": 2056,
      "evidenceIds": [
        "evidence-6f82033b74b443a284741993ab4907b9",
        "review-limitation:Finding is limited to one repository-local claim comparison; no human semantic review was available",
        "acceptance-check-limitation:ak-docs audit documentation --json is unsupported by the installed CLI and emitted help instead of audit JSON"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 1,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 192512,
        "reasoningOutputTokens": 416,
        "stderrBytes": 0,
        "providerTokenCostUnits": 231919
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "544fd461af0e90f0ade35f846e0867bf8d40e1b9016ce2b2bb9da71a184c530c",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:45:06.757Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-04-documentation",
        "repositoryId": "consumer-04",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 59826,
        "responseBytes": 473,
        "stderrBytes": 0,
        "stdoutHash": "1774714dc6735c27a82b213c81e08a2b786066d60a1c06a5c28dae61606bb33e",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 179033,
        "outputTokens": 1287,
        "tokenMethod": "provider",
        "toolCalls": 5
      },
      "contextBytes": 2047,
      "evidenceIds": [
        "evidence-8c439b887fb90bd570b8ba0786dfc1f8",
        "evidence-da367f72413b349e7eb34447457fa564"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 134912,
        "reasoningOutputTokens": 351,
        "stderrBytes": 0,
        "providerTokenCostUnits": 180320
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "a596945e739c02a3a0b2093685c8a3ad7c7637be42f3fc04fd6b84c08efb62b0",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:46:28.249Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-06-documentation",
        "repositoryId": "consumer-06",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 81439,
        "responseBytes": 398,
        "stderrBytes": 0,
        "stdoutHash": "7d3c4868677540bb0751b77b2d83172aea03df8227ccc1f540da950b8c24391c",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 210923,
        "outputTokens": 2226,
        "tokenMethod": "provider",
        "toolCalls": 6
      },
      "contextBytes": 2057,
      "evidenceIds": [
        "documentation-evidence",
        "review-limitation"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 186624,
        "reasoningOutputTokens": 861,
        "stderrBytes": 0,
        "providerTokenCostUnits": 213149
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "0b4130c8f1109a1be721c859817df55d4393e1bc179b3dcb781aafd82d4e8c7c",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:47:05.926Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-03-discovery",
        "repositoryId": "consumer-03",
        "category": "discovery",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-a",
        "difficulty": "easy"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 37633,
        "responseBytes": 773,
        "stderrBytes": 0,
        "stdoutHash": "570bf2c0b021002c7ca7df9556fddad1b2669c0c127089fb1cab075374944f66",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 178140,
        "outputTokens": 1451,
        "tokenMethod": "provider",
        "toolCalls": 5
      },
      "contextBytes": 1865,
      "evidenceIds": [
        "entrypoint-evidence",
        "README.md",
        "docs/for-agents/index.md",
        "docs/architecture/overview.md",
        "docs/architecture/upstream-adoption.md",
        "docs/for-agents/packages/chat.md",
        "docs/for-agents/packages/protocol.md",
        "docs/for-agents/packages/server.md",
        "docs/for-agents/packages/cli.md",
        "packages/chat/src/index.ts",
        "packages/protocol/src/index.ts",
        "packages/server/src/index.ts",
        "packages/cli/src/index.ts",
        "discovery-check"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 150016,
        "reasoningOutputTokens": 601,
        "stderrBytes": 0,
        "providerTokenCostUnits": 179591
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "5c1e77b6f17ed3e42e2df9b75fd59b9f7704ce200273d0b3a68553a2d18792f6",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:48:11.484Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-05-documentation",
        "repositoryId": "consumer-05",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 65519,
        "responseBytes": 857,
        "stderrBytes": 0,
        "stdoutHash": "0a7df94d148de719456b47ff9b59eafc4a2d34f50518670a81bec328940334fe",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 165026,
        "outputTokens": 1400,
        "tokenMethod": "provider",
        "toolCalls": 5
      },
      "contextBytes": 2047,
      "evidenceIds": [
        "documentation-evidence:docs/architecture.md:27-36-claims-AgentsKit-Chat-0.3.x-while-package.json:43-pins-@agentskit/chat-0.4.0-classification-contradictory",
        "review-limitation:semantic-classification-high-confidence-but-ak-docs-audit-could-not-run-because-read-only-filesystem-blocked-.doc-bridge-workflow-lock-creation",
        "recommended-next-action:update-0.3.x-and-0.3-headings-to-0.4.0-or-version-neutral-wording-then-rerun-ak-docs-audit-documentation-json"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 1,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 136704,
        "reasoningOutputTokens": 403,
        "stderrBytes": 0,
        "providerTokenCostUnits": 166426
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "170b186df612a6dfa962613c1c1183b4c702e4590653fb5da2648c39029a33fb",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:48:57.824Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-02-implementation",
        "repositoryId": "consumer-02",
        "category": "implementation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 46292,
        "responseBytes": 556,
        "stderrBytes": 0,
        "stdoutHash": "95ba616ea727ccc0de1b64be0101cabf3540f08452402324f71a3bbfb9264967",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 280690,
        "outputTokens": 1810,
        "tokenMethod": "provider",
        "toolCalls": 7
      },
      "contextBytes": 2094,
      "evidenceIds": [
        ".codex/verification.json",
        ".doc-bridge/workflow/manifest.json",
        ".doc-bridge/workflow/artifacts/report-c7b8d915bd677d64087b53d5cc312c2503dec7b1da86aa486d81915df19f6d9d.json"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "blocked",
      "evidenceQuality": "low",
      "safetyOutcome": "safe",
      "clarificationRequests": 1,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "documentationFindingCount": 0,
        "cachedInputTokens": 234496,
        "reasoningOutputTokens": 678,
        "stderrBytes": 0,
        "providerTokenCostUnits": 282500
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "d18bc3507a03d988d510f1dbb71b7b88cafc5ed29a72807f529f71580e021510",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:49:50.471Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-06-documentation",
        "repositoryId": "consumer-06",
        "category": "documentation",
        "scenarioId": "deterministic-doc-bridge",
        "modelId": "low-cost-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "low-cost-model",
        "role": "low-cost",
        "provider": "codex",
        "model": "gpt-5.6-sol",
        "version": "codex-cli-0.149.0",
        "parametersHash": "6dc5481139cc4fc6960a38ffd4c7e68605108d0d3eac9ac3c269c92703200273",
        "contextLimit": 272000,
        "toolConfigurationHash": "a08b38a4fa7e559811713cabc2e2f62cb05ef266d406dcb38cf46be627417e20",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "deterministic-doc-bridge",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 52609,
        "responseBytes": 1050,
        "stderrBytes": 0,
        "stdoutHash": "f3053a75201a4979a779bffb97c2e636c24eef36fb15b7aa253c3271490d40a2",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 140097,
        "outputTokens": 1180,
        "tokenMethod": "provider",
        "toolCalls": 4
      },
      "contextBytes": 2056,
      "evidenceIds": [
        "evidence-067b19cf083889ce3f9684a42d959745",
        "source-evidence:packages/pack-bundle has no package.json and is not a workspace package; the accurate term is 76 package workspaces, not 76 package dirs",
        "generated-index-evidence:docs/internal/index.generated.json:96 reproduces the stale AGENTS.md claim",
        "review-limitation:semantic classification is high-confidence for directory count wording, but ak-docs was unavailable, so the required audit could not independently emit or score the finding",
        "recommended-next-action:change AGENTS.md:3 from 76 package dirs to 76 package workspaces, then regenerate the internal index and run the documentation audit"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "high",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "errorRate": 1,
        "documentationFindingCount": 1,
        "cachedInputTokens": 106496,
        "reasoningOutputTokens": 314,
        "stderrBytes": 0,
        "providerTokenCostUnits": 141277
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "6cd7628adf67b401ab8d376b0b1f764706c81ea156c110a969385ca88bf8c98f",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:50:44.035Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-02-implementation",
        "repositoryId": "consumer-02",
        "category": "implementation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 53518,
        "responseBytes": 666,
        "stderrBytes": 0,
        "stdoutHash": "67dad6e9e57ff8fbac9e807f0d30e3678a98e1d09a06a51800228194e4461d4a",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 300662,
        "outputTokens": 1917,
        "tokenMethod": "provider",
        "toolCalls": 7
      },
      "contextBytes": 2085,
      "evidenceIds": [
        "patch-evidence:docs/PHASE-1-REVIEW.md:115-120",
        "target:apps/docs-next/content/docs/reference/recipes/simulate-stream.mdx",
        "target:apps/docs-next/content/docs/reference/recipes/cost-guard.mdx",
        "target:apps/docs-next/content/docs/reference/recipes/edit-and-regenerate.mdx",
        "verification-plan:ak-docs-check-json"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "safe",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 249088,
        "reasoningOutputTokens": 925,
        "stderrBytes": 0,
        "providerTokenCostUnits": 302579
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "cd9636c89a1513773cb948b87538a7db6558275114525ebb2578f050db362243",
      "contentHashAlgo": "sha256-normalized-v1"
    },
    {
      "type": "controlled-study-observation",
      "schemaVersion": 1,
      "observationVersion": "v1",
      "observedAt": "2026-08-31T21:51:25.216Z",
      "runId": "phase-9-ab-adjudicated-cost-03",
      "planHash": "f61bf70063fcceb9dfe823ac9d0707f0f004a075143cae6e9518523c64ba1b4f",
      "task": {
        "taskId": "consumer-03-documentation",
        "repositoryId": "consumer-03",
        "category": "documentation",
        "scenarioId": "repository-only",
        "modelId": "reference-model",
        "replicate": 0,
        "variantId": "variant-b",
        "difficulty": "hard"
      },
      "model": {
        "id": "reference-model",
        "role": "reference",
        "provider": "codex",
        "model": "gpt-5.6-luna",
        "version": "codex-cli-0.149.0",
        "parametersHash": "785593a535c713ae8015defb031047fb9ae55926ec4ac0cb0cafcdaa5cba1e30",
        "contextLimit": 272000,
        "toolConfigurationHash": "d2fe8bbf1fa6add2b357c4a199a37224c03bbe4edf29901ff45ae0ef8ee99366",
        "promptContractHash": "5ba19f0e4ea57c4206d2965a55fc1bbd081348a8ccf89c86ff64a81df93bf98b"
      },
      "scenario": {
        "id": "repository-only",
        "network": false
      },
      "execution": {
        "status": "completed",
        "exitCode": 0,
        "signal": null,
        "durationMs": 41135,
        "responseBytes": 410,
        "stderrBytes": 0,
        "stdoutHash": "7c39c7faf467f93d5f8a9f170bc88bac89d4fe134829addb3a533264ace0f855",
        "stderrHash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
        "inputTokens": 208640,
        "outputTokens": 1536,
        "tokenMethod": "provider",
        "toolCalls": 6
      },
      "contextBytes": 2048,
      "evidenceIds": [
        "documentation-evidence",
        "review-limitation"
      ],
      "round": "ab-adjudicated-cost-2026-08-31",
      "taskOutcome": "partial",
      "evidenceQuality": "medium",
      "safetyOutcome": "not-applicable",
      "clarificationRequests": 0,
      "reworkCount": 0,
      "measurements": {
        "acceptanceChecksPassed": 0,
        "acceptanceChecksTotal": 1,
        "cachedInputTokens": 183552,
        "reasoningOutputTokens": 587,
        "stderrBytes": 0,
        "providerTokenCostUnits": 210176
      },
      "adjudication": {
        "status": "automated",
        "actor": "deterministic-rubric-v1",
        "method": "deterministic-rubric-v1",
        "outcome": "partial",
        "reason": "Independent bounded evaluation of execution status, acceptance metrics, and evidence count."
      },
      "contentHash": "0f3fb148c8a84ddd2945366f413713760fd03744dc92b4f25684a0c531242cb6",
      "contentHashAlgo": "sha256-normalized-v1"
    }
  ],
  "contentHash": "3f2c80e6768ffd33a14f38ca8fb9eeb482f4722c9f028ec29adc7d09d54aaef6",
  "contentHashAlgo": "sha256-normalized-v1"
}
