{
  "schemaVersion": "1",
  "kind": "post-hoc-verifier-sensitivity",
  "recordedAt": "2026-09-08T08:45:21.399932+00:00",
  "originalResultId": "15dfe11b-2e6d-4f47-b676-601f2ffa37d0",
  "originalResultSha256": "3176505ee2526d46d9425cb5b0b548a09313db9812c29938efcd7f32f67db7ba",
  "originalManifestHash": "c746dd4e9df8f01a446c584bcd78cf61aed58137ff6585139efd37d902534f29",
  "originalSourceCommit": "8d86dda1be42db57df4c8b82cecd1749ad04bc42",
  "verifierVersions": {
    "original": "1",
    "sensitivity": "2"
  },
  "defect": "Version 1 required the word inconclusive in a scope quotation. It rejected source-grounded statements that explicitly denied generalization using different wording. A failed original verdict is therefore not automatically an agent failure.",
  "repair": "Version 2 accepts three equivalent statements found in the frozen source while preserving all other checks. Original verdicts and source files are unchanged. Every retained answer was assessed under both versions, including failures.",
  "inference": "This is a post-hoc verifier diagnosis and sensitivity analysis, not a preregistered success comparison. Neither the flawed original predicate nor the repaired counts establish that the navigation index improves agent performance.",
  "productionBoundary": "Real agent execution on frozen production response files, prepared for production-platform upload. No per-trial HTTP, authenticated account operations, browser or assistive-technology coverage.",
  "counts": {
    "baseline": {
      "planned": 4,
      "originalPassed": 0,
      "postHocAccepted": 2,
      "repairedFalseNegatives": 2
    },
    "candidate": {
      "planned": 4,
      "originalPassed": 0,
      "postHocAccepted": 4,
      "repairedFalseNegatives": 4
    }
  },
  "trials": [
    {
      "trialId": "1-candidate",
      "variant": "candidate",
      "order": 0,
      "originalStatus": "failed",
      "originalReason": "sources.scope: quotation does not cover the required boundary",
      "postHocOutcomeStatus": "passed",
      "originalVerifier": {
        "passed": false,
        "reason": "sources.scope: quotation does not cover the required boundary",
        "metrics": {
          "failedChecks": 1
        }
      },
      "postHocVerifierV2": {
        "passed": true,
        "reason": "Complete handoff accepted with an equivalent source-grounded scope quotation (post-hoc verifier v2).",
        "metrics": {
          "failedChecks": 0
        }
      },
      "changedByScopeRepair": true,
      "classification": "verifier-false-negative",
      "agentDurationMs": 106321.029708,
      "usage": {
        "inputTokens": 215139,
        "cachedInputTokens": 146048,
        "outputTokens": 2281,
        "reasoningTokens": 36,
        "costUsd": null,
        "source": "agent-reported"
      },
      "diagnostics": {
        "commandCalls": 11,
        "nonzeroCommandExits": 0,
        "events": []
      },
      "scopeSource": {
        "claim": "scope",
        "file": "production/full.md",
        "quote": "Four pairs have little power under this conservative bound. Use a representative task collection, an independently reviewed verifier and a declared sampling plan before making a release-wide claim. This task demonstrates how to perform a complete experiment; it does not establish that indexes help agents in general."
      },
      "outputHash": "358e7e9427fee2d92274e2761a900eb20c28fb034ec7f94a691e3a4eebcdc4aa"
    },
    {
      "trialId": "1-baseline",
      "variant": "baseline",
      "order": 1,
      "originalStatus": "failed",
      "originalReason": "sources.scope: quotation absent from frozen source; sources.scope: quotation does not cover the required boundary",
      "postHocOutcomeStatus": "failed",
      "originalVerifier": {
        "passed": false,
        "reason": "sources.scope: quotation absent from frozen source; sources.scope: quotation does not cover the required boundary",
        "metrics": {
          "failedChecks": 2
        }
      },
      "postHocVerifierV2": {
        "passed": false,
        "reason": "sources.scope: quotation absent from frozen source; sources.scope: quotation does not cover the required boundary",
        "metrics": {
          "failedChecks": 2
        }
      },
      "changedByScopeRepair": false,
      "classification": "retained-verifier-failure",
      "agentDurationMs": 115822.92279100002,
      "usage": {
        "inputTokens": 243214,
        "cachedInputTokens": 207744,
        "outputTokens": 2345,
        "reasoningTokens": 71,
        "costUsd": null,
        "source": "agent-reported"
      },
      "diagnostics": {
        "commandCalls": 11,
        "nonzeroCommandExits": 0,
        "events": []
      },
      "scopeSource": {
        "claim": "scope",
        "file": "production/full.md",
        "quote": "Four pairs have little power under this conservative bound. Use a representative task collection, an independently reviewed verifier and a declared sampling plan before making a release-wide claim. This task demonstrates how to perform a complete experiment; it does not establish that indexes help agents in general.\n\nThis was a selected authentication-source snapshot, not the full repository. The observations do not establish that indexes help or harm agents in other environments. Local plan timestamps and client-produced evidence are not independent third-party attestation."
      },
      "outputHash": "81bb349eda6a5fb47c3f358c11687c6ba77be824f4213e97aa3b5e6f6c99191b"
    },
    {
      "trialId": "2-baseline",
      "variant": "baseline",
      "order": 2,
      "originalStatus": "failed",
      "originalReason": "sources.scope: quotation does not cover the required boundary",
      "postHocOutcomeStatus": "passed",
      "originalVerifier": {
        "passed": false,
        "reason": "sources.scope: quotation does not cover the required boundary",
        "metrics": {
          "failedChecks": 1
        }
      },
      "postHocVerifierV2": {
        "passed": true,
        "reason": "Complete handoff accepted with an equivalent source-grounded scope quotation (post-hoc verifier v2).",
        "metrics": {
          "failedChecks": 0
        }
      },
      "changedByScopeRepair": true,
      "classification": "verifier-false-negative",
      "agentDurationMs": 106127.61379199999,
      "usage": {
        "inputTokens": 193084,
        "cachedInputTokens": 161408,
        "outputTokens": 2268,
        "reasoningTokens": 28,
        "costUsd": null,
        "source": "agent-reported"
      },
      "diagnostics": {
        "commandCalls": 9,
        "nonzeroCommandExits": 0,
        "events": []
      },
      "scopeSource": {
        "claim": "scope",
        "file": "production/full.md",
        "quote": "Four pairs have little power under this conservative bound. Use a representative task collection, an independently reviewed verifier and a declared sampling plan before making a release-wide claim. This task demonstrates how to perform a complete experiment; it does not establish that indexes help agents in general."
      },
      "outputHash": "50ee465afb8b0f0ae2eb09b07db06e8d5b2ed64a262f0a41dfbef9134c8463dd"
    },
    {
      "trialId": "2-candidate",
      "variant": "candidate",
      "order": 3,
      "originalStatus": "failed",
      "originalReason": "sources.scope: quotation does not cover the required boundary",
      "postHocOutcomeStatus": "passed",
      "originalVerifier": {
        "passed": false,
        "reason": "sources.scope: quotation does not cover the required boundary",
        "metrics": {
          "failedChecks": 1
        }
      },
      "postHocVerifierV2": {
        "passed": true,
        "reason": "Complete handoff accepted with an equivalent source-grounded scope quotation (post-hoc verifier v2).",
        "metrics": {
          "failedChecks": 0
        }
      },
      "changedByScopeRepair": true,
      "classification": "verifier-false-negative",
      "agentDurationMs": 95607.83937500004,
      "usage": {
        "inputTokens": 194568,
        "cachedInputTokens": 165248,
        "outputTokens": 2067,
        "reasoningTokens": 47,
        "costUsd": null,
        "source": "agent-reported"
      },
      "diagnostics": {
        "commandCalls": 9,
        "nonzeroCommandExits": 0,
        "events": []
      },
      "scopeSource": {
        "claim": "scope",
        "file": "production/full.md",
        "quote": "Four pairs have little power under this conservative bound. Use a representative task collection, an independently reviewed verifier and a declared sampling plan before making a release-wide claim. This task demonstrates how to perform a complete experiment; it does not establish that indexes help agents in general."
      },
      "outputHash": "50ee465afb8b0f0ae2eb09b07db06e8d5b2ed64a262f0a41dfbef9134c8463dd"
    },
    {
      "trialId": "3-candidate",
      "variant": "candidate",
      "order": 4,
      "originalStatus": "failed",
      "originalReason": "sources.scope: quotation does not cover the required boundary",
      "postHocOutcomeStatus": "passed",
      "originalVerifier": {
        "passed": false,
        "reason": "sources.scope: quotation does not cover the required boundary",
        "metrics": {
          "failedChecks": 1
        }
      },
      "postHocVerifierV2": {
        "passed": true,
        "reason": "Complete handoff accepted with an equivalent source-grounded scope quotation (post-hoc verifier v2).",
        "metrics": {
          "failedChecks": 0
        }
      },
      "changedByScopeRepair": true,
      "classification": "verifier-false-negative",
      "agentDurationMs": 89096.70545800001,
      "usage": {
        "inputTokens": 217246,
        "cachedInputTokens": 179712,
        "outputTokens": 2003,
        "reasoningTokens": 31,
        "costUsd": null,
        "source": "agent-reported"
      },
      "diagnostics": {
        "commandCalls": 8,
        "nonzeroCommandExits": 0,
        "events": []
      },
      "scopeSource": {
        "claim": "scope",
        "file": "production/full.md",
        "quote": "Four pairs have little power under this conservative bound. Use a representative task collection, an independently reviewed verifier and a declared sampling plan before making a release-wide claim. This task demonstrates how to perform a complete experiment; it does not establish that indexes help agents in general."
      },
      "outputHash": "50ee465afb8b0f0ae2eb09b07db06e8d5b2ed64a262f0a41dfbef9134c8463dd"
    },
    {
      "trialId": "3-baseline",
      "variant": "baseline",
      "order": 5,
      "originalStatus": "failed",
      "originalReason": "sources.scope: quotation does not cover the required boundary",
      "postHocOutcomeStatus": "passed",
      "originalVerifier": {
        "passed": false,
        "reason": "sources.scope: quotation does not cover the required boundary",
        "metrics": {
          "failedChecks": 1
        }
      },
      "postHocVerifierV2": {
        "passed": true,
        "reason": "Complete handoff accepted with an equivalent source-grounded scope quotation (post-hoc verifier v2).",
        "metrics": {
          "failedChecks": 0
        }
      },
      "changedByScopeRepair": true,
      "classification": "verifier-false-negative",
      "agentDurationMs": 110116.500458,
      "usage": {
        "inputTokens": 237281,
        "cachedInputTokens": 207488,
        "outputTokens": 2339,
        "reasoningTokens": 74,
        "costUsd": null,
        "source": "agent-reported"
      },
      "diagnostics": {
        "commandCalls": 11,
        "nonzeroCommandExits": 0,
        "events": []
      },
      "scopeSource": {
        "claim": "scope",
        "file": "production/full.md",
        "quote": "This was a selected authentication-source snapshot, not the full repository. The observations do not establish that indexes help or harm agents in other environments. Local plan timestamps and client-produced evidence are not independent third-party attestation."
      },
      "outputHash": "aa91bd6b13a17a20f50b5880e844f99521db550db2f4a1eef152a1372600b4b0"
    },
    {
      "trialId": "4-baseline",
      "variant": "baseline",
      "order": 6,
      "originalStatus": "failed",
      "originalReason": "answer.policy.validateCommand: incorrect CLI workflow; sources.scope: quotation does not cover the required boundary",
      "postHocOutcomeStatus": "failed",
      "originalVerifier": {
        "passed": false,
        "reason": "answer.policy.validateCommand: incorrect CLI workflow; sources.scope: quotation does not cover the required boundary",
        "metrics": {
          "failedChecks": 2
        }
      },
      "postHocVerifierV2": {
        "passed": false,
        "reason": "answer.policy.validateCommand: incorrect CLI workflow",
        "metrics": {
          "failedChecks": 1
        }
      },
      "changedByScopeRepair": false,
      "classification": "retained-verifier-failure",
      "agentDurationMs": 95577.63791699999,
      "usage": {
        "inputTokens": 194944,
        "cachedInputTokens": 165760,
        "outputTokens": 2140,
        "reasoningTokens": 17,
        "costUsd": null,
        "source": "agent-reported"
      },
      "diagnostics": {
        "commandCalls": 9,
        "nonzeroCommandExits": 0,
        "events": []
      },
      "scopeSource": {
        "claim": "scope",
        "file": "production/full.md",
        "quote": "Four pairs have little power under this conservative bound. Use a representative task collection, an independently reviewed verifier and a declared sampling plan before making a release-wide claim. This task demonstrates how to perform a complete experiment; it does not establish that indexes help agents in general."
      },
      "outputHash": "ba899f215df29105d8353fa6e5036b35b7ced7d4d546fb8865c7c7ed4a5281fd"
    },
    {
      "trialId": "4-candidate",
      "variant": "candidate",
      "order": 7,
      "originalStatus": "failed",
      "originalReason": "sources.scope: quotation does not cover the required boundary",
      "postHocOutcomeStatus": "passed",
      "originalVerifier": {
        "passed": false,
        "reason": "sources.scope: quotation does not cover the required boundary",
        "metrics": {
          "failedChecks": 1
        }
      },
      "postHocVerifierV2": {
        "passed": true,
        "reason": "Complete handoff accepted with an equivalent source-grounded scope quotation (post-hoc verifier v2).",
        "metrics": {
          "failedChecks": 0
        }
      },
      "changedByScopeRepair": true,
      "classification": "verifier-false-negative",
      "agentDurationMs": 100834.882125,
      "usage": {
        "inputTokens": 197566,
        "cachedInputTokens": 171008,
        "outputTokens": 2108,
        "reasoningTokens": 21,
        "costUsd": null,
        "source": "agent-reported"
      },
      "diagnostics": {
        "commandCalls": 8,
        "nonzeroCommandExits": 0,
        "events": []
      },
      "scopeSource": {
        "claim": "scope",
        "file": "production/full.md",
        "quote": "Four pairs have little power under this conservative bound. Use a representative task collection, an independently reviewed verifier and a declared sampling plan before making a release-wide claim. This task demonstrates how to perform a complete experiment; it does not establish that indexes help agents in general."
      },
      "outputHash": "50ee465afb8b0f0ae2eb09b07db06e8d5b2ed64a262f0a41dfbef9134c8463dd"
    }
  ],
  "verifierFingerprints": {
    "verify.py": "a528400d84d516370f0f999b9eb6f2f66a8793907ccc4603df92d32f673cbdfd",
    "verify_v2.py": "ca0bacae64bd085f8d60cc5dc54a6834012547606c4483928360e33729068567"
  }
}
