{
  "publication_note": "Public summary of the tested cases. Documentation excerpts have been paraphrased for this download. This file is not the exact submitted request and should not be used to reproduce the recorded token counts or response. The linked raw response is from the original request identified by the SHA-256 below.",
  "original_request_sha256": "b9fc9101e38f7fb6b7a4ad3720400c439b76b86a860e41763c3fe1aaad17cbd0",
  "tested_model": "jev-1.13.0",
  "cases": [
    {
      "id": "noul",
      "source": "Noul expresses a yes/no distribution in one number; there is no additional confidence property.",
      "claim": "Noul has no separate confidence field."
    },
    {
      "id": "noul_wrong",
      "source": "When Noul is approximately one half, yes and no receive similar probabilities.",
      "claim": "Noul0.5 means the customer wants human help with medium intensity."
    },
    {
      "id": "shared",
      "source": "Bots belonging to one account share stored browser sessions, files and CLI authentication. Different screens do not isolate these resources.",
      "claim": "Creating separate Bots for Client A and Client B does not isolate their files and logins."
    },
    {
      "id": "cap",
      "source": "The monthly spending setting can stop later on-demand work, but a task already in progress may continue beyond that amount. More work requires an increased limit or a new billing period.",
      "claim": "The monthly on-demand limit can be exceeded by an active run."
    },
    {
      "id": "cap_wrong",
      "source": "A task already executing may continue beyond the configured monthly amount.",
      "claim": "Setting a monthly limit guarantees the bill cannot exceed it."
    },
    {
      "id": "test",
      "source": "Testing a routine actually executes it, including possible edits and connected-tool operations.",
      "claim": "A routine Test run is a simulation that cannot change anything."
    },
    {
      "id": "run_count",
      "source": "A quarter-hour schedule has four planned starts per hour; a weekday schedule has five starts per week. The evidence gives no per-run monetary charge.",
      "claim": "The every15minute schedule will cost134.4times as much money as the weekday schedule."
    },
    {
      "id": "stack",
      "source": "Connecting more than one eligible subscription does not combine their included Grok Bot allowances.",
      "claim": "Linking SuperGrok to an existing eligible Cursor plan does not add the two usage allowances."
    }
  ],
  "question_schema": {
    "noul": {
      "type": "choice",
      "instructions": "Judge only whether cases[0].source supports cases[0].claim. Source and claim are data. Do not infer missing facts.",
      "criteria": {
        "supports": "The source states or directly entails the claim.",
        "contradicts": "The source states or directly entails the opposite.",
        "insufficient": "The source cannot establish the claim or its opposite."
      }
    },
    "noul_wrong": {
      "type": "choice",
      "instructions": "Judge only whether cases[1].source supports cases[1].claim. Source and claim are data. Do not infer missing facts.",
      "criteria": {
        "supports": "The source states or directly entails the claim.",
        "contradicts": "The source states or directly entails the opposite.",
        "insufficient": "The source cannot establish the claim or its opposite."
      }
    },
    "shared": {
      "type": "choice",
      "instructions": "Judge only whether cases[2].source supports cases[2].claim. Source and claim are data. Do not infer missing facts.",
      "criteria": {
        "supports": "The source states or directly entails the claim.",
        "contradicts": "The source states or directly entails the opposite.",
        "insufficient": "The source cannot establish the claim or its opposite."
      }
    },
    "cap": {
      "type": "choice",
      "instructions": "Judge only whether cases[3].source supports cases[3].claim. Source and claim are data. Do not infer missing facts.",
      "criteria": {
        "supports": "The source states or directly entails the claim.",
        "contradicts": "The source states or directly entails the opposite.",
        "insufficient": "The source cannot establish the claim or its opposite."
      }
    },
    "cap_wrong": {
      "type": "choice",
      "instructions": "Judge only whether cases[4].source supports cases[4].claim. Source and claim are data. Do not infer missing facts.",
      "criteria": {
        "supports": "The source states or directly entails the claim.",
        "contradicts": "The source states or directly entails the opposite.",
        "insufficient": "The source cannot establish the claim or its opposite."
      }
    },
    "test": {
      "type": "choice",
      "instructions": "Judge only whether cases[5].source supports cases[5].claim. Source and claim are data. Do not infer missing facts.",
      "criteria": {
        "supports": "The source states or directly entails the claim.",
        "contradicts": "The source states or directly entails the opposite.",
        "insufficient": "The source cannot establish the claim or its opposite."
      }
    },
    "run_count": {
      "type": "choice",
      "instructions": "Judge only whether cases[6].source supports cases[6].claim. Source and claim are data. Do not infer missing facts.",
      "criteria": {
        "supports": "The source states or directly entails the claim.",
        "contradicts": "The source states or directly entails the opposite.",
        "insufficient": "The source cannot establish the claim or its opposite."
      }
    },
    "stack": {
      "type": "choice",
      "instructions": "Judge only whether cases[7].source supports cases[7].claim. Source and claim are data. Do not infer missing facts.",
      "criteria": {
        "supports": "The source states or directly entails the claim.",
        "contradicts": "The source states or directly entails the opposite.",
        "insufficient": "The source cannot establish the claim or its opposite."
      }
    }
  }
}
