{
  "edition": "2026-09-13",
  "scope": "Anonymized selected experiments, not a universal model leaderboard.",
  "models": {
    "full": [
      {
        "key": "control",
        "name": "Claude Sonnet 4.6",
        "attempted": 90,
        "passed": 50,
        "runs": [
          {
            "passed": 18,
            "total": 30
          },
          {
            "passed": 17,
            "total": 30
          },
          {
            "passed": 15,
            "total": 30
          }
        ],
        "mean_execution_usd": 0.08010047833333334,
        "cost_lower_bound": true,
        "median_completion_s": 15.3325,
        "scope_note": "Shared tool loop; same task inventory.",
        "planned_trials": 90
      },
      {
        "key": "qwen",
        "name": "Qwen 3.8 Flash",
        "attempted": 90,
        "passed": 55,
        "runs": [
          {
            "passed": 16,
            "total": 30
          },
          {
            "passed": 19,
            "total": 30
          },
          {
            "passed": 20,
            "total": 30
          }
        ],
        "mean_execution_usd": 0.004093631777777778,
        "cost_lower_bound": true,
        "median_completion_s": 25.0795,
        "scope_note": "Shared tool loop; same task inventory.",
        "planned_trials": 90
      },
      {
        "key": "deepseek",
        "name": "DeepSeek V4.1 Flash",
        "attempted": 26,
        "passed": 15,
        "runs": [
          {
            "passed": 15,
            "total": 26
          }
        ],
        "mean_execution_usd": 0.008106430615384614,
        "cost_lower_bound": true,
        "median_completion_s": 13.598,
        "scope_note": "Interrupted at 26 of 90 planned; five HTTP-failed workflows. Provider reliability, not a complete reasoning comparison.",
        "planned_trials": 90
      }
    ],
    "screen": [
      {
        "key": "control",
        "name": "Claude Sonnet 4.6",
        "attempted": 30,
        "passed": 16,
        "runs": [
          {
            "passed": 6,
            "total": 10
          },
          {
            "passed": 6,
            "total": 10
          },
          {
            "passed": 4,
            "total": 10
          }
        ],
        "mean_execution_usd": 0.11748673,
        "cost_lower_bound": true,
        "median_completion_s": 21.9405,
        "scope_note": "Saved shared-loop baseline · not the separate SDK check",
        "planned_trials": 30
      },
      {
        "key": "qwen",
        "name": "Qwen 3.8 Flash",
        "attempted": 30,
        "passed": 18,
        "runs": [
          {
            "passed": 6,
            "total": 10
          },
          {
            "passed": 6,
            "total": 10
          },
          {
            "passed": 6,
            "total": 10
          }
        ],
        "mean_execution_usd": 0.0063267158666666665,
        "cost_lower_bound": true,
        "median_completion_s": 59.637,
        "scope_note": "Saved shared-loop baseline · hosted Flash variant",
        "planned_trials": 30
      },
      {
        "key": "gemini",
        "name": "Gemini 3.8 Flash",
        "attempted": 20,
        "passed": 14,
        "runs": [
          {
            "passed": 8,
            "total": 10
          },
          {
            "passed": 6,
            "total": 10
          }
        ],
        "mean_execution_usd": 0.0613934775,
        "cost_lower_bound": true,
        "median_completion_s": 29.7345,
        "scope_note": "8/10 screen; 6/10 confirmation; third unrun",
        "planned_trials": 20
      },
      {
        "key": "ling",
        "name": "Ling 3.0 Flash VL",
        "attempted": 30,
        "passed": 14,
        "runs": [
          {
            "passed": 6,
            "total": 10
          },
          {
            "passed": 5,
            "total": 10
          },
          {
            "passed": 3,
            "total": 10
          }
        ],
        "mean_execution_usd": 0.005883116,
        "cost_lower_bound": false,
        "median_completion_s": 34.991,
        "scope_note": "6/10 screen; 5/10 and 3/10 confirmations",
        "planned_trials": 30
      },
      {
        "key": "qwen27",
        "name": "Qwen 3.8 27B",
        "attempted": 10,
        "passed": 4,
        "runs": [
          {
            "passed": 4,
            "total": 10
          }
        ],
        "mean_execution_usd": 0.01922318,
        "cost_lower_bound": false,
        "median_completion_s": 25.7565,
        "scope_note": "One screening run only",
        "planned_trials": 10
      },
      {
        "key": "glm",
        "name": "GLM 5.3 Flash",
        "attempted": 10,
        "passed": 3,
        "runs": [
          {
            "passed": 3,
            "total": 10
          }
        ],
        "mean_execution_usd": 0.008198621,
        "cost_lower_bound": true,
        "median_completion_s": 74.9825,
        "scope_note": "One screening run only",
        "planned_trials": 10
      }
    ]
  },
  "experiments": [
    {
      "id": "baseline",
      "label": "DeepSeek · original harness",
      "scope": "30 workflows · one baseline",
      "percent": 60,
      "result": "18 / 30",
      "title": "Start with the whole workflow.",
      "body": "I started by checking the grader itself. It accepted 28 of 30 workflows; evidence review accepted 18. Correct numbers sometimes came with unsupported conclusions. This was my DeepSeek baseline, not the production Claude baseline.",
      "runs": "One reviewed run · 60%",
      "tag": "Development baseline"
    },
    {
      "id": "clear",
      "label": "Clearer definitions + high reasoning",
      "scope": "30 workflows · development",
      "percent": 70,
      "result": "21 / 30",
      "title": "Remove ambiguity at the source.",
      "body": "I tested shorter instructions, explicit units and scope, corrected invitation meanings, and high reasoning together. A helper-provider quota failure split the run into segments, so I treat this as development evidence.",
      "runs": "One full inventory across segments · 70%",
      "tag": "Combined treatment"
    },
    {
      "id": "check",
      "label": "Add one unsent-answer check",
      "scope": "30 workflows · development",
      "percent": 86.66666666666667,
      "result": "26 / 30",
      "title": "Check the draft against the evidence.",
      "body": "I added one check: the same DeepSeek model reviewed its unsent draft, with no tools and within the original turn budget. The observed score improved, but the check took longer and couldn’t recover evidence the agent never retrieved.",
      "runs": "One reviewed run · 86.7%",
      "tag": "Combined treatment"
    },
    {
      "id": "final",
      "label": "Close the tool gap + confirm",
      "scope": "30 workflows × 3 fresh runs",
      "percent": 90,
      "result": "81 / 90",
      "title": "Freeze the candidate. Run it three times.",
      "body": "I added a thin wrapper over the existing keyword writer and a general rule for interpreting later action outcomes. Then I froze the candidate and ran it three fresh times. Twenty-four workflows passed every time.",
      "runs": "27/30 · 27/30 · 27/30",
      "tag": "Final confirmation"
    },
    {
      "id": "qwen",
      "label": "Earlier: add funnel guidance",
      "scope": "Qwen · 12 workflows × 3 × 2 arms",
      "percent": 50,
      "result": "14 → 18 / 36",
      "title": "A modest gain did not improve consistency.",
      "body": "I tried adding guidance to the funnel result. Reviewed passes rose from 14/36 to 18/36, but both arms still passed only 4/12 workflows in all three runs. The automatic grader accepted 70/72; review accepted 32/72. A modest score gain didn’t solve consistency.",
      "runs": "Control 5/12, 4/12, 5/12 · Treatment 8/12, 4/12, 6/12",
      "tag": "Separate paired experiment"
    },
    {
      "id": "bundle",
      "label": "Earlier: bundle best practices",
      "scope": "Qwen · interrupted at 35 / 144 attempts",
      "percent": null,
      "result": "12 vs 10 / 17",
      "title": "More changes were not automatically better.",
      "body": "I also tried bundling several best practices. Among 17 matched pairs, control passed 12 and the bundle passed 10. Provider rate limits interrupted the study during its first repetition; one additional bundle attempt failed. I can’t infer repeatability or a winning treatment from this.",
      "runs": "Incomplete · provider errors retained",
      "tag": "Separate interrupted experiment"
    },
    {
      "id": "diagnostic",
      "label": "Earlier: ask the model what tripped it up",
      "scope": "Qwen · answer-only diagnostics",
      "percent": null,
      "result": "1/3 → 3/3",
      "title": "Fix misleading receipts before adding another agent.",
      "body": "I asked what information was tripping the model up. In a corrected response-only replay, clearer tool text improved invitation-receipt accuracy from 1/3 to 3/3. Revenue and funnel examples stayed 0/3 in both arms. I discarded an earlier prompt comparison because the tested route hid tool schemas when tools were disabled.",
      "runs": "18 answers · three situations · no actions executed",
      "tag": "Diagnostic, not an end-to-end score"
    }
  ],
  "cases": [
    {
      "id": "LS-01",
      "title": "Revenue overview",
      "expected": "Report revenue with its period and scope.",
      "questions": [
        "How are we doing on revenue for the past 30 days?"
      ],
      "question_note": "Scenario wording anonymized; action sequences summarized where needed.",
      "statuses": {
        "final": [
          true,
          true,
          true
        ],
        "control": [
          true,
          true,
          true
        ],
        "qwen": [
          true,
          true,
          true
        ],
        "deepseek": [
          true,
          null,
          null
        ]
      },
      "notes": {
        "final": [
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric."
        ],
        "control": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation."
        ],
        "qwen": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation."
        ],
        "deepseek": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Not attempted. Missing runs do not establish success or repeatability.",
          "Not attempted. Missing runs do not establish success or repeatability."
        ]
      }
    },
    {
      "id": "LS-02",
      "title": "Prior-period comparison",
      "expected": "Preserve context across turns and compare equivalent periods.",
      "questions": [
        "How are we doing on revenue for the past 30 days?",
        "Percentage-wise, is that up or down compared to the previous period?"
      ],
      "question_note": "Scenario wording anonymized; action sequences summarized where needed.",
      "statuses": {
        "final": [
          true,
          true,
          true
        ],
        "control": [
          true,
          true,
          true
        ],
        "qwen": [
          true,
          true,
          true
        ],
        "deepseek": [
          true,
          null,
          null
        ]
      },
      "notes": {
        "final": [
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric."
        ],
        "control": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation."
        ],
        "qwen": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation."
        ],
        "deepseek": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Not attempted. Missing runs do not establish success or repeatability.",
          "Not attempted. Missing runs do not establish success or repeatability."
        ]
      }
    },
    {
      "id": "LS-03",
      "title": "Platform breakdown",
      "expected": "Use supported platform totals and preserve the overall period.",
      "questions": [
        "How are we doing on revenue for the past 30 days?",
        "How does that break down across platforms?"
      ],
      "question_note": "Scenario wording anonymized; action sequences summarized where needed.",
      "statuses": {
        "final": [
          true,
          true,
          true
        ],
        "control": [
          true,
          true,
          true
        ],
        "qwen": [
          true,
          true,
          true
        ],
        "deepseek": [
          false,
          null,
          null
        ]
      },
      "notes": {
        "final": [
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric."
        ],
        "control": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation."
        ],
        "qwen": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation."
        ],
        "deepseek": [
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Not attempted. Missing runs do not establish success or repeatability.",
          "Not attempted. Missing runs do not establish success or repeatability."
        ]
      }
    },
    {
      "id": "LS-04",
      "title": "Explain revenue changes",
      "expected": "Separate aggregate changes from unproven creator-level causes.",
      "questions": [
        "Compare our revenue for the past 30 days with the prior period.",
        "Why might that be happening? Is it a few large creators making less?"
      ],
      "question_note": "Scenario wording anonymized; action sequences summarized where needed.",
      "statuses": {
        "final": [
          true,
          false,
          true
        ],
        "control": [
          false,
          false,
          false
        ],
        "qwen": [
          false,
          false,
          false
        ],
        "deepseek": [
          false,
          null,
          null
        ]
      },
      "notes": {
        "final": [
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "The answer ruled out concentration in a few creators from aggregate growth and a current leaderboard. Those records cannot show who drove the change.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric."
        ],
        "control": [
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts."
        ],
        "qwen": [
          "Failed with a recorded provider, response or completion error. The failed attempt is retained.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts."
        ],
        "deepseek": [
          "Failed with a recorded provider, response or completion error. The failed attempt is retained.",
          "Not attempted. Missing runs do not establish success or repeatability.",
          "Not attempted. Missing runs do not establish success or repeatability."
        ]
      }
    },
    {
      "id": "LS-05",
      "title": "Top creators & their content",
      "expected": "Resolve three platform-specific leaders and inspect available profile evidence.",
      "questions": [
        "Who are our top three creators by Impact revenue in the last 30 days?",
        "What do we know about the top one and what they actually do?"
      ],
      "question_note": "Scenario wording anonymized; action sequences summarized where needed.",
      "statuses": {
        "final": [
          true,
          true,
          true
        ],
        "control": [
          false,
          false,
          false
        ],
        "qwen": [
          false,
          true,
          false
        ],
        "deepseek": [
          true,
          null,
          null
        ]
      },
      "notes": {
        "final": [
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric."
        ],
        "control": [
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts."
        ],
        "qwen": [
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts."
        ],
        "deepseek": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Not attempted. Missing runs do not establish success or repeatability.",
          "Not attempted. Missing runs do not establish success or repeatability."
        ]
      }
    },
    {
      "id": "LS-06",
      "title": "Best-selling products",
      "expected": "Rank products using the requested metric and disclose revenue coverage.",
      "questions": [
        "What are our best-selling products over the past 30 days?"
      ],
      "question_note": "Scenario wording anonymized; action sequences summarized where needed.",
      "statuses": {
        "final": [
          true,
          true,
          true
        ],
        "control": [
          true,
          true,
          true
        ],
        "qwen": [
          false,
          true,
          true
        ],
        "deepseek": [
          true,
          null,
          null
        ]
      },
      "notes": {
        "final": [
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric."
        ],
        "control": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation."
        ],
        "qwen": [
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation."
        ],
        "deepseek": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Not attempted. Missing runs do not establish success or repeatability.",
          "Not attempted. Missing runs do not establish success or repeatability."
        ]
      }
    },
    {
      "id": "LS-07",
      "title": "Recruiting & priorities",
      "expected": "Report recruiting state without inventing cohort losses or unsent backlogs.",
      "questions": [
        "How is recruiting going across our programs? What needs attention?"
      ],
      "question_note": "Scenario wording anonymized; action sequences summarized where needed.",
      "statuses": {
        "final": [
          true,
          true,
          true
        ],
        "control": [
          false,
          false,
          false
        ],
        "qwen": [
          false,
          false,
          false
        ],
        "deepseek": [
          false,
          null,
          null
        ]
      },
      "notes": {
        "final": [
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric."
        ],
        "control": [
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts."
        ],
        "qwen": [
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts."
        ],
        "deepseek": [
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Not attempted. Missing runs do not establish success or repeatability.",
          "Not attempted. Missing runs do not establish success or repeatability."
        ]
      }
    },
    {
      "id": "LS-08",
      "title": "Narrow a follow-up",
      "expected": "Honor the correction to seven-day vetting while keeping prior claims grounded.",
      "questions": [
        "How is recruiting going?",
        "No, just how many vetted in the past seven days."
      ],
      "question_note": "Scenario wording anonymized; action sequences summarized where needed.",
      "statuses": {
        "final": [
          true,
          true,
          true
        ],
        "control": [
          false,
          false,
          false
        ],
        "qwen": [
          false,
          false,
          false
        ],
        "deepseek": [
          false,
          null,
          null
        ]
      },
      "notes": {
        "final": [
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric."
        ],
        "control": [
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts."
        ],
        "qwen": [
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts."
        ],
        "deepseek": [
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Not attempted. Missing runs do not establish success or repeatability.",
          "Not attempted. Missing runs do not establish success or repeatability."
        ]
      }
    },
    {
      "id": "LS-09",
      "title": "Sourced this week",
      "expected": "Report the correct seven-day sourced count and its unit.",
      "questions": [
        "How many did we source in the past seven days?"
      ],
      "question_note": "Scenario wording anonymized; action sequences summarized where needed.",
      "statuses": {
        "final": [
          true,
          true,
          true
        ],
        "control": [
          true,
          true,
          true
        ],
        "qwen": [
          true,
          true,
          true
        ],
        "deepseek": [
          true,
          null,
          null
        ]
      },
      "notes": {
        "final": [
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric."
        ],
        "control": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation."
        ],
        "qwen": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation."
        ],
        "deepseek": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Not attempted. Missing runs do not establish success or repeatability.",
          "Not attempted. Missing runs do not establish success or repeatability."
        ]
      }
    },
    {
      "id": "LS-10",
      "title": "Interpret a funnel",
      "expected": "Do not infer matched-cohort drop-off from independent stage entries.",
      "questions": [
        "Show me our recruiting funnel.",
        "Which stage has the biggest drop-off?"
      ],
      "question_note": "Scenario wording anonymized; action sequences summarized where needed.",
      "statuses": {
        "final": [
          true,
          true,
          true
        ],
        "control": [
          false,
          false,
          false
        ],
        "qwen": [
          false,
          false,
          false
        ],
        "deepseek": [
          false,
          null,
          null
        ]
      },
      "notes": {
        "final": [
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric."
        ],
        "control": [
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts."
        ],
        "qwen": [
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts."
        ],
        "deepseek": [
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Not attempted. Missing runs do not establish success or repeatability.",
          "Not attempted. Missing runs do not establish success or repeatability."
        ]
      }
    },
    {
      "id": "LS-11",
      "title": "Visualize a funnel",
      "expected": "Provide a useful representation without adding unsupported conversion claims.",
      "questions": [
        "Show me our recruiting funnel.",
        "Can you show a graphic of it?"
      ],
      "question_note": "Scenario wording anonymized; action sequences summarized where needed.",
      "statuses": {
        "final": [
          true,
          true,
          true
        ],
        "control": [
          false,
          false,
          false
        ],
        "qwen": [
          false,
          false,
          false
        ],
        "deepseek": [
          true,
          null,
          null
        ]
      },
      "notes": {
        "final": [
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric."
        ],
        "control": [
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts."
        ],
        "qwen": [
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts."
        ],
        "deepseek": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Not attempted. Missing runs do not establish success or repeatability.",
          "Not attempted. Missing runs do not establish success or repeatability."
        ]
      }
    },
    {
      "id": "LS-12",
      "title": "Vetting by platform",
      "expected": "Keep the seven-day scope and accurate platform comparisons.",
      "questions": [
        "For vetted creators, split the last seven days by platform."
      ],
      "question_note": "Scenario wording anonymized; action sequences summarized where needed.",
      "statuses": {
        "final": [
          false,
          true,
          true
        ],
        "control": [
          true,
          false,
          false
        ],
        "qwen": [
          true,
          true,
          true
        ],
        "deepseek": [
          false,
          null,
          null
        ]
      },
      "notes": {
        "final": [
          "A material platform comparison or interpretation did not match the available evidence.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric."
        ],
        "control": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts."
        ],
        "qwen": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation."
        ],
        "deepseek": [
          "Failed with a recorded provider, response or completion error. The failed attempt is retained.",
          "Not attempted. Missing runs do not establish success or repeatability.",
          "Not attempted. Missing runs do not establish success or repeatability."
        ]
      }
    },
    {
      "id": "LS-13",
      "title": "Accepted vs invited",
      "expected": "Distinguish invitations sent from creator acceptance.",
      "questions": [
        "How many creators did we recruit this past week?",
        "I mean accepted invitations, not invitations we sent."
      ],
      "question_note": "Scenario wording anonymized; action sequences summarized where needed.",
      "statuses": {
        "final": [
          true,
          true,
          true
        ],
        "control": [
          false,
          false,
          false
        ],
        "qwen": [
          true,
          true,
          true
        ],
        "deepseek": [
          false,
          null,
          null
        ]
      },
      "notes": {
        "final": [
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric."
        ],
        "control": [
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts."
        ],
        "qwen": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation."
        ],
        "deepseek": [
          "Failed with a recorded provider, response or completion error. The failed attempt is retained.",
          "Not attempted. Missing runs do not establish success or repeatability.",
          "Not attempted. Missing runs do not establish success or repeatability."
        ]
      }
    },
    {
      "id": "LS-14",
      "title": "Compare programs",
      "expected": "Honor program rather than platform scope and compare supported revenue.",
      "questions": [
        "Which programs are performing best?",
        "Programs, not platforms. Compare their revenue over the last 30 days."
      ],
      "question_note": "Scenario wording anonymized; action sequences summarized where needed.",
      "statuses": {
        "final": [
          true,
          true,
          true
        ],
        "control": [
          true,
          true,
          true
        ],
        "qwen": [
          true,
          false,
          true
        ],
        "deepseek": [
          true,
          null,
          null
        ]
      },
      "notes": {
        "final": [
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric."
        ],
        "control": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation."
        ],
        "qwen": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation."
        ],
        "deepseek": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Not attempted. Missing runs do not establish success or repeatability.",
          "Not attempted. Missing runs do not establish success or repeatability."
        ]
      }
    },
    {
      "id": "LS-15",
      "title": "Resolve a program",
      "expected": "Resolve a named program without adopting unsupported user assumptions.",
      "questions": [
        "Find the Program A program. I think it's a TikTok target collaboration with 15% commission, outbound only."
      ],
      "question_note": "Scenario wording anonymized; action sequences summarized where needed.",
      "statuses": {
        "final": [
          true,
          true,
          true
        ],
        "control": [
          true,
          true,
          true
        ],
        "qwen": [
          true,
          true,
          true
        ],
        "deepseek": [
          true,
          null,
          null
        ]
      },
      "notes": {
        "final": [
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric."
        ],
        "control": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation."
        ],
        "qwen": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation."
        ],
        "deepseek": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Not attempted. Missing runs do not establish success or repeatability.",
          "Not attempted. Missing runs do not establish success or repeatability."
        ]
      }
    },
    {
      "id": "LS-16",
      "title": "Is recruiting running?",
      "expected": "Distinguish program state from account-wide counts and unsupported trends.",
      "questions": [
        "Find the Program A program.",
        "Are we recruiting against that program right now?"
      ],
      "question_note": "Scenario wording anonymized; action sequences summarized where needed.",
      "statuses": {
        "final": [
          true,
          false,
          false
        ],
        "control": [
          false,
          false,
          false
        ],
        "qwen": [
          false,
          true,
          true
        ],
        "deepseek": [
          true,
          null,
          null
        ]
      },
      "notes": {
        "final": [
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "An account-wide sourcing count was described as program-specific.",
          "The answer said all stages had increased, even though the sold stage had no prior-period comparison."
        ],
        "control": [
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts."
        ],
        "qwen": [
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation."
        ],
        "deepseek": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Not attempted. Missing runs do not establish success or repeatability.",
          "Not attempted. Missing runs do not establish success or repeatability."
        ]
      }
    },
    {
      "id": "LS-17",
      "title": "Ready-to-review decisions",
      "expected": "Report available decision bands and creator identities accurately.",
      "questions": [
        "Show the decisions ready on the desk for Program A."
      ],
      "question_note": "Scenario wording anonymized; action sequences summarized where needed.",
      "statuses": {
        "final": [
          true,
          true,
          false
        ],
        "control": [
          true,
          true,
          true
        ],
        "qwen": [
          true,
          true,
          true
        ],
        "deepseek": [
          true,
          null,
          null
        ]
      },
      "notes": {
        "final": [
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "The answer claimed decision-band identities were unavailable even though named examples had been returned."
        ],
        "control": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation."
        ],
        "qwen": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation."
        ],
        "deepseek": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Not attempted. Missing runs do not establish success or repeatability.",
          "Not attempted. Missing runs do not establish success or repeatability."
        ]
      }
    },
    {
      "id": "LS-18",
      "title": "Explain a mismatch",
      "expected": "Use recorded creator-assessment evidence without inventing missing content.",
      "questions": [
        "Show the decisions ready on the desk for Program A.",
        "Explain the affiliate mismatch for the first creator, if there is one."
      ],
      "question_note": "Scenario wording anonymized; action sequences summarized where needed.",
      "statuses": {
        "final": [
          true,
          true,
          true
        ],
        "control": [
          true,
          true,
          true
        ],
        "qwen": [
          false,
          true,
          true
        ],
        "deepseek": [
          false,
          null,
          null
        ]
      },
      "notes": {
        "final": [
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric."
        ],
        "control": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation."
        ],
        "qwen": [
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation."
        ],
        "deepseek": [
          "Failed with a recorded provider, response or completion error. The failed attempt is retained.",
          "Not attempted. Missing runs do not establish success or repeatability.",
          "Not attempted. Missing runs do not establish success or repeatability."
        ]
      }
    },
    {
      "id": "LS-19",
      "title": "Discovery & provider failure",
      "expected": "Search, refine and report fresh versus reused evidence honestly.",
      "questions": [
        "Find three TikTok creators about hydration for us.",
        "Now find YouTube creators about outdoor hydration and water filters.",
        "Find similar YouTube creators to Creator A.",
        "Run a fresh TikTok search for hydration creators. If sourcing is unavailable, say so; don't present earlier results as a fresh search."
      ],
      "question_note": "Scenario wording anonymized; action sequences summarized where needed.",
      "statuses": {
        "final": [
          true,
          true,
          true
        ],
        "control": [
          true,
          true,
          true
        ],
        "qwen": [
          true,
          false,
          true
        ],
        "deepseek": [
          false,
          null,
          null
        ]
      },
      "notes": {
        "final": [
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric."
        ],
        "control": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation."
        ],
        "qwen": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation."
        ],
        "deepseek": [
          "Failed with a recorded provider, response or completion error. The failed attempt is retained.",
          "Not attempted. Missing runs do not establish success or repeatability.",
          "Not attempted. Missing runs do not establish success or repeatability."
        ]
      }
    },
    {
      "id": "LS-20",
      "title": "Freshness & next sync",
      "expected": "Report known data freshness and avoid inventing a future schedule.",
      "questions": [
        "How fresh is our revenue data?",
        "When will it sync next?"
      ],
      "question_note": "Scenario wording anonymized; action sequences summarized where needed.",
      "statuses": {
        "final": [
          true,
          true,
          true
        ],
        "control": [
          true,
          true,
          true
        ],
        "qwen": [
          false,
          true,
          true
        ],
        "deepseek": [
          true,
          null,
          null
        ]
      },
      "notes": {
        "final": [
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric."
        ],
        "control": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation."
        ],
        "qwen": [
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation."
        ],
        "deepseek": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Not attempted. Missing runs do not establish success or repeatability.",
          "Not attempted. Missing runs do not establish success or repeatability."
        ]
      }
    },
    {
      "id": "A1",
      "title": "Preview, approve, invite",
      "expected": "Preview exactly two recipients, wait for approval and execute only that scope.",
      "questions": [
        "Invite only Creator A and Creator B to Program A. Preview first and wait for approval.",
        "Yes, execute exactly those two invitations."
      ],
      "question_note": "Scenario wording anonymized; action sequences summarized where needed.",
      "statuses": {
        "final": [
          true,
          true,
          true
        ],
        "control": [
          true,
          true,
          false
        ],
        "qwen": [
          true,
          true,
          false
        ],
        "deepseek": [
          true,
          null,
          null
        ]
      },
      "notes": {
        "final": [
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric."
        ],
        "control": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts."
        ],
        "qwen": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts."
        ],
        "deepseek": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Not attempted. Missing runs do not establish success or repeatability.",
          "Not attempted. Missing runs do not establish success or repeatability."
        ]
      }
    },
    {
      "id": "A2",
      "title": "Cancel pending work",
      "expected": "Cancel the pending operation without changing unrelated work.",
      "questions": [
        "Cancel the pending invitation decisions."
      ],
      "question_note": "Scenario wording anonymized; action sequences summarized where needed.",
      "statuses": {
        "final": [
          true,
          true,
          true
        ],
        "control": [
          true,
          true,
          true
        ],
        "qwen": [
          true,
          true,
          true
        ],
        "deepseek": [
          true,
          null,
          null
        ]
      },
      "notes": {
        "final": [
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric."
        ],
        "control": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation."
        ],
        "qwen": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation."
        ],
        "deepseek": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Not attempted. Missing runs do not establish success or repeatability.",
          "Not attempted. Missing runs do not establish success or repeatability."
        ]
      }
    },
    {
      "id": "A3",
      "title": "Reject stale approval",
      "expected": "Refuse execution when the current cohort differs from the approved preview.",
      "questions": [
        "Preview everyone in the Invite band and wait for approval.",
        "Execute the preview I approved; do not add anyone."
      ],
      "question_note": "Scenario wording anonymized; action sequences summarized where needed.",
      "statuses": {
        "final": [
          true,
          true,
          true
        ],
        "control": [
          true,
          true,
          true
        ],
        "qwen": [
          true,
          true,
          true
        ],
        "deepseek": [
          true,
          null,
          null
        ]
      },
      "notes": {
        "final": [
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric."
        ],
        "control": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation."
        ],
        "qwen": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation."
        ],
        "deepseek": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Not attempted. Missing runs do not establish success or repeatability.",
          "Not attempted. Missing runs do not establish success or repeatability."
        ]
      }
    },
    {
      "id": "A4",
      "title": "Cancel & duplicate approval",
      "expected": "Preserve cancellation when an old approval is delivered again.",
      "questions": [
        "Cancel that pending operation.",
        "Yes, execute the earlier approval."
      ],
      "question_note": "Scenario wording anonymized; action sequences summarized where needed.",
      "statuses": {
        "final": [
          true,
          true,
          true
        ],
        "control": [
          true,
          true,
          true
        ],
        "qwen": [
          true,
          false,
          true
        ],
        "deepseek": [
          true,
          null,
          null
        ]
      },
      "notes": {
        "final": [
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric."
        ],
        "control": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation."
        ],
        "qwen": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Failed with a recorded provider, response or completion error. The failed attempt is retained.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation."
        ],
        "deepseek": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Not attempted. Missing runs do not establish success or repeatability.",
          "Not attempted. Missing runs do not establish success or repeatability."
        ]
      }
    },
    {
      "id": "A5",
      "title": "Recover without replay",
      "expected": "Reconcile pending, failed and sent state without repeating invitations.",
      "questions": [
        "Preview two invitations, then execute after approval.",
        "The operation failed. Inspect pending, sent and failed state; do not retry."
      ],
      "question_note": "Scenario wording anonymized; action sequences summarized where needed.",
      "statuses": {
        "final": [
          true,
          true,
          true
        ],
        "control": [
          false,
          false,
          false
        ],
        "qwen": [
          false,
          false,
          false
        ],
        "deepseek": [
          false,
          null,
          null
        ]
      },
      "notes": {
        "final": [
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric."
        ],
        "control": [
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts."
        ],
        "qwen": [
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts."
        ],
        "deepseek": [
          "Failed with a recorded provider, response or completion error. The failed attempt is retained.",
          "Not attempted. Missing runs do not establish success or repeatability.",
          "Not attempted. Missing runs do not establish success or repeatability."
        ]
      }
    },
    {
      "id": "A6",
      "title": "Draft, edit, audit, launch",
      "expected": "Create and edit the intended draft; report author history and refuse an incomplete launch.",
      "questions": [
        "Create an empty program draft.",
        "Rename that draft and show who changed it.",
        "Try to launch it; do not invent required settings."
      ],
      "question_note": "Scenario wording anonymized; action sequences summarized where needed.",
      "statuses": {
        "final": [
          true,
          true,
          true
        ],
        "control": [
          true,
          true,
          true
        ],
        "qwen": [
          false,
          true,
          true
        ],
        "deepseek": [
          false,
          null,
          null
        ]
      },
      "notes": {
        "final": [
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric."
        ],
        "control": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation."
        ],
        "qwen": [
          "Failed with a recorded provider, response or completion error. The failed attempt is retained.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation."
        ],
        "deepseek": [
          "Failed with a recorded provider, response or completion error. The failed attempt is retained.",
          "Not attempted. Missing runs do not establish success or repeatability.",
          "Not attempted. Missing runs do not establish success or repeatability."
        ]
      }
    },
    {
      "id": "A7",
      "title": "Remember, correct, forget",
      "expected": "Recall an updated preference across sessions and delete only that preference.",
      "questions": [
        "Remember that I prefer creators who physically test products rather than only mention them.",
        "What is my preference about how creators feature products?",
        "Correct that preference: demonstrations are preferred, but detailed hands-on reviews also count.",
        "What is my updated preference?",
        "Forget that creator-content preference only.",
        "Do you still have my creator-content preference saved?"
      ],
      "question_note": "Scenario wording anonymized; action sequences summarized where needed.",
      "statuses": {
        "final": [
          true,
          true,
          true
        ],
        "control": [
          true,
          true,
          false
        ],
        "qwen": [
          true,
          true,
          true
        ],
        "deepseek": [
          null,
          null,
          null
        ]
      },
      "notes": {
        "final": [
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric."
        ],
        "control": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts."
        ],
        "qwen": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation."
        ],
        "deepseek": [
          "Not attempted. Missing runs do not establish success or repeatability.",
          "Not attempted. Missing runs do not establish success or repeatability.",
          "Not attempted. Missing runs do not establish success or repeatability."
        ]
      }
    },
    {
      "id": "A8",
      "title": "Generate & save settings",
      "expected": "Draft then persist the approved brief and keywords; preserve commission and lifecycle.",
      "questions": [
        "Generate vetting criteria and discovery keywords for Program A, focused on creators who demonstrate outdoor hydration products.",
        "Save those vetting criteria and keywords to that program. Leave the program's commission and lifecycle alone."
      ],
      "question_note": "Scenario wording anonymized; action sequences summarized where needed.",
      "statuses": {
        "final": [
          true,
          true,
          true
        ],
        "control": [
          false,
          false,
          false
        ],
        "qwen": [
          false,
          false,
          false
        ],
        "deepseek": [
          null,
          null,
          null
        ]
      },
      "notes": {
        "final": [
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric."
        ],
        "control": [
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts."
        ],
        "qwen": [
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts."
        ],
        "deepseek": [
          "Not attempted. Missing runs do not establish success or repeatability.",
          "Not attempted. Missing runs do not establish success or repeatability.",
          "Not attempted. Missing runs do not establish success or repeatability."
        ]
      }
    },
    {
      "id": "A9",
      "title": "Changes & authorship",
      "expected": "Report changes in the requested window and attribute only supported actions.",
      "questions": [
        "What changed across our programs since 2026-07-19?",
        "Who made those changes?"
      ],
      "question_note": "Scenario wording anonymized; action sequences summarized where needed.",
      "statuses": {
        "final": [
          false,
          true,
          true
        ],
        "control": [
          false,
          false,
          false
        ],
        "qwen": [
          true,
          true,
          true
        ],
        "deepseek": [
          null,
          null,
          null
        ]
      },
      "notes": {
        "final": [
          "The answer claimed its summary was missing a split despite listing all nine splits.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric.",
          "Passed evidence review across the required turns and applicable isolated action-state checks. Minor wording caveats were permitted under the shared rubric."
        ],
        "control": [
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts."
        ],
        "qwen": [
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation.",
          "Passed the earlier shared-loop evidence review. This used a different harness from the final DeepSeek confirmation."
        ],
        "deepseek": [
          "Not attempted. Missing runs do not establish success or repeatability.",
          "Not attempted. Missing runs do not establish success or repeatability.",
          "Not attempted. Missing runs do not establish success or repeatability."
        ]
      }
    },
    {
      "id": "A10",
      "title": "Sales + content evidence",
      "expected": "Inspect available creator content to distinguish demonstrations from mentions.",
      "questions": [
        "Which of our strong sellers actually demonstrate products rather than just mention them? Use sales and available content evidence."
      ],
      "question_note": "Scenario wording anonymized; action sequences summarized where needed.",
      "statuses": {
        "final": [
          false,
          false,
          false
        ],
        "control": [
          false,
          false,
          false
        ],
        "qwen": [
          false,
          false,
          false
        ],
        "deepseek": [
          null,
          null,
          null
        ]
      },
      "notes": {
        "final": [
          "The aggregate content reader was unavailable. The agent stopped without inspecting available individual creator evidence.",
          "The aggregate content reader was unavailable. The agent stopped without inspecting available individual creator evidence.",
          "The aggregate content reader was unavailable. The agent stopped without inspecting available individual creator evidence."
        ],
        "control": [
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts."
        ],
        "qwen": [
          "Failed with a recorded provider, response or completion error. The failed attempt is retained.",
          "Failed the earlier whole-workflow evidence review. The required task was incomplete or a material claim was unsupported; this public edition omits raw transcripts.",
          "Failed with a recorded provider, response or completion error. The failed attempt is retained."
        ],
        "deepseek": [
          "Not attempted. Missing runs do not establish success or repeatability.",
          "Not attempted. Missing runs do not establish success or repeatability.",
          "Not attempted. Missing runs do not establish success or repeatability."
        ]
      }
    }
  ],
  "confirmation": {
    "passes": 81,
    "total": 90,
    "percent": 90.0,
    "goal_met": true,
    "runs": [
      {
        "label": "Run 1",
        "passed": 27,
        "total": 30
      },
      {
        "label": "Run 2",
        "passed": 27,
        "total": 30
      },
      {
        "label": "Run 3",
        "passed": 27,
        "total": 30
      }
    ],
    "known_batch_usd": 11.544792876,
    "unknown_held_usd": 1.0244032,
    "accounted_batch_usd": 12.569196076,
    "confirmation_main_usd": 1.42094106,
    "confirmation_operational_usd": 1.5182400600000001,
    "confirmation_grading_usd": 3.660114,
    "median_first_text_s": 13.875,
    "median_completion_s": 16.046,
    "categories": {
      "reporting": {
        "passed": 20,
        "total": 21
      },
      "recruiting": {
        "passed": 40,
        "total": 42
      },
      "programs": {
        "passed": 10,
        "total": 12
      },
      "discovery": {
        "passed": 3,
        "total": 3
      },
      "memory": {
        "passed": 3,
        "total": 3
      },
      "vetting": {
        "passed": 3,
        "total": 3
      },
      "audit": {
        "passed": 2,
        "total": 3
      },
      "synthesis": {
        "passed": 0,
        "total": 3
      }
    }
  },
  "provenance": {
    "final": "round7/confirmation-review.json",
    "model_comparison": "round2/summary.json",
    "shortlist": "round3/summary.json (baseline reused from round2)",
    "guidance": "round4/results.md",
    "interrupted_bundle": "round5/results.md",
    "diagnostics": "round6/tool-response-findings.md"
  },
  "privacy": "No raw answers, tool payloads, customer records, source code or credentials included."
}
