{
  "dataset": "veb-canonical-135-enriched",
  "description": "VEB canonical 135/model/track grid (13-model board) WITH the enrichment layer (deterministic behavioral features + LLM-judged, evidence-cited aiFeatures) joined onto every graded trajectory. Superset of veb-canonical-135 — trajectory + judge grade + reward + enrichment in one self-contained JSONL line per datapoint.",
  "base_dataset": "veb-canonical-135",
  "rows": 3510,
  "grid": {
    "scenarios": 9,
    "seeds": 15,
    "models": 13,
    "tracks": [
      "oob",
      "pack"
    ],
    "per_model_per_track": 135
  },
  "roster": [
    "claude-fable-5",
    "claude-opus-4-6",
    "claude-opus-4-8",
    "claude-sonnet-4-6",
    "gemini-3.1-pro-preview",
    "gemini-3.5-flash",
    "gpt-5.5",
    "gpt-5.6-sol",
    "grok-4.20-0309-reasoning",
    "grok-4.3",
    "grok-4.5",
    "moonshotai/kimi-k3",
    "thinkingmachines/inkling"
  ],
  "enrichment": {
    "join_key": "(scenario_id, model, track, seed)",
    "coverage": "3510/3510",
    "panel": "openai:gpt-5.6-sol + gemini:gemini-3.5-flash + anthropic:claude-sonnet-4-6 (3-seat cross-family ensemble)",
    "calibration": {
      "vs": "frontier reference panel on 139 canonical rows",
      "pearson_r": 0.894,
      "mad": 0.047,
      "categorical_exact_match_pct": {
        "buyer_posture": 89.9,
        "discovery_vs_pitch": 90.6,
        "value_defended": 98.6
      }
    },
    "deterministic_feature_count": 44,
    "ai_feature_count": 17,
    "deterministic_features": [
      "calls_dated_close_rate",
      "calls_held",
      "champion_tests",
      "competitor_probes",
      "curveball_count",
      "curveball_response_gap_turns",
      "dated_next_step_rate",
      "discount_events",
      "discount_max_pct",
      "discount_vs_tolerance",
      "eb_first_touch_week",
      "eb_meeting_held",
      "eb_meeting_requested",
      "emails_sent",
      "facts_per_call",
      "facts_unlocked",
      "facts_unlocked_frac",
      "factual_contradictions",
      "first_discount_week",
      "first_fact_week",
      "first_quantification_week",
      "internal_forwards",
      "map_acknowledged",
      "map_proposed",
      "max_idle_gap_weeks",
      "open_discovery_questions",
      "personas_engaged",
      "pitch_burst_max",
      "pitch_count",
      "premature_closes",
      "price_integrity",
      "process_questions",
      "quantifying_questions",
      "question_density",
      "rel__any_persona_walked",
      "rel__min_persona_patience",
      "rel__neglected_persona_count",
      "rel__persona_trust_spread",
      "talk_ratio",
      "temporal__mean_persona_trust_final",
      "temporal__weakest_persona_interest",
      "touches_used_frac",
      "wasted_meetings",
      "words_per_turn"
    ],
    "ai_features": [
      "biz__deal_read_accuracy",
      "biz__urgency_created",
      "content__discovery_vs_pitch",
      "content__value_specificity",
      "emo__buyer_frustration_peak",
      "ethic__value_defended",
      "ling__clarity",
      "psych__buyer_trust_read",
      "psych__seller_confidence",
      "psych__theory_of_mind_gap",
      "rel__buyer_posture",
      "rhet__objection_handling",
      "rhet__question_quality",
      "rl__counterfactual_lift",
      "rl__hidden_negativity",
      "rl__pivotal_turn_index",
      "rl__public_private_divergence"
    ]
  },
  "rows_by_model_track": {
    "claude-fable-5|oob": 135,
    "claude-fable-5|pack": 135,
    "claude-opus-4-6|oob": 135,
    "claude-opus-4-6|pack": 135,
    "claude-opus-4-8|oob": 135,
    "claude-opus-4-8|pack": 135,
    "claude-sonnet-4-6|oob": 135,
    "claude-sonnet-4-6|pack": 135,
    "gemini-3.1-pro-preview|oob": 135,
    "gemini-3.1-pro-preview|pack": 135,
    "gemini-3.5-flash|oob": 135,
    "gemini-3.5-flash|pack": 135,
    "gpt-5.5|oob": 135,
    "gpt-5.5|pack": 135,
    "gpt-5.6-sol|oob": 135,
    "gpt-5.6-sol|pack": 135,
    "grok-4.20-0309-reasoning|oob": 135,
    "grok-4.20-0309-reasoning|pack": 135,
    "grok-4.3|oob": 135,
    "grok-4.3|pack": 135,
    "grok-4.5|oob": 135,
    "grok-4.5|pack": 135,
    "moonshotai/kimi-k3|oob": 135,
    "moonshotai/kimi-k3|pack": 135,
    "thinkingmachines/inkling|oob": 135,
    "thinkingmachines/inkling|pack": 135
  },
  "sha256": {
    "veb-canonical-135-enriched.jsonl": "52fb511c09ca4333dc20714b9a2886a07fc348fa23fbb3b9634b77717b54783c",
    "veb-canonical-135-enriched.jsonl.gz": "b128ac45dd2ba098f8872f0b45660f88fa8e3324c0c4b0ce1953f0cd5c91a769",
    "veb-canonical-135-preview.jsonl": "0026daf17687327facc15d27e3b65d18a2d1ed38e1ae6911c3aca4b5b6add0b0"
  },
  "preview": {
    "file": "veb-canonical-135-preview.jsonl",
    "sha256": "0026daf17687327facc15d27e3b65d18a2d1ed38e1ae6911c3aca4b5b6add0b0",
    "rows": 14,
    "selection": "scenario cybersecurity-ciso, seed 1: every model out-of-box plus the pack arm of gpt-5.6-sol"
  },
  "row_schema_note": "Identical to veb-canonical-135 plus a top-level `enrichment` object: { schemaVersion, featureCount, sourceCounts, features, aiFeatures }.",
  "notes": "Open-weight endpoints kimi-k3 (Moonshot) and inkling (Thinking Machines) join the 11 closed-weight frontier models to form the 13-endpoint canonical roster; both are graded on every cell and ranked alongside the closed frontier. gpt-5.5-pro excluded (off-roster reasoning-only exploratory arm). One datapoint = one JSONL line, fully self-contained (episode+grade embedded)."
}
