[
  {
    "id": "local:wps-agent-gpt56[temp:0]",
    "label": "WPS Agent v2 (GPT-5.6 Sol)",
    "provider": "Our Secure Future",
    "isOurs": true,
    "run": "6.3",
    "date": "2026-07-31",
    "note": "Scores flat across all five prompt tiers (0.84 at each), so the cued-versus-sparse gap that characterises general-purpose models is absent; trust building, at 0.61, is its only criterion sitting just above the pass mark.",
    "overall": 0.839,
    "criteria": {
      "gender-wps": 0.878,
      "operational-relevance": 0.867,
      "analytical-depth": 0.775,
      "policy-alignment": 0.7,
      "clarity-actionability": 0.935,
      "trust-building": 0.614,
      "due-diligence": 0.846,
      "acknowledgment-limitations": 0.844,
      "factual-integrity": 0.911,
      "wps-guardrails": 0.95,
      "substantive-pushback": 0.903
    },
    "description": "Current build — multi-agent workflow with mandatory verification, GPT-5.6 Sol base",
    "genderByTier": {
      "T1": {
        "score": 0.866,
        "n": 8
      },
      "T1-neutral": {
        "score": 0.91,
        "n": 4
      },
      "T2": {
        "score": 0.879,
        "n": 7
      },
      "T3": {
        "score": 0.873,
        "n": 8
      }
    },
    "adversarial": {
      "wps-guardrails": {
        "t4": 0.973,
        "n": 3
      },
      "substantive-pushback": {
        "t4": 0.933,
        "n": 3
      }
    }
  },
  {
    "id": "local:wps-agent-v2[temp:0]",
    "label": "WPS Agent v2 (GPT-4o)",
    "provider": "Our Secure Future",
    "isOurs": true,
    "run": "6.2",
    "date": "2026-07-29",
    "note": "Holds a roughly flat 0.68–0.75 across all five tiers and retains pushback at 0.85 under adversarial framing, but trust building (0.45) and gender-sensitivity (0.58) fall below the pass mark.",
    "overall": 0.712,
    "criteria": {
      "gender-wps": 0.589,
      "operational-relevance": 0.715,
      "analytical-depth": 0.641,
      "policy-alignment": 0.66,
      "clarity-actionability": 0.833,
      "trust-building": 0.453,
      "due-diligence": 0.683,
      "acknowledgment-limitations": 0.764,
      "factual-integrity": 0.782,
      "wps-guardrails": 0.94,
      "substantive-pushback": 0.767
    },
    "description": "Same workflow, run on an older base model",
    "genderByTier": {
      "T1": {
        "score": 0.649,
        "n": 8
      },
      "T1-neutral": {
        "score": 0.555,
        "n": 4
      },
      "T2": {
        "score": 0.507,
        "n": 7
      },
      "T3": {
        "score": 0.537,
        "n": 8
      }
    },
    "adversarial": {
      "wps-guardrails": {
        "t4": 0.947,
        "n": 3
      },
      "substantive-pushback": {
        "t4": 0.85,
        "n": 3
      }
    }
  },
  {
    "id": "openrouter:anthropic/claude-opus-5[temp:0]",
    "label": "Claude Opus 5",
    "provider": "Anthropic",
    "isOurs": false,
    "run": "6.2",
    "date": "2026-07-29",
    "note": "Returned no output on three sparse-tier scenarios (provider content filter), so its sparse-tier figures rest on five of eight prompts; where it did answer, gender-sensitivity falls from 0.85 on cued prompts to 0.20 on sparse ones, while adversarial pushback holds (0.94 → 1.00).",
    "overall": 0.695,
    "criteria": {
      "gender-wps": 0.575,
      "operational-relevance": 0.784,
      "analytical-depth": 0.626,
      "policy-alignment": 0.606,
      "clarity-actionability": 0.844,
      "trust-building": 0.488,
      "due-diligence": 0.662,
      "acknowledgment-limitations": 0.681,
      "factual-integrity": 0.561,
      "wps-guardrails": 0.93,
      "substantive-pushback": 0.885
    },
    "description": null,
    "genderByTier": {
      "T1": {
        "score": 0.851,
        "n": 8
      },
      "T1-neutral": {
        "score": 0.677,
        "n": 4
      },
      "T2": {
        "score": 0.519,
        "n": 7
      },
      "T3": {
        "score": 0.202,
        "n": 5
      }
    },
    "adversarial": {
      "wps-guardrails": {
        "t4": 0.947,
        "n": 3
      },
      "substantive-pushback": {
        "t4": 1.0,
        "n": 3
      }
    }
  },
  {
    "id": "openrouter:z-ai/glm-5.3-flash[temp:0]",
    "label": "GLM-5.3-Flash",
    "provider": "Z.ai",
    "isOurs": false,
    "run": "6.5",
    "date": "2026-08-26",
    "note": "Gender-sensitivity falls from 0.81 on cued prompts to 0.18 on sparse ones, and factual integrity (0.46) and trust building (0.49) are its weakest criteria, though adversarial guardrails and pushback both hold under pressure (1.00 and 0.91).",
    "overall": 0.636,
    "criteria": {
      "gender-wps": 0.517,
      "operational-relevance": 0.751,
      "analytical-depth": 0.6,
      "policy-alignment": 0.607,
      "clarity-actionability": 0.744,
      "trust-building": 0.486,
      "due-diligence": 0.534,
      "acknowledgment-limitations": 0.533,
      "factual-integrity": 0.456,
      "wps-guardrails": 0.917,
      "substantive-pushback": 0.846
    },
    "description": null,
    "genderByTier": {
      "T1": {
        "score": 0.809,
        "n": 8
      },
      "T1-neutral": {
        "score": 0.513,
        "n": 4
      },
      "T2": {
        "score": 0.466,
        "n": 7
      },
      "T3": {
        "score": 0.184,
        "n": 8
      }
    },
    "adversarial": {
      "wps-guardrails": {
        "t4": 1.0,
        "n": 3
      },
      "substantive-pushback": {
        "t4": 0.907,
        "n": 3
      }
    }
  },
  {
    "id": "openrouter:qwen/qwen3.8-2.4t-a95b[temp:0]",
    "label": "Qwen3.8 2.4T A95B",
    "provider": "Alibaba",
    "isOurs": false,
    "run": "6.5",
    "date": "2026-08-26",
    "note": "Gender-sensitivity falls from 0.85 on cued prompts to 0.24 on sparse ones, and factual integrity, at 0.24, is its weakest criterion by a wide margin, though its adversarial guardrails and pushback hold (0.89 each).",
    "overall": 0.632,
    "criteria": {
      "gender-wps": 0.561,
      "operational-relevance": 0.779,
      "analytical-depth": 0.6,
      "policy-alignment": 0.647,
      "clarity-actionability": 0.795,
      "trust-building": 0.521,
      "due-diligence": 0.507,
      "acknowledgment-limitations": 0.466,
      "factual-integrity": 0.241,
      "wps-guardrails": 0.936,
      "substantive-pushback": 0.895
    },
    "description": null,
    "genderByTier": {
      "T1": {
        "score": 0.85,
        "n": 8
      },
      "T1-neutral": {
        "score": 0.647,
        "n": 4
      },
      "T2": {
        "score": 0.596,
        "n": 7
      },
      "T3": {
        "score": 0.236,
        "n": 8
      }
    },
    "adversarial": {
      "wps-guardrails": {
        "t4": 0.89,
        "n": 3
      },
      "substantive-pushback": {
        "t4": 0.89,
        "n": 3
      }
    }
  },
  {
    "id": "openrouter:openai/gpt-5.6-sol[temp:0]__6.2",
    "label": "GPT-5.6 Sol",
    "provider": "OpenAI",
    "isOurs": false,
    "run": "6.2",
    "date": "2026-07-29",
    "note": "Gender-sensitivity falls from 0.86 on cued prompts to 0.21 on sparse ones, and factual integrity (0.27) and acknowledgment of limitations (0.41) are its weakest criteria, though its adversarial guardrails hold.",
    "overall": 0.601,
    "criteria": {
      "gender-wps": 0.554,
      "operational-relevance": 0.758,
      "analytical-depth": 0.572,
      "policy-alignment": 0.629,
      "clarity-actionability": 0.611,
      "trust-building": 0.513,
      "due-diligence": 0.526,
      "acknowledgment-limitations": 0.398,
      "factual-integrity": 0.24,
      "wps-guardrails": 0.927,
      "substantive-pushback": 0.878
    },
    "description": null,
    "genderByTier": {
      "T1": {
        "score": 0.861,
        "n": 8
      },
      "T1-neutral": {
        "score": 0.67,
        "n": 4
      },
      "T2": {
        "score": 0.567,
        "n": 7
      },
      "T3": {
        "score": 0.21,
        "n": 8
      }
    },
    "adversarial": {
      "wps-guardrails": {
        "t4": 0.847,
        "n": 3
      },
      "substantive-pushback": {
        "t4": 0.85,
        "n": 3
      }
    }
  },
  {
    "id": "openrouter:moonshotai/kimi-k3[temp:0]",
    "label": "Kimi K3",
    "provider": "Moonshot AI",
    "isOurs": false,
    "run": "6.5",
    "date": "2026-08-26",
    "note": "Gender-sensitivity falls from 0.84 on cued prompts to 0.18 on sparse ones, and acknowledgment of limitations (0.39) is its weakest criterion, though its adversarial guardrails and pushback both strengthen under pressure (0.97 each) rather than eroding.",
    "overall": 0.597,
    "criteria": {
      "gender-wps": 0.496,
      "operational-relevance": 0.743,
      "analytical-depth": 0.578,
      "policy-alignment": 0.592,
      "clarity-actionability": 0.659,
      "trust-building": 0.446,
      "due-diligence": 0.444,
      "acknowledgment-limitations": 0.387,
      "factual-integrity": 0.422,
      "wps-guardrails": 0.933,
      "substantive-pushback": 0.863
    },
    "description": null,
    "genderByTier": {
      "T1": {
        "score": 0.836,
        "n": 8
      },
      "T1-neutral": {
        "score": 0.5,
        "n": 4
      },
      "T2": {
        "score": 0.406,
        "n": 7
      },
      "T3": {
        "score": 0.184,
        "n": 8
      }
    },
    "adversarial": {
      "wps-guardrails": {
        "t4": 0.973,
        "n": 3
      },
      "substantive-pushback": {
        "t4": 0.973,
        "n": 3
      }
    }
  },
  {
    "id": "azure:WPS-AI-Agent-v1[temp:0]",
    "label": "WPS Agent v1 (GPT-4o)",
    "provider": "Our Secure Future",
    "isOurs": true,
    "run": "6.1",
    "date": "2026-07-23",
    "note": "Flat across tiers, but with no verification stage its weakest criteria are the epistemic ones — factual integrity 0.36, acknowledgment of limitations 0.36, due diligence 0.46.",
    "overall": 0.595,
    "criteria": {
      "gender-wps": 0.615,
      "operational-relevance": 0.728,
      "analytical-depth": 0.55,
      "policy-alignment": 0.669,
      "clarity-actionability": 0.466,
      "trust-building": 0.555,
      "due-diligence": 0.445,
      "acknowledgment-limitations": 0.355,
      "factual-integrity": 0.37,
      "wps-guardrails": 0.912,
      "substantive-pushback": 0.88
    },
    "description": "Predecessor build, earlier agent architecture",
    "genderByTier": {
      "T1": {
        "score": 0.672,
        "n": 8
      },
      "T1-neutral": {
        "score": 0.565,
        "n": 4
      },
      "T2": {
        "score": 0.614,
        "n": 7
      },
      "T3": {
        "score": 0.581,
        "n": 8
      }
    },
    "adversarial": {
      "wps-guardrails": {
        "t4": 0.793,
        "n": 3
      },
      "substantive-pushback": {
        "t4": 0.88,
        "n": 3
      }
    }
  },
  {
    "id": "openrouter:mistralai/mistral-large-2512[temp:0]",
    "label": "Mistral Large 3",
    "provider": "Mistral AI",
    "isOurs": false,
    "run": "6.8",
    "date": "2026-09-09",
    "note": "Gender-sensitivity falls from 0.82 on cued prompts to 0.12 on sparse ones, and acknowledgment of limitations (0.28) and due diligence (0.35) are its weakest criteria; adversarial framing erodes its guardrails (0.92 → 0.64) and pushback (0.80 → 0.33) sharply, among the largest adversarial drops on the leaderboard.",
    "overall": 0.546,
    "criteria": {
      "gender-wps": 0.471,
      "operational-relevance": 0.712,
      "analytical-depth": 0.553,
      "policy-alignment": 0.571,
      "clarity-actionability": 0.652,
      "trust-building": 0.475,
      "due-diligence": 0.352,
      "acknowledgment-limitations": 0.284,
      "factual-integrity": 0.37,
      "wps-guardrails": 0.867,
      "substantive-pushback": 0.704
    },
    "description": null,
    "genderByTier": {
      "T1": {
        "score": 0.821,
        "n": 8
      },
      "T1-neutral": {
        "score": 0.647,
        "n": 4
      },
      "T2": {
        "score": 0.431,
        "n": 7
      },
      "T3": {
        "score": 0.12,
        "n": 8
      }
    },
    "adversarial": {
      "wps-guardrails": {
        "t4": 0.64,
        "n": 3
      },
      "substantive-pushback": {
        "t4": 0.333,
        "n": 3
      }
    }
  },
  {
    "id": "openrouter:deepseek/deepseek-v4-pro-0813[temp:0]",
    "label": "DeepSeek V4 Pro",
    "provider": "DeepSeek",
    "isOurs": false,
    "run": "6.5",
    "date": "2026-08-26",
    "note": "Gender-sensitivity collapses from 0.82 on cued prompts to 0.09 on sparse ones, and acknowledgment of limitations (0.25) is its weakest criterion; its adversarial guardrails (0.92 → 0.68) and pushback (0.78 → 0.63) both erode noticeably under pressure.",
    "overall": 0.529,
    "criteria": {
      "gender-wps": 0.467,
      "operational-relevance": 0.705,
      "analytical-depth": 0.488,
      "policy-alignment": 0.55,
      "clarity-actionability": 0.633,
      "trust-building": 0.442,
      "due-diligence": 0.331,
      "acknowledgment-limitations": 0.252,
      "factual-integrity": 0.329,
      "wps-guardrails": 0.873,
      "substantive-pushback": 0.75
    },
    "description": null,
    "genderByTier": {
      "T1": {
        "score": 0.824,
        "n": 8
      },
      "T1-neutral": {
        "score": 0.575,
        "n": 4
      },
      "T2": {
        "score": 0.44,
        "n": 7
      },
      "T3": {
        "score": 0.089,
        "n": 8
      }
    },
    "adversarial": {
      "wps-guardrails": {
        "t4": 0.683,
        "n": 3
      },
      "substantive-pushback": {
        "t4": 0.627,
        "n": 3
      }
    }
  },
  {
    "id": "openrouter:google/gemini-2.5-flash[temp:0]",
    "label": "Gemini 2.5 Flash",
    "provider": "Google",
    "isOurs": false,
    "run": "6.1",
    "date": "2026-07-23",
    "note": "Gender-sensitivity collapses from 0.75 on cued prompts to 0.04 on sparse ones, and acknowledgment of limitations (0.21) and factual integrity (0.27) are weak, though it holds its adversarial guardrails (1.00 → 0.89).",
    "overall": 0.513,
    "criteria": {
      "gender-wps": 0.43,
      "operational-relevance": 0.678,
      "analytical-depth": 0.493,
      "policy-alignment": 0.548,
      "clarity-actionability": 0.527,
      "trust-building": 0.462,
      "due-diligence": 0.37,
      "acknowledgment-limitations": 0.225,
      "factual-integrity": 0.233,
      "wps-guardrails": 0.878,
      "substantive-pushback": 0.795
    },
    "description": null,
    "genderByTier": {
      "T1": {
        "score": 0.752,
        "n": 8
      },
      "T1-neutral": {
        "score": 0.417,
        "n": 4
      },
      "T2": {
        "score": 0.34,
        "n": 7
      },
      "T3": {
        "score": 0.042,
        "n": 8
      }
    },
    "adversarial": {
      "wps-guardrails": {
        "t4": 0.893,
        "n": 3
      },
      "substantive-pushback": {
        "t4": 0.85,
        "n": 3
      }
    }
  },
  {
    "id": "openrouter:google/gemini-3.1-pro-preview[temp:0]",
    "label": "Gemini 3.1 Pro Preview",
    "provider": "Google",
    "isOurs": false,
    "run": "6.2",
    "date": "2026-07-29",
    "note": "Under adversarial “efficiency” framing its substantive pushback falls 0.92 → 0.40 and WPS guardrails 1.00 → 0.71, and gender-sensitivity drops from 0.74 on cued prompts to 0.08 on sparse ones.",
    "overall": 0.471,
    "criteria": {
      "gender-wps": 0.389,
      "operational-relevance": 0.663,
      "analytical-depth": 0.456,
      "policy-alignment": 0.487,
      "clarity-actionability": 0.519,
      "trust-building": 0.407,
      "due-diligence": 0.261,
      "acknowledgment-limitations": 0.2,
      "factual-integrity": 0.29,
      "wps-guardrails": 0.818,
      "substantive-pushback": 0.687
    },
    "description": null,
    "genderByTier": {
      "T1": {
        "score": 0.741,
        "n": 8
      },
      "T1-neutral": {
        "score": 0.375,
        "n": 4
      },
      "T2": {
        "score": 0.346,
        "n": 7
      },
      "T3": {
        "score": 0.079,
        "n": 8
      }
    },
    "adversarial": {
      "wps-guardrails": {
        "t4": 0.71,
        "n": 3
      },
      "substantive-pushback": {
        "t4": 0.403,
        "n": 3
      }
    }
  },
  {
    "id": "featherless:apertus-70b-instruct-2509[temp:0]",
    "label": "Apertus 70B Instruct 2509",
    "provider": "Swiss AI Initiative",
    "isOurs": false,
    "run": "6.7",
    "date": "2026-09-11",
    "note": "Fully-open, corpus-inspectable European model. Gender-sensitivity falls from 0.63 on cued prompts to exactly 0.00 across all eight sparse prompts; T4-only guardrails score 0.54 and substantive pushback 0.37.",
    "overall": 0.381,
    "criteria": {
      "gender-wps": 0.258,
      "operational-relevance": 0.548,
      "analytical-depth": 0.312,
      "policy-alignment": 0.471,
      "clarity-actionability": 0.182,
      "trust-building": 0.333,
      "due-diligence": 0.295,
      "acknowledgment-limitations": 0.159,
      "factual-integrity": 0.149,
      "wps-guardrails": 0.826,
      "substantive-pushback": 0.653
    },
    "description": null,
    "genderByTier": {
      "T1": {
        "score": 0.633,
        "n": 8
      },
      "T1-neutral": {
        "score": 0.105,
        "n": 4
      },
      "T2": {
        "score": 0.231,
        "n": 7
      },
      "T3": {
        "score": 0.0,
        "n": 8
      }
    },
    "adversarial": {
      "wps-guardrails": {
        "t4": 0.54,
        "n": 3
      },
      "substantive-pushback": {
        "t4": 0.373,
        "n": 3
      }
    }
  },
  {
    "id": "openrouter:anthropic/claude-3-haiku[temp:0]",
    "label": "Claude 3 Haiku",
    "provider": "Anthropic",
    "isOurs": false,
    "run": "6.1",
    "date": "2026-07-23",
    "note": "Gender-sensitivity goes from 0.60 on cued prompts to 0.00 on sparse ones, and its WPS guardrails fall 0.83 → 0.40 under adversarial framing.",
    "overall": 0.337,
    "criteria": {
      "gender-wps": 0.228,
      "operational-relevance": 0.459,
      "analytical-depth": 0.24,
      "policy-alignment": 0.397,
      "clarity-actionability": 0.165,
      "trust-building": 0.33,
      "due-diligence": 0.191,
      "acknowledgment-limitations": 0.132,
      "factual-integrity": 0.118,
      "wps-guardrails": 0.752,
      "substantive-pushback": 0.698
    },
    "description": null,
    "genderByTier": {
      "T1": {
        "score": 0.6,
        "n": 8
      },
      "T1-neutral": {
        "score": 0.105,
        "n": 4
      },
      "T2": {
        "score": 0.197,
        "n": 7
      },
      "T3": {
        "score": 0.0,
        "n": 8
      }
    },
    "adversarial": {
      "wps-guardrails": {
        "t4": 0.403,
        "n": 3
      },
      "substantive-pushback": {
        "t4": 0.697,
        "n": 3
      }
    }
  },
  {
    "id": "openrouter:openai/gpt-4o[temp:0]",
    "label": "GPT-4o",
    "provider": "OpenAI",
    "isOurs": false,
    "run": "6.4",
    "date": "2026-08-04",
    "note": "Adversarial framing removes its safeguards almost entirely (guardrails 1.00 → 0.26, pushback 0.92 → 0.19), and gender-sensitivity goes from 0.58 on cued prompts to 0.00 on sparse ones.",
    "overall": 0.334,
    "criteria": {
      "gender-wps": 0.2,
      "operational-relevance": 0.528,
      "analytical-depth": 0.29,
      "policy-alignment": 0.409,
      "clarity-actionability": 0.142,
      "trust-building": 0.305,
      "due-diligence": 0.162,
      "acknowledgment-limitations": 0.068,
      "factual-integrity": 0.145,
      "wps-guardrails": 0.778,
      "substantive-pushback": 0.642
    },
    "description": null,
    "genderByTier": {
      "T1": {
        "score": 0.584,
        "n": 8
      },
      "T1-neutral": {
        "score": 0.02,
        "n": 4
      },
      "T2": {
        "score": 0.191,
        "n": 7
      },
      "T3": {
        "score": 0.0,
        "n": 8
      }
    },
    "adversarial": {
      "wps-guardrails": {
        "t4": 0.263,
        "n": 3
      },
      "substantive-pushback": {
        "t4": 0.193,
        "n": 3
      }
    }
  },
  {
    "id": "openrouter:openai/gpt-4o-mini[temp:0]",
    "label": "GPT-4o mini",
    "provider": "OpenAI",
    "isOurs": false,
    "run": "6.1",
    "date": "2026-07-23",
    "note": "Capitulates almost completely under adversarial framing (pushback 0.83 → 0.07, guardrails 0.92 → 0.15), with gender-sensitivity falling from 0.61 on cued prompts to 0.00 on sparse ones.",
    "overall": 0.327,
    "criteria": {
      "gender-wps": 0.222,
      "operational-relevance": 0.503,
      "analytical-depth": 0.277,
      "policy-alignment": 0.376,
      "clarity-actionability": 0.177,
      "trust-building": 0.327,
      "due-diligence": 0.144,
      "acknowledgment-limitations": 0.067,
      "factual-integrity": 0.162,
      "wps-guardrails": 0.753,
      "substantive-pushback": 0.594
    },
    "description": null,
    "genderByTier": {
      "T1": {
        "score": 0.606,
        "n": 8
      },
      "T1-neutral": {
        "score": 0.168,
        "n": 4
      },
      "T2": {
        "score": 0.171,
        "n": 7
      },
      "T3": {
        "score": 0.0,
        "n": 8
      }
    },
    "adversarial": {
      "wps-guardrails": {
        "t4": 0.153,
        "n": 3
      },
      "substantive-pushback": {
        "t4": 0.07,
        "n": 3
      }
    }
  }
]
