{
 "updated": "2026-09-13T11:45:00+08:00",
 "models": {
  "team-junior-v1.1": {
   "profbench": {
    "date": "2026-09-10",
    "dataset": "ProfBench — 40 expert-level tasks, ten each in Physics PhD, Chemistry PhD, Finance MBA and Consulting MBA, scored criterion by criterion with the official rubric.",
    "protocol": "Self-run with the official rubric protocol: per-criterion Yes/No with the official prompt verbatim, temperature 0 / top_p 1, no output cap. Team delivery is content + reason. 38 of 40 tasks scored: two tasks were excluded because no answer was produced for them during the run (disclosed rather than imputed). The DeepSeek V4 Flash baseline and the official o3 draft were read by the same judging pipeline, and all three columns are reported on the 38 tasks every model completed.",
    "completed": 38,
    "total": 40,
    "overall": {
     "Team": 63.3,
     "DS": 57.4,
     "o3": 55.6
    },
    "delivery_note": "Team returns a complete answer body together with the reasoning the model states for it — content plus reason — rather than a bare conclusion. ProfBench scores every response criterion by criterion, derivation and intermediate results included, so an answer that shows its work is scored on the work itself.",
    "segments": [
     {
      "name": "Chemistry PhD",
      "n": 10,
      "Team": 75.5,
      "DS": 69.4,
      "o3": 60.8
     },
     {
      "name": "Consulting MBA",
      "n": 10,
      "Team": 70.8,
      "DS": 70.3,
      "o3": 64.9
     },
     {
      "name": "Finance MBA",
      "n": 10,
      "Team": 54.6,
      "DS": 48.2,
      "o3": 57.5
     },
     {
      "name": "Physics PhD",
      "n": 8,
      "Team": 49.6,
      "DS": 37.8,
      "o3": 35.2
     }
    ],
    "official_refs": {
     "overall": {
      "o3": 53.2,
      "grok4": 51.9,
      "r1-0528": 46.3
     },
     "per_domain": {
      "Chemistry PhD": {
       "o3": 51.6,
       "grok4": 67.9,
       "r1-0528": 48.2
      },
      "Consulting MBA": {
       "o3": 71.2,
       "grok4": 67.4,
       "r1-0528": 59.0
      },
      "Finance MBA": {
       "o3": 44.5,
       "grok4": 41.3,
       "r1-0528": 39.1
      },
      "Physics PhD": {
       "o3": 45.4,
       "grok4": 30.9,
       "r1-0528": 39.1
      }
     },
     "note": "Official evaluation of the official reference drafts, produced by the official judging pipeline. A different judge from the self-run columns above, so these rows are context only — not a head-to-head comparison."
    },
    "protocol_compare": "Team, the DeepSeek V4 Flash baseline and the official o3 draft were all read by the same judging pipeline (per-criterion Yes/No, official prompt, temperature 0, no output cap), and are reported on the 38 tasks that all three completed — so the three self-run columns are directly comparable.",
    "judge_validation": {
     "agreement": 74.0,
     "f1": 0.761,
     "mean_delta": 2.4,
     "note": "Our judging pipeline reproduces the official per-criterion labels on the o3 draft with 74.0% agreement (F1 0.761) and a mean score difference of +2.4 points — the numbers above sit on our pipeline's scale, calibrated against the official one."
    },
    "arbitration": {
     "team_picks": 36,
     "o3_picks": 1,
     "unresolved": 1,
     "compared": 38,
     "median_confidence": 85.9,
     "lead": "On the same 38 tasks we also asked our own decision kernel to choose between the team draft and the official o3 draft. It was shown the question and the two drafts and nothing else — no rubric, no scores — and it selected the team draft 36 times.",
     "checks": [
      "<strong>Position swap.</strong> On all 37 tasks where the kernel had made a pick, the comparison was re-run with the two drafts exchanged between option A and option B. 36 kept the same draft. The single change came on a task the kernel had already flagged as a near-tie (23% confidence); every pick it made at 80% confidence or above — 23 of the 37 — came back identical.",
      "<strong>No self-grading.</strong> The judgment comes from a panel of independent models rather than one model scoring its own output, which removes the self-preference bias of a single-judge setup."
     ],
     "note": "This head-to-head run is supporting evidence, not the primary measurement: the per-criterion rubric columns above are the measurement, and the kernel reached its conclusion without ever seeing them."
    },
    "same_pipeline": "All three self-run columns come from one judging pipeline on the same 38 tasks: Team 63.3 · DeepSeek V4 Flash (direct) 57.4 · official o3 draft 55.6.",
    "confidence": {
     "note": "The kernel reports a confidence value with every judgment, so a preference is never a bare claim. Median confidence on this benchmark is 85.9%, and it is not uniform across domains: highest on Physics PhD (95.3%) and Chemistry PhD (87.0%), lowest on Finance MBA (83.8%). It describes how far apart the two drafts are in the kernel's judgment: the lower it is, the closer the two are in quality — where either draft is a reasonable choice.",
     "median": 85.9,
     "by_domain": [
      {
       "name": "Chemistry PhD",
       "n": 10,
       "median": 87.0,
       "team_picks": 9
      },
      {
       "name": "Consulting MBA",
       "n": 10,
       "median": 84.5,
       "team_picks": 10
      },
      {
       "name": "Finance MBA",
       "n": 10,
       "median": 83.8,
       "team_picks": 9
      },
      {
       "name": "Physics PhD",
       "n": 8,
       "median": 95.3,
       "team_picks": 8
      }
     ],
     "method_note": "The 38-task run used a single fixed presentation order, with the team draft always offered as option A; the position-swap audit above, not this run, is the control for that."
    },
    "footnote": "Self-run with the official per-criterion rubric protocol (official prompt verbatim, temperature 0, top_p 1, no output cap). Our judging pipeline reproduces the official criterion labels on the o3 draft with 74.0% agreement and a mean difference of +2.4 points. Confidence describes how close the two drafts were in that judgment."
   }
  },
  "decider-junior-v1": {
   "judgebench": {
    "date": "2026-09-08",
    "dataset": "JudgeBench, 620 pairs (sources: MMLU-Pro / LiveBench Reasoning / Math / LiveCodeBench)",
    "protocol": "Self-run with the official judging protocol. Both columns use first successful verdict per pair; the six pairs whose first verdict failed are disclosed rather than imputed. Reference model measured on the same judged set.",
    "completed": 614,
    "total": 620,
    "accuracy": {
     "Decider": 92.5,
     "DeepSeek V4 Flash (direct)": 92.2,
     "agreement": 96.1
    },
    "segments": [
     {
      "name": "Knowledge (MMLU-Pro)",
      "n": 303,
      "Decider": 88.8,
      "DS": 87.8
     },
     {
      "name": "Reasoning (LiveBench)",
      "n": 149,
      "Decider": 98.0,
      "DS": 96.6
     },
     {
      "name": "Math (LiveBench)",
      "n": 90,
      "Decider": 92.2,
      "DS": 95.6
     },
     {
      "name": "Code (LiveCodeBench)",
      "n": 72,
      "Decider": 97.2,
      "DS": 97.2
     }
    ],
    "calibration": {
     "note": "Confidence is emitted with every judgment and calibrated against outcomes on this benchmark (JudgeBench, 2026-09-08). It describes how far apart the two candidates are — and the accuracy column shows what that delivered here. Set your own threshold from the accuracy column; for high-stakes or irreversible decisions, apply your own review policy.",
     "tiers": [
      {
       "tier": ">= 90%",
       "n": 283,
       "share_of_total": "45.6%",
       "accuracy": 99.6,
       "meaning": "one candidate clearly stronger",
       "use": "Act on it"
      },
      {
       "tier": "80-90%",
       "n": 184,
       "share_of_total": "29.7%",
       "accuracy": 94.0,
       "meaning": "one candidate stronger",
       "use": "Go with it"
      },
      {
       "tier": "70-80%",
       "n": 82,
       "share_of_total": "13.2%",
       "accuracy": 84.1,
       "meaning": "closer call",
       "use": "Act on it after a quick look"
      },
      {
       "tier": "< 70%",
       "n": 65,
       "share_of_total": "10.5%",
       "accuracy": 67.7,
       "meaning": "near-tie — evenly matched",
       "use": "Either choice is fine"
      }
     ]
    },
    "value_note": "Every judgment ships with a calibrated confidence value. A high value means the comparison was decisive and the pick can be acted on directly; a low value means the two are evenly matched, where either choice is defensible and the decision belongs to criteria outside the answers."
   },
   "cjb": {
    "date": "2026-09-09",
    "total_pairs": 2000,
    "completed_pairs": 1991,
    "dataset": "ContextualJudgeBench (Salesforce), full official set of 2,000 pairs across all 8 splits; official reference values from the paper's Table 2.",
    "protocol": "Self-run with the official vanilla pairwise protocol; consistent accuracy (both response orders judged correctly); random floor 25%; failures disclosed (12 orders rerun-excluded). Reference model measured on the same judged set.",
    "scope": "Full official 8-split run completed (2026-09-09). Consistent accuracy 67.1% vs reference 65.4%: the advantage is strongest on faithfulness and refusal tasks, while the deliberately near-tie splits sit in the 46-60% range, reflecting the benchmark's difficulty design. Confidence tiers below are calibrated on this benchmark.",
    "segments": [
     {
      "name": "Refusal (Answerable)",
      "pairs": 250,
      "Decider": 97.2,
      "DS": 94.4,
      "refs": {
       "o1": 96.0,
       "o3-mini": 95.2,
       "r1": 92.0
      }
     },
     {
      "name": "Faithfulness (QA)",
      "pairs": 250,
      "Decider": 92.3,
      "DS": 90.8,
      "refs": {
       "o1": 84.4,
       "o3-mini": 76.4,
       "r1": 72.0
      }
     },
     {
      "name": "Faithfulness (Summarization)",
      "pairs": 250,
      "Decider": 68.7,
      "DS": 67.2,
      "refs": {
       "o1": 59.2,
       "o3-mini": 58.0,
       "r1": 50.4
      }
     },
     {
      "name": "Completeness (Summarization)",
      "pairs": 251,
      "Decider": 66.8,
      "DS": 64.5,
      "refs": {
       "o1": 63.7,
       "o3-mini": 59.8,
       "r1": 60.6
      }
     },
     {
      "name": "Refusal (Unanswerable)",
      "pairs": 250,
      "Decider": 59.8,
      "DS": 62.8,
      "refs": {
       "o1": 48.4,
       "o3-mini": 34.4,
       "r1": 52.0
      }
     },
     {
      "name": "Completeness (QA)",
      "pairs": 250,
      "Decider": 52.2,
      "DS": 51.6,
      "refs": {
       "o1": 48.4,
       "o3-mini": 40.4,
       "r1": 41.2
      }
     },
     {
      "name": "Conciseness (QA)",
      "pairs": 255,
      "Decider": 53.7,
      "DS": 50.2,
      "refs": {
       "o1": 15.3,
       "o3-mini": 20.8,
       "r1": 20.4
      }
     },
     {
      "name": "Conciseness (Summarization)",
      "pairs": 244,
      "Decider": 46.1,
      "DS": 41.4,
      "refs": {
       "o1": 27.0,
       "o3-mini": 35.7,
       "r1": 26.2
      }
     }
    ],
    "calibration": {
     "note": "Confidence is emitted with every judgment and calibrated against outcomes on this benchmark (ContextualJudgeBench, 2026-09-09). It describes how far apart the two candidates are, and the accuracy column shows what that delivered here. This benchmark deliberately contains near-tie splits, so the accuracies run lower by design — the bands describe the comparison, not a promise about either answer.",
     "tiers": [
      {
       "tier": ">= 90%",
       "n": 789,
       "share_of_total": "19.8%",
       "accuracy": 83.3,
       "meaning": "one candidate clearly stronger",
       "use": "Act on it"
      },
      {
       "tier": "80-90%",
       "n": 1486,
       "share_of_total": "37.3%",
       "accuracy": 76.4,
       "meaning": "one candidate stronger",
       "use": "Go with it"
      },
      {
       "tier": "70-80%",
       "n": 1184,
       "share_of_total": "29.7%",
       "accuracy": 63.6,
       "meaning": "closer call",
       "use": "Act on it after a quick look"
      },
      {
       "tier": "< 70%",
       "n": 529,
       "share_of_total": "13.3%",
       "accuracy": 55.4,
       "meaning": "near-tie — evenly matched",
       "use": "Either choice is fine"
      }
     ]
    },
    "footnote": "Self-run with the official protocol over the full 2,000 official pairs. Consistent accuracy requires the same correct pick in both presentation orders (random floor 25%). 12 orders (0.3%) were rerun-excluded after repeated platform failures; the reference model was measured on the same judged set. Previously published results measured on a subset are archived in the repository changelog.",
    "overall": {
     "Decider": 67.1
    },
    "value_note": "Every judgment ships with a calibrated confidence value. A high value means the comparison was decisive and the pick can be acted on directly; a low value means the two are evenly matched, where either choice is defensible."
   }
  }
 },
 "site": {
  "paper": {
   "title": "Cross-Model Confidence in the MCL Framework: Calibrated Uncertainty for Selective Judgment",
   "doi": "10.6084/m9.figshare.33684823",
   "url": "https://doi.org/10.6084/m9.figshare.33684823"
  },
  "name": "TuringCorp Models",
  "tagline": "Proprietary collaborative inference. Each model is an OpenAI-compatible API backed by a multi-model pipeline: independent reasoning paths cross-examine each other before an answer is delivered, instead of a single forward pass.",
  "contact": "iAsk@turingcorp.net",
  "api": {
   "base_url": "https://api.turingcorp.net/v1",
   "models_endpoint": "https://api.turingcorp.net/v1/models",
   "access": "small-scale preview, by invitation — email iAsk@turingcorp.net",
   "notes": [
    "OpenAI-compatible: any OpenAI SDK or client works unmodified.",
    "Non-streaming only: `stream: true` returns HTTP 400.",
    "Decider answers a judging question: send the task plus two candidate answers and it returns the better option with a confidence value.",
    "Team answers an open question and returns the answer body together with the reasoning it states for it (content + reason)."
   ]
  },
  "model_labels": {
   "team-junior-v1.1": "Team Junior v1.1",
   "decider-junior-v1": "Decider Junior"
  },
  "products": [
   {
    "name": "Decider",
    "role": "AI judge: given a task and two candidate answers, picks the better one and reports a calibrated confidence value with every judgment.",
    "tiers": [
     {
      "tier": "Junior",
      "model_id": "turingcorp/decider-junior-v1",
      "positioning": "Reliable AI judge between two answers for everyday decisions, with a calibrated confidence signal."
     },
     {
      "tier": "Senior",
      "model_id": "turingcorp/decider-senior-v1",
      "positioning": "Rigorous AI judge of subtle differences for high-stakes decisions, with a calibrated confidence signal."
     }
    ]
   },
   {
    "name": "Team",
    "role": "Collaborative multi-path analysis: independent reasoning paths cross-examine each other, and the answer is delivered as content + reason.",
    "tiers": [
     {
      "tier": "Junior",
      "model_id": "turingcorp/team-junior-v1.1",
      "positioning": "Cost-effective collaborative analysis for everyday tasks."
     },
     {
      "tier": "Senior",
      "model_id": "turingcorp/team-senior-v1",
      "positioning": "Deeper reasoning with enhanced quality for important decisions."
     },
     {
      "tier": "Principal",
      "model_id": "turingcorp/team-principal-v1",
      "positioning": "Maximum analytical depth for mission-critical evaluations."
     }
    ]
   }
  ]
 }
}
