{
  "title": "Beyond 0-100: How Confidence Scale Design Shapes Verbalized Confidence and LLM Metacognition",
  "type": "ScholarlyArticle",
  "status": "preprint",
  "year": 2026,
  "authors": [
    {"name": "Yuyang Dai", "affiliation": "INSAIT", "email": "y9657422@gmail.com"},
    {"name": "Yuxia Wang", "affiliation": "INSAIT", "email": "yuxia.wang@insait.ai"}
  ],
  "keywords": [
    "large language models",
    "LLM calibration",
    "verbalized confidence",
    "confidence scale",
    "metacognition",
    "uncertainty estimation",
    "expected calibration error",
    "meta-d-prime"
  ],
  "datasets": ["MMLU", "GSM8K", "TruthfulQA"],
  "primary_model_count": 6,
  "additional_robustness_models": ["Llama-3-8B-Instruct"],
  "key_findings": [
    "Under the standard 0-100 scale, the three most frequent values account for 78.2%-92.1% of responses.",
    "Models use only 15-28 of the 101 available integer confidence values.",
    "Within the evaluated settings, a 0-20 scale yields higher metacognitive efficiency than 0-100.",
    "Aggressive lower-bound compression degrades metacognitive performance.",
    "Round-number preferences persist under irregular numerical ranges."
  ],
  "limitations": "The experiments focus on three multiple-choice benchmarks. Generalization to open-ended generation, other prompts, and future model versions is not established."
}
