{
 "@context": "https://schema.org",
 "@type": "Claim",
 "@id": "https://wulfkaal.github.io/claims/4855607-018",
 "identifier": "kaal:claim:4855607-018",
 "text": "RLHF fails on several fronts at once: humans can pursue harmful goals either innocently or maliciously, human feedback degrades when examples are hard to evaluate and especially when RLHF is applied to superhuman models, and reward models diverge from humans through misspecification and misgeneralization.",
 "author": {
  "@type": "Person",
  "name": "Wulf A. Kaal",
  "identifier": "https://orcid.org/0000-0003-0757-275X"
 },
 "citation": "Wulf A. Kaal, How AI Models are Optimized Through Web3 Governance (2024). SSRN: https://ssrn.com/abstract=4855607",
 "isBasedOn": {
  "@type": "ScholarlyArticle",
  "name": "How AI Models are Optimized Through Web3 Governance",
  "datePublished": "2024",
  "url": "https://ssrn.com/abstract=4855607",
  "sha256": "eb0b3e62374b45a8fa888c6bde9725e606bcb46cf4b5e74a6e851d9f25099113",
  "encoding": "https://raw.githubusercontent.com/wulfkaal/Academic-Papers/main/papers/pdf/Kaal%20-%202024%20-%20How%20AI%20Models%20are%20Optimized%20Through%20Web3%20Governance.pdf"
 },
 "keywords": [
  "ai-and-agents"
 ],
 "about": [
  "rlhf",
  "scalable-oversight",
  "reward-misspecification",
  "misgeneralization",
  "alignment"
 ],
 "abstract": "Moreover, humans can pursue harmful goals, either innocently or maliciously, and can provide poor feedback when examples are hard to evaluate, especially when applying RLHF to superhuman models. Reward models can differ from humans due to misspecification and misgeneralization",
 "additionalProperty": [
  {
   "@type": "PropertyValue",
   "name": "claim_type",
   "value": "failure"
  },
  {
   "@type": "PropertyValue",
   "name": "confidence",
   "value": "evidenced"
  },
  {
   "@type": "PropertyValue",
   "name": "scope_conditions",
   "value": [
    "especially acute when RLHF is applied to models more capable than their human evaluators"
   ]
  },
  {
   "@type": "PropertyValue",
   "name": "is_failure_mode",
   "value": true
  },
  {
   "@type": "PropertyValue",
   "name": "attestations",
   "value": "https://wulfkaal.github.io/colloquium/attestations/acd93c15e8213d7da6ed93a3807610113a4f6957575392ecd7fabd2c449912b0.json"
  },
  {
   "@type": "PropertyValue",
   "name": "edges",
   "value": []
  }
 ],
 "isPartOf": {
  "@id": "https://wulfkaal.github.io/claims/index.json"
 },
 "dateModified": "2026-07-27",
 "version": "1.0",
 "sha256": "acd93c15e8213d7da6ed93a3807610113a4f6957575392ecd7fabd2c449912b0",
 "canonicalForm": "https://wulfkaal.github.io/claims/4855607-018.md"
}