{
  "@context": "https://schema.org",
  "@type": "Claim",
  "@id": "https://wulfkaal.github.io/claims/7261018-001",
  "identifier": "kaal:claim:7261018-001",
  "text": "The evaluation of multi-agent systems built from large language models has, to date, been an evaluation of capability under incentive-free conditions.",
  "author": {
    "@type": "Person",
    "name": "Wulf A. Kaal",
    "identifier": "https://orcid.org/0009-0008-7840-1847"
  },
  "citation": "Wulf A. Kaal, Empirical Evaluation of the Agentic Reputation Substrate: Deliberation, the Composition of Error, and the Registered Measurement of Agency Costs in a Controlled Multi-Model Cohort (2026). SSRN: https://ssrn.com/abstract=7261018",
  "isBasedOn": {
    "@type": "ScholarlyArticle",
    "name": "Empirical Evaluation of the Agentic Reputation Substrate: Deliberation, the Composition of Error, and the Registered Measurement of Agency Costs in a Controlled Multi-Model Cohort",
    "datePublished": "2026",
    "url": "https://ssrn.com/abstract=7261018",
    "sha256": "1d6cbe544bd0055133f7cf8ff308be4fde955867bd8dc764992b8d516af15fa8",
    "encoding": "https://raw.githubusercontent.com/wulfkaal/Academic-Papers/main/papers/pdf/Kaal%20-%202026%20-%20Empirical%20Evaluation%20of%20the%20Agentic%20Reputation%20Substrate%20Deliberation%2C%20the%20Composition%20of%20Error%2C%20and%20the%20Registered%20Measurement%20of%20Agency%20Costs%20in%20a%20Controlled%20Multi-Model%20Cohort.pdf"
  },
  "keywords": [
    "ai-and-agents",
    "research-methods"
  ],
  "about": [],
  "abstract": "The evaluation of multi-agent systems built from large language models has, to date, been an evaluation of capability under incentive-free conditions.",
  "additionalProperty": [
    {
      "@type": "PropertyValue",
      "name": "claim_type",
      "value": "condition"
    },
    {
      "@type": "PropertyValue",
      "name": "confidence",
      "value": "argued"
    },
    {
      "@type": "PropertyValue",
      "name": "scope_conditions",
      "value": [
        "multi-agent LLM evaluation as characterized by the reviewed benchmark literature"
      ]
    },
    {
      "@type": "PropertyValue",
      "name": "is_failure_mode",
      "value": false
    },
    {
      "@type": "PropertyValue",
      "name": "edges",
      "value": []
    }
  ],
  "isPartOf": {
    "@id": "https://wulfkaal.github.io/claims/index.json"
  },
  "dateModified": "2026-08-10",
  "version": "1.0",
  "sha256": "8259b644c716b3815312434f67407b9a6426a4321973ffd0f28431c164c9790c",
  "canonicalForm": "https://wulfkaal.github.io/claims/7261018-001.md"
}
