{
 "@context": "https://schema.org",
 "@type": "DefinedTerm",
 "@id": "https://wulfkaal.github.io/entities/evaluation-validity",
 "identifier": "kaal:entity:evaluation-validity",
 "name": "Evaluation validity",
 "termCode": "evaluation-validity",
 "inDefinedTermSet": {
  "@id": "https://wulfkaal.github.io/entities/index.json"
 },
 "author": {
  "@type": "Person",
  "name": "Wulf A. Kaal",
  "identifier": "https://orcid.org/0000-0003-0757-275X"
 },
 "dateModified": "2026-07-29",
 "canonicalForm": "https://wulfkaal.github.io/entities/evaluation-validity.md",
 "sha256": "11fe6460b5ceb84858e7f886a50e32060ae085a3cec6fdd9fde4600b80bf644a",
 "additionalProperty": [
  {
   "@type": "PropertyValue",
   "name": "status",
   "value": "derived"
  },
  {
   "@type": "PropertyValue",
   "name": "claim_count",
   "value": 2
  },
  {
   "@type": "PropertyValue",
   "name": "work_count",
   "value": 1
  },
  {
   "@type": "PropertyValue",
   "name": "year_span",
   "value": [
    "2025",
    "2025"
   ]
  },
  {
   "@type": "PropertyValue",
   "name": "non_current_claims",
   "value": 0
  }
 ],
 "subjectOf": [
  {
   "@type": "Claim",
   "@id": "https://wulfkaal.github.io/claims/5541658-008",
   "identifier": "kaal:claim:5541658-008",
   "text": "Prediction on a test set of existing judgments is not the same task as predicting outcomes for a party mid-litigation, because the precise formulation of facts used by such models emerges only once the judgment has been issued.",
   "abstract": "for instance, a lawyer advising a client on the probable outcome of a court hearing—does not have access to the precise formulation of facts presented in a judgment, as this formulation emerges only once the judgment has been issued.",
   "citation": "Wulf A. Kaal, Morgan A. Gray, The Evolving Role of Artificial Intelligence in Law (2025). SSRN: https://ssrn.com/abstract=5541658",
   "datePublished": "2025",
   "claim_type": "failure",
   "confidence": "argued",
   "is_failure_mode": true,
   "scope_conditions": [
    "legal judgment prediction systems trained and evaluated on published judgments"
   ],
   "source_pdf_sha256": "e543a2d698fcd522d4d02e034cc9ee1344d0015d2c824b40b9e05ab7c0728c60",
   "status": "current"
  },
  {
   "@type": "Claim",
   "@id": "https://wulfkaal.github.io/claims/5541658-032",
   "identifier": "kaal:claim:5541658-032",
   "text": "Benchmark results for legal LLMs may overstate capability because of data contamination: if a model saw a benchmark's ground truth answers during training, its measured performance reflects memorization rather than genuine generalization.",
   "abstract": "If a model has already seen a benchmark's ground-truth answers during training, its performance may reflect memorization rather than genuine generalization, making it hard to assess its ability on truly unseen tasks.",
   "citation": "Wulf A. Kaal, Morgan A. Gray, The Evolving Role of Artificial Intelligence in Law (2025). SSRN: https://ssrn.com/abstract=5541658",
   "datePublished": "2025",
   "claim_type": "failure",
   "confidence": "evidenced",
   "is_failure_mode": true,
   "scope_conditions": [
    "closed source LLMs evaluated on public benchmarks"
   ],
   "source_pdf_sha256": "e543a2d698fcd522d4d02e034cc9ee1344d0015d2c824b40b9e05ab7c0728c60",
   "status": "current"
  }
 ],
 "description": "2 claims in the published works of Wulf A. Kaal carry the concept tag 'evaluation-validity'. Derived node: a roster, not an adjudicated definition."
}