{
 "@context": "https://schema.org",
 "@type": "DefinedTerm",
 "@id": "https://wulfkaal.github.io/entities/llm-training",
 "identifier": "kaal:entity:llm-training",
 "name": "Llm training",
 "termCode": "llm-training",
 "inDefinedTermSet": {
  "@id": "https://wulfkaal.github.io/entities/index.json"
 },
 "author": {
  "@type": "Person",
  "name": "Wulf A. Kaal",
  "identifier": "https://orcid.org/0000-0003-0757-275X"
 },
 "dateModified": "2026-07-29",
 "canonicalForm": "https://wulfkaal.github.io/entities/llm-training.md",
 "sha256": "93b0ec5bce8dce33f899abf221cc1000bf6755618c9266787685e6e53d1a1886",
 "additionalProperty": [
  {
   "@type": "PropertyValue",
   "name": "status",
   "value": "derived"
  },
  {
   "@type": "PropertyValue",
   "name": "claim_count",
   "value": 2
  },
  {
   "@type": "PropertyValue",
   "name": "work_count",
   "value": 1
  },
  {
   "@type": "PropertyValue",
   "name": "year_span",
   "value": [
    "2024",
    "2024"
   ]
  },
  {
   "@type": "PropertyValue",
   "name": "non_current_claims",
   "value": 0
  }
 ],
 "subjectOf": [
  {
   "@type": "Claim",
   "@id": "https://wulfkaal.github.io/claims/4755632-002",
   "identifier": "kaal:claim:4755632-002",
   "text": "Transformer neural network architecture removes the scale constraint on training data but not the quality constraint, so data quality continues to be a major unsolved issue for large language models even where internet scale corpora are available.",
   "abstract": "data quality continues to be a huge issue for LLMs",
   "citation": "Wulf A. Kaal, AI Learning - Decentralized Governance to Optimize Human Output Datasets for AI Learning (2024). SSRN: https://ssrn.com/abstract=4755632",
   "datePublished": "2024",
   "claim_type": "failure",
   "confidence": "argued",
   "is_failure_mode": true,
   "scope_conditions": [
    "applies to LLMs trained on internet derived corpora"
   ],
   "source_pdf_sha256": "972ccebf0c06ac1767a9e443bb95942b7670e806a63c25ee817c368a64c8eca8",
   "status": "current"
  },
  {
   "@type": "Claim",
   "@id": "https://wulfkaal.github.io/claims/4755632-003",
   "identifier": "kaal:claim:4755632-003",
   "text": "The move by AI developers toward smaller training datasets raises the risk of overfitting, especially with complex models, which forces LLM developers to rely on regularization to counteract overfitting of the model to the training data.",
   "abstract": "However, with small datasets in LLMs, the risk of overfitting also rises, especially with complex models. Therefore, LLM developers have to turn to regularization in an effort to address overfitting of the model with the training data.",
   "citation": "Wulf A. Kaal, AI Learning - Decentralized Governance to Optimize Human Output Datasets for AI Learning (2024). SSRN: https://ssrn.com/abstract=4755632",
   "datePublished": "2024",
   "claim_type": "failure",
   "confidence": "argued",
   "is_failure_mode": true,
   "scope_conditions": [
    "holds for smaller datasets used in LLM development",
    "risk increases with model complexity"
   ],
   "source_pdf_sha256": "972ccebf0c06ac1767a9e443bb95942b7670e806a63c25ee817c368a64c8eca8",
   "status": "current"
  }
 ],
 "description": "2 claims in the published works of Wulf A. Kaal carry the concept tag 'llm-training'. Derived node: a roster, not an adjudicated definition."
}