{
 "@context": "https://schema.org",
 "@type": "DefinedTerm",
 "@id": "https://wulfkaal.github.io/entities/training-data",
 "identifier": "kaal:entity:training-data",
 "name": "Training data",
 "termCode": "training-data",
 "inDefinedTermSet": {
  "@id": "https://wulfkaal.github.io/entities/index.json"
 },
 "author": {
  "@type": "Person",
  "name": "Wulf A. Kaal",
  "identifier": "https://orcid.org/0000-0003-0757-275X"
 },
 "dateModified": "2026-07-29",
 "canonicalForm": "https://wulfkaal.github.io/entities/training-data.md",
 "sha256": "898890f2f7c932aa040042a62e8d274e607f9359c74c509aad669a46331fe9de",
 "additionalProperty": [
  {
   "@type": "PropertyValue",
   "name": "status",
   "value": "derived"
  },
  {
   "@type": "PropertyValue",
   "name": "claim_count",
   "value": 11
  },
  {
   "@type": "PropertyValue",
   "name": "work_count",
   "value": 6
  },
  {
   "@type": "PropertyValue",
   "name": "year_span",
   "value": [
    "2018",
    "2026"
   ]
  },
  {
   "@type": "PropertyValue",
   "name": "non_current_claims",
   "value": 0
  }
 ],
 "subjectOf": [
  {
   "@type": "Claim",
   "@id": "https://wulfkaal.github.io/claims/3128900-003",
   "identifier": "kaal:claim:3128900-003",
   "text": "The performance of an AI neural network's learning algorithm during supervised training rises with the quality and quantity of the labelled datasets it is trained on, which ties AI progress directly to micro task work.",
   "abstract": "The higher the quality and quantity of such labelled datasets the better the AI neural network's learning algorithm during the supervised training process.",
   "citation": "Wulf A. Kaal, Decentralized Mechanical Turk Through Verified Reputation (2018). SSRN: https://ssrn.com/abstract=3128900",
   "datePublished": "2018",
   "claim_type": "mechanism",
   "confidence": "evidenced",
   "is_failure_mode": false,
   "scope_conditions": [
    "supervised training process"
   ],
   "source_pdf_sha256": "381d72e85d976e2af852e8ec3bba87bc6a05ccb3349a2c3dd6f9e64de96ab4b0",
   "status": "current"
  },
  {
   "@type": "Claim",
   "@id": "https://wulfkaal.github.io/claims/3128900-005",
   "identifier": "kaal:claim:3128900-005",
   "text": "Existing centralized micro task marketplaces cannot adequately meet the rising demand for high quality labelled AI training data.",
   "abstract": "The existing centralized marketplaces for micro task work cannot adequately fulfil the increasing demand for high quality micro task work for AI labelled training datasets.",
   "citation": "Wulf A. Kaal, Decentralized Mechanical Turk Through Verified Reputation (2018). SSRN: https://ssrn.com/abstract=3128900",
   "datePublished": "2018",
   "claim_type": "failure",
   "confidence": "argued",
   "is_failure_mode": true,
   "scope_conditions": [
    "existing centralized marketplaces",
    "demand for high quality AI training datasets"
   ],
   "source_pdf_sha256": "381d72e85d976e2af852e8ec3bba87bc6a05ccb3349a2c3dd6f9e64de96ab4b0",
   "status": "current"
  },
  {
   "@type": "Claim",
   "@id": "https://wulfkaal.github.io/claims/4796714-011",
   "identifier": "kaal:claim:4796714-011",
   "text": "Bias in AI systems arises when algorithms incorporate discriminatory practices carried in their training data, and the resulting outputs reveal a profound misalignment between AI operations and societal values, ethics, and norms.",
   "abstract": "AI governance does encounter a critical challenge in mitigating biases within AI systems, where biases can inadvertently arise through algorithms incorporating discriminatory practices due to data used in training.",
   "citation": "Wulf A. Kaal, AI Governance (2024). SSRN: https://ssrn.com/abstract=4796714",
   "datePublished": "2024",
   "claim_type": "mechanism",
   "confidence": "evidenced",
   "is_failure_mode": true,
   "scope_conditions": [
    "arises through the training data pathway",
    "observable in deployed systems such as chatbots and screening tools"
   ],
   "source_pdf_sha256": "59fa63bae179e8f9b6b8efbdf90cee28400276512a1b04f9f579a48641305c93",
   "status": "current"
  },
  {
   "@type": "Claim",
   "@id": "https://wulfkaal.github.io/claims/4796714-034",
   "identifier": "kaal:claim:4796714-034",
   "text": "Routing proposals through the Forum and then through Validation Pool review is what allows the input parameters and learning data of AI systems to be governed by expert community consensus, because only vetted and consensus backed data and parameters reach AI development.",
   "abstract": "This process involves submitting proposals to the Forum and undergoing Validation Pool review, ensuring that only vetted and consensus-backed data and parameters are utilized in AI development.",
   "citation": "Wulf A. Kaal, AI Governance (2024). SSRN: https://ssrn.com/abstract=4796714",
   "datePublished": "2024",
   "claim_type": "mechanism",
   "confidence": "argued",
   "is_failure_mode": false,
   "scope_conditions": [
    "requires an operating DAO with reputation tokens and validation pools",
    "applies to input parameters and training data selection"
   ],
   "source_pdf_sha256": "59fa63bae179e8f9b6b8efbdf90cee28400276512a1b04f9f579a48641305c93",
   "status": "current"
  },
  {
   "@type": "Claim",
   "@id": "https://wulfkaal.github.io/claims/4796714-035",
   "identifier": "kaal:claim:4796714-035",
   "text": "AI learning is degraded by Web2 platforms because their engagement driven algorithms amplify extreme viewpoints and negativity, so the human sentiment and ethics the models absorb from that data are systematically distorted.",
   "abstract": "AI learning is afflicted by web2 systems that bring out suboptimal human generated outcomes. WEB2 platforms often amplify extreme viewpoints and negativity due to their engagement-driven algorithms, presenting a distorted view of human sentiment and ethics.",
   "citation": "Wulf A. Kaal, AI Governance (2024). SSRN: https://ssrn.com/abstract=4796714",
   "datePublished": "2024",
   "claim_type": "mechanism",
   "confidence": "argued",
   "is_failure_mode": true,
   "scope_conditions": [
    "applies to models trained on unfiltered web and social media data",
    "arises from engagement optimization in Web2 platforms"
   ],
   "source_pdf_sha256": "59fa63bae179e8f9b6b8efbdf90cee28400276512a1b04f9f579a48641305c93",
   "status": "current"
  },
  {
   "@type": "Claim",
   "@id": "https://wulfkaal.github.io/claims/4855607-003",
   "identifier": "kaal:claim:4855607-003",
   "text": "Deep learning models inadvertently learn and amplify whatever biases exist in their training data, so the composition of the training corpus, not the architecture, is the source of unfair or discriminatory outcomes.",
   "abstract": "Depending on the data used for training, deep learning models can inadvertently learn and amplify biases present in the training data, potentially leading to unfair or discriminatory outcomes.",
   "citation": "Wulf A. Kaal, How AI Models are Optimized Through Web3 Governance (2024). SSRN: https://ssrn.com/abstract=4855607",
   "datePublished": "2024",
   "claim_type": "failure",
   "confidence": "evidenced",
   "is_failure_mode": true,
   "scope_conditions": [
    "depends on the data used for training"
   ],
   "source_pdf_sha256": "eb0b3e62374b45a8fa888c6bde9725e606bcb46cf4b5e74a6e851d9f25099113",
   "status": "current"
  },
  {
   "@type": "Claim",
   "@id": "https://wulfkaal.github.io/claims/4855607-026",
   "identifier": "kaal:claim:4855607-026",
   "text": "Requiring community members to stake reputation tokens in order to validate data quality is what produces robust and reliable training datasets, and this participatory validation improves annotation accuracy while reducing bias.",
   "abstract": "Community members stake reputation tokens to validate data quality, ensuring robust and reliable datasets for training AI models. This participatory approach can improve data annotation accuracy and reduce biases.",
   "citation": "Wulf A. Kaal, How AI Models are Optimized Through Web3 Governance (2024). SSRN: https://ssrn.com/abstract=4855607",
   "datePublished": "2024",
   "claim_type": "mechanism",
   "confidence": "asserted",
   "is_failure_mode": false,
   "scope_conditions": [
    "training data validation carried out by a reputation staking community"
   ],
   "source_pdf_sha256": "eb0b3e62374b45a8fa888c6bde9725e606bcb46cf4b5e74a6e851d9f25099113",
   "status": "current"
  },
  {
   "@type": "Claim",
   "@id": "https://wulfkaal.github.io/claims/4941807-015",
   "identifier": "kaal:claim:4941807-015",
   "text": "A critical unsolved challenge for AI governance is bias mitigation, because biases enter inadvertently when algorithms incorporate discriminatory practices carried in the data used for training.",
   "abstract": "AI governance does encounter a critical challenge in mitigating biases within AI systems, where biases can inadvertently arise through algorithms incorporating discriminatory practices due to data used in training.",
   "citation": "Wulf A. Kaal, AI Governance Via Web3 Reputation System (2024). SSRN: https://ssrn.com/abstract=4941807",
   "datePublished": "2024",
   "claim_type": "failure",
   "confidence": "evidenced",
   "is_failure_mode": true,
   "scope_conditions": [
    "where training data embeds discriminatory practices"
   ],
   "source_pdf_sha256": "ab66c1e99a88da1fa36b0c6b536df5184231fe6aa427f3dd53287a4e0ac79853",
   "status": "current"
  },
  {
   "@type": "Claim",
   "@id": "https://wulfkaal.github.io/claims/5541658-013",
   "identifier": "kaal:claim:5541658-013",
   "text": "Bias in judicial AI arises because models are trained on historical data that reflect past inequities, and the standard remedy of fairness through unawareness, meaning the omission of protected characteristics such as race, fails because proxy variables continue to correlate with the omitted attribute.",
   "abstract": "This bias arises because AI models rely on historical data that reflect past inequities, and even attempts at \"fairness through unawareness\" (omitting protected characteristics like race) fail due to proxy variables that correlate with bias.",
   "citation": "Wulf A. Kaal, Morgan A. Gray, The Evolving Role of Artificial Intelligence in Law (2025). SSRN: https://ssrn.com/abstract=5541658",
   "datePublished": "2025",
   "claim_type": "failure",
   "confidence": "argued",
   "is_failure_mode": true,
   "scope_conditions": [
    "machine learning models trained on historical judicial data"
   ],
   "source_pdf_sha256": "e543a2d698fcd522d4d02e034cc9ee1344d0015d2c824b40b9e05ab7c0728c60",
   "status": "current"
  },
  {
   "@type": "Claim",
   "@id": "https://wulfkaal.github.io/claims/5541658-030",
   "identifier": "kaal:claim:5541658-030",
   "text": "Because AI systems are predominantly developed in the West and trained mostly on Western data, their outputs are liable to carry cultural biases that inadequately represent non-Western cultures and the values inherent in them.",
   "abstract": "Such western AI system domination can be further exacerbated through mostly western training data for AI systems. This may lead to cultural biases in AI outputs as non-western cultures and non-western values inherent in such cultures are inadequately represented.",
   "citation": "Wulf A. Kaal, Morgan A. Gray, The Evolving Role of Artificial Intelligence in Law (2025). SSRN: https://ssrn.com/abstract=5541658",
   "datePublished": "2025",
   "claim_type": "mechanism",
   "confidence": "argued",
   "is_failure_mode": true,
   "scope_conditions": [
    "AI systems deployed in non-Western legal systems",
    "training corpora dominated by Western sources"
   ],
   "source_pdf_sha256": "e543a2d698fcd522d4d02e034cc9ee1344d0015d2c824b40b9e05ab7c0728c60",
   "status": "current"
  },
  {
   "@type": "Claim",
   "@id": "https://wulfkaal.github.io/claims/6607458-017",
   "identifier": "kaal:claim:6607458-017",
   "text": "Computational abundance does not eliminate information asymmetry; it transforms its locus, since traditional informational advantages such as knowledge of market conditions, contract terms, and domain expertise become accessible at negligible cost.",
   "abstract": "Computational abundance does not eliminate information asymmetry, it transforms its locus. Traditional informational advantages, knowledge of market conditions, understanding of contract terms, possession of domain expertise, become accessible at negligible cost.",
   "citation": "Wulf A. Kaal, Computative Economics A Framework for Economic Analysis under Computational Abundance (2026). SSRN: https://ssrn.com/abstract=6607458",
   "datePublished": "2026",
   "claim_type": "mechanism",
   "confidence": "argued",
   "is_failure_mode": false,
   "scope_conditions": [],
   "source_pdf_sha256": "75cca35350378eb8904866d068b5d2f3e1504629e6031457ce6c12152a88e3e8",
   "status": "current"
  }
 ],
 "description": "11 claims in the published works of Wulf A. Kaal carry the concept tag 'training-data'. Derived node: a roster, not an adjudicated definition."
}