{
 "failure_mode": "ai-model-and-training-failure",
 "specific_names": [
  "Distribution shift brittleness",
  "GNN adversarial vulnerability",
  "Homogeneous graph assumption",
  "Overfitting from repeated reuse of the same data set",
  "Plausibility Truth Orthogonality",
  "Restricted legal dataset access",
  "Scale Without Quality",
  "Small Dataset Overfitting",
  "Sparse reward learning failure",
  "Training data bias amplification",
  "Western training data bias",
  "absent reliability guarantee",
  "annotator bias propagation",
  "centralized discernment deficit",
  "clinical bias amplification",
  "emotive-cognitive gap in NLP",
  "engagement driven data distortion",
  "expert system brittleness",
  "fairness through unawareness failure",
  "overfitting-in-repeated-dataset-use",
  "privacy driven data starvation",
  "quality-filtered supply shortfall",
  "recursive synthetic dilution",
  "regression failure on novel situations",
  "representation skew",
  "residual hallucination under RAG",
  "systemic bias invisible to validation",
  "tragedy of the commons in microdemocracy",
  "training data bias propagation",
  "training data bias transmission",
  "unexplainable prediction"
 ],
 "count": 31,
 "claims": [
  {
   "id": "kaal:claim:2998033-021",
   "url": "https://wulfkaal.github.io/claims/2998033-021",
   "claim": "Repeated use of the same dataset by data scientists creates an overfitting risk: the training model fits the test set so closely that its performance on a different dataset degrades.",
   "specific_name": "overfitting-in-repeated-dataset-use",
   "conditions": [
    "applies to adaptive data analysis where the same data set is used repetitively"
   ],
   "source": "Blockchain Innovation for Private Investment Funds",
   "year": "2017",
   "quote": "When data scientists use the same data set repetitively a risk exists that the training model will overfit the test set of data which can limit the performance of the applied model on a different dataset.",
   "citation": "Wulf A. Kaal, Blockchain Innovation for Private Investment Funds (2017). SSRN: https://ssrn.com/abstract=2998033"
  },
  {
   "id": "kaal:claim:3409548-009",
   "url": "https://wulfkaal.github.io/claims/3409548-009",
   "claim": "Repeated use of the same data set by data scientists creates an overfitting risk: the training model overfits the test set, which limits the performance of the applied model on a different dataset.",
   "specific_name": "Overfitting from repeated reuse of the same data set",
   "conditions": [
    "data scientists reusing a single data set repetitively"
   ],
   "source": "Financial Technology and Hedge Funds",
   "year": "2019",
   "quote": "When data scientists use the same data set repetitively a risk exists that the training model will overfit the test set of data, which can limit the performance of the applied model on a different dataset.",
   "citation": "Kaal, Financial Technology and Hedge Funds (2019). SSRN: https://papers.ssrn.com/sol3/papers.cfm?abstract_id=3409548"
  },
  {
   "id": "kaal:claim:3709041-020",
   "url": "https://wulfkaal.github.io/claims/3709041-020",
   "claim": "Microdemocracies must confront the tragedy of the commons, because voters acting on self-interest independent of the totality of voters may, without controls, deplete or spoil shared resources such as the environment and public goods.",
   "specific_name": "tragedy of the commons in microdemocracy",
   "conditions": [
    "applies to microdemocratic shared resource systems without controls"
   ],
   "source": "Blockchain Technology for Good",
   "year": "2020",
   "quote": "Acting for their own self-interest may be contrary to the common good of all voters. Without controls, some self-interested voters may be depleting or spoiling resources they share with the rest of society, such as the environment, public goods, among others.",
   "citation": "Kaal, Blockchain Technology for Good (2020). SSRN: https://papers.ssrn.com/sol3/papers.cfm?abstract_id=3709041"
  },
  {
   "id": "kaal:claim:3782201-012",
   "url": "https://wulfkaal.github.io/claims/3782201-012",
   "claim": "Suggestions that artificial intelligence will solve the rigidity of centralized reputation systems are misguided and will fail for the same reason, because neural networks are merely a complex mathematical architecture for statistical regression, which is always extremely unreliable when applied to novel situations.",
   "specific_name": "regression failure on novel situations",
   "conditions": [
    "where new social situations continually arise that lie outside the training distribution"
   ],
   "source": "Future of Decentralization",
   "year": "2021",
   "quote": "Suggestions that AI will solve these problems are misguided, and will fail for the same reason. Neural networks are merely a complex mathematical ar- chitecture for performing statistical regression, which is always extremely unreliable when applied to novel situations.",
   "citation": "Craig Calcaterra, Wulf A. Kaal, Future of Decentralization (2021). SSRN: https://ssrn.com/abstract=3782201"
  },
  {
   "id": "kaal:claim:4796714-010",
   "url": "https://wulfkaal.github.io/claims/4796714-010",
   "claim": "Federated learning lacks theoretical guarantees of reliability and robustness, which makes its behavior unpredictable in practical applications.",
   "specific_name": "absent reliability guarantee",
   "conditions": [
    "applies to federated learning as practiced"
   ],
   "source": "AI Governance",
   "year": "2024",
   "quote": "Additionally, FL lacks theoretical guarantees that ensure reliability and robustness, making it less predictable for practical applications.",
   "citation": "Wulf A. Kaal, AI Governance (2024). SSRN: https://ssrn.com/abstract=4796714"
  },
  {
   "id": "kaal:claim:4796714-011",
   "url": "https://wulfkaal.github.io/claims/4796714-011",
   "claim": "Bias in AI systems arises when algorithms incorporate discriminatory practices carried in their training data, and the resulting outputs reveal a profound misalignment between AI operations and societal values, ethics, and norms.",
   "specific_name": "training data bias transmission",
   "conditions": [
    "arises through the training data pathway",
    "observable in deployed systems such as chatbots and screening tools"
   ],
   "source": "AI Governance",
   "year": "2024",
   "quote": "AI governance does encounter a critical challenge in mitigating biases within AI systems, where biases can inadvertently arise through algorithms incorporating discriminatory practices due to data used in training.",
   "citation": "Wulf A. Kaal, AI Governance (2024). SSRN: https://ssrn.com/abstract=4796714"
  },
  {
   "id": "kaal:claim:4796714-035",
   "url": "https://wulfkaal.github.io/claims/4796714-035",
   "claim": "AI learning is degraded by Web2 platforms because their engagement driven algorithms amplify extreme viewpoints and negativity, so the human sentiment and ethics the models absorb from that data are systematically distorted.",
   "specific_name": "engagement driven data distortion",
   "conditions": [
    "applies to models trained on unfiltered web and social media data",
    "arises from engagement optimization in Web2 platforms"
   ],
   "source": "AI Governance",
   "year": "2024",
   "quote": "AI learning is afflicted by web2 systems that bring out suboptimal human generated outcomes. WEB2 platforms often amplify extreme viewpoints and negativity due to their engagement-driven algorithms, presenting a distorted view of human sentiment and ethics.",
   "citation": "Wulf A. Kaal, AI Governance (2024). SSRN: https://ssrn.com/abstract=4796714"
  },
  {
   "id": "kaal:claim:4796714-038",
   "url": "https://wulfkaal.github.io/claims/4796714-038",
   "claim": "Broad community governance of AI training identifies and mitigates bias more effectively than data validation alone, because validation focused approaches can overlook systemic biases already embedded in the pretrained model.",
   "specific_name": "systemic bias invisible to validation",
   "conditions": [
    "compares community governance of training against post hoc data validation",
    "concerns systemic bias baked in during pretraining"
   ],
   "source": "AI Governance",
   "year": "2024",
   "quote": "Moreover, involving a broad community in the governance of AI training can help identify and mitigate biases more effectively than a focus on data validation alone, which might overlook systemic biases embedded in the pretrained models.",
   "citation": "Wulf A. Kaal, AI Governance (2024). SSRN: https://ssrn.com/abstract=4796714"
  },
  {
   "id": "kaal:claim:4941807-009",
   "url": "https://wulfkaal.github.io/claims/4941807-009",
   "claim": "Concrete cases show the cost of AI opacity: Nvidia self driving cars that learn from human behavior might confuse the moon for a traffic light, and the DeepPatient project predicted disease onset accurately from medical records while offering no explanation for its predictions.",
   "specific_name": "unexplainable prediction",
   "conditions": [
    "in deep learning systems deployed in driving and medical prediction"
   ],
   "source": "AI Governance Via Web3 Reputation System",
   "year": "2024",
   "quote": "For example, Nvidia's self-driving cars learn from human behavior but might confuse the moon for a traffic light, and the DeepPatient project accurately predicted disease onset from medical records without providing explanations for its predictions",
   "citation": "Wulf A. Kaal, AI Governance Via Web3 Reputation System (2024). SSRN: https://ssrn.com/abstract=4941807"
  },
  {
   "id": "kaal:claim:4941807-010",
   "url": "https://wulfkaal.github.io/claims/4941807-010",
   "claim": "Strict data privacy regulation such as the GDPR imposes stringent conditions on data sharing that limit the amount and variety of data available to AI systems, which can reduce model performance and exacerbate bias because the training dataset is restricted.",
   "specific_name": "privacy driven data starvation",
   "conditions": [
    "under stringent data sharing regimes such as the GDPR"
   ],
   "source": "AI Governance Via Web3 Reputation System",
   "year": "2024",
   "quote": "GDPR imposes stringent conditions on data sharing, which can limit the amount and variety of data AI systems use, potentially reducing their performance and exacerbating biases due to the restricted dataset.",
   "citation": "Wulf A. Kaal, AI Governance Via Web3 Reputation System (2024). SSRN: https://ssrn.com/abstract=4941807"
  },
  {
   "id": "kaal:claim:4941807-015",
   "url": "https://wulfkaal.github.io/claims/4941807-015",
   "claim": "A critical unsolved challenge for AI governance is bias mitigation, because biases enter inadvertently when algorithms incorporate discriminatory practices carried in the data used for training.",
   "specific_name": "training data bias propagation",
   "conditions": [
    "where training data embeds discriminatory practices"
   ],
   "source": "AI Governance Via Web3 Reputation System",
   "year": "2024",
   "quote": "AI governance does encounter a critical challenge in mitigating biases within AI systems, where biases can inadvertently arise through algorithms incorporating discriminatory practices due to data used in training.",
   "citation": "Wulf A. Kaal, AI Governance Via Web3 Reputation System (2024). SSRN: https://ssrn.com/abstract=4941807"
  },
  {
   "id": "kaal:claim:4755632-002",
   "url": "https://wulfkaal.github.io/claims/4755632-002",
   "claim": "Transformer neural network architecture removes the scale constraint on training data but not the quality constraint, so data quality continues to be a major unsolved issue for large language models even where internet scale corpora are available.",
   "specific_name": "Scale Without Quality",
   "conditions": [
    "applies to LLMs trained on internet derived corpora"
   ],
   "source": "AI Learning - Decentralized Governance to Optimize Human Output Datasets for AI Learning",
   "year": "2024",
   "quote": "data quality continues to be a huge issue for LLMs",
   "citation": "Wulf A. Kaal, AI Learning - Decentralized Governance to Optimize Human Output Datasets for AI Learning (2024). SSRN: https://ssrn.com/abstract=4755632"
  },
  {
   "id": "kaal:claim:4755632-003",
   "url": "https://wulfkaal.github.io/claims/4755632-003",
   "claim": "The move by AI developers toward smaller training datasets raises the risk of overfitting, especially with complex models, which forces LLM developers to rely on regularization to counteract overfitting of the model to the training data.",
   "specific_name": "Small Dataset Overfitting",
   "conditions": [
    "holds for smaller datasets used in LLM development",
    "risk increases with model complexity"
   ],
   "source": "AI Learning - Decentralized Governance to Optimize Human Output Datasets for AI Learning",
   "year": "2024",
   "quote": "However, with small datasets in LLMs, the risk of overfitting also rises, especially with complex models. Therefore, LLM developers have to turn to regularization in an effort to address overfitting of the model with the training data.",
   "citation": "Wulf A. Kaal, AI Learning - Decentralized Governance to Optimize Human Output Datasets for AI Learning (2024). SSRN: https://ssrn.com/abstract=4755632"
  },
  {
   "id": "kaal:claim:4855607-003",
   "url": "https://wulfkaal.github.io/claims/4855607-003",
   "claim": "Deep learning models inadvertently learn and amplify whatever biases exist in their training data, so the composition of the training corpus, not the architecture, is the source of unfair or discriminatory outcomes.",
   "specific_name": "Training data bias amplification",
   "conditions": [
    "depends on the data used for training"
   ],
   "source": "How AI Models are Optimized Through Web3 Governance",
   "year": "2024",
   "quote": "Depending on the data used for training, deep learning models can inadvertently learn and amplify biases present in the training data, potentially leading to unfair or discriminatory outcomes.",
   "citation": "Wulf A. Kaal, How AI Models are Optimized Through Web3 Governance (2024). SSRN: https://ssrn.com/abstract=4855607"
  },
  {
   "id": "kaal:claim:4855607-004",
   "url": "https://wulfkaal.github.io/claims/4855607-004",
   "claim": "Deep learning models adapt to changes in data distribution far less readily than human learning does, which limits their reliability once the operating environment diverges from the training data.",
   "specific_name": "Distribution shift brittleness",
   "conditions": [
    "settings where the data distribution shifts after training"
   ],
   "source": "How AI Models are Optimized Through Web3 Governance",
   "year": "2024",
   "quote": "Furthermore, deep learning models are not as adaptable to changes in data distribution as human learning, which can adapt more quickly to new situations and contexts.",
   "citation": "Wulf A. Kaal, How AI Models are Optimized Through Web3 Governance (2024). SSRN: https://ssrn.com/abstract=4855607"
  },
  {
   "id": "kaal:claim:4855607-009",
   "url": "https://wulfkaal.github.io/claims/4855607-009",
   "claim": "In the legal domain the adoption of transformer based language models is blocked less by capability than by resources and access: training and deployment are resource intensive and large, quality tagged legal datasets are usually restricted.",
   "specific_name": "Restricted legal dataset access",
   "conditions": [
    "legal applications of transformer based language models"
   ],
   "source": "How AI Models are Optimized Through Web3 Governance",
   "year": "2024",
   "quote": "In the legal domain, training and deploying Transformer-based Language Models (TLMs) is resource-intensive, and access to large, quality-tagged legal datasets is often restricted, hindering widespread adoption.",
   "citation": "Wulf A. Kaal, How AI Models are Optimized Through Web3 Governance (2024). SSRN: https://ssrn.com/abstract=4855607"
  },
  {
   "id": "kaal:claim:4855607-011",
   "url": "https://wulfkaal.github.io/claims/4855607-011",
   "claim": "GNNs are vulnerable to adversarial attacks that target both node features and graph structure, and their lack of interpretability remains a major obstacle to applying them to real world problems.",
   "specific_name": "GNN adversarial vulnerability",
   "conditions": [
    "real world GNN deployments where adversaries can influence the graph"
   ],
   "source": "How AI Models are Optimized Through Web3 Governance",
   "year": "2024",
   "quote": "GNNs can be vulnerable to adversarial attacks on both node features and graph structure, and interpretability remains a major obstacle for applying GNNs to real-world problems.",
   "citation": "Wulf A. Kaal, How AI Models are Optimized Through Web3 Governance (2024). SSRN: https://ssrn.com/abstract=4855607"
  },
  {
   "id": "kaal:claim:4855607-012",
   "url": "https://wulfkaal.github.io/claims/4855607-012",
   "claim": "Most GNN architectures assume homogeneous graph structures, so adapting them to heterogeneous graphs with diverse node and edge types remains an unsolved research challenge, and full batch training on large graphs suffers memory overflow.",
   "specific_name": "Homogeneous graph assumption",
   "conditions": [
    "graphs with diverse node and edge types",
    "large graphs trained full batch"
   ],
   "source": "How AI Models are Optimized Through Web3 Governance",
   "year": "2024",
   "quote": "Training GNNs on large graphs can be resource-intensive, and full-batch training methods suffer from memory overflow issues. Many GNNs assume homogeneous graph structures, and adapting these models to heterogeneous graphs with diverse node and edge types remains a significant research challenge.",
   "citation": "Wulf A. Kaal, How AI Models are Optimized Through Web3 Governance (2024). SSRN: https://ssrn.com/abstract=4855607"
  },
  {
   "id": "kaal:claim:4855607-014",
   "url": "https://wulfkaal.github.io/claims/4855607-014",
   "claim": "Deep reinforcement learning demands large amounts of training data, which suggests its algorithms differ fundamentally from human learning, and learning without supervision becomes particularly hard when rewards are sparse, as they typically are in sequence generation tasks.",
   "specific_name": "Sparse reward learning failure",
   "conditions": [
    "sparse reward settings",
    "sequence generation tasks"
   ],
   "source": "How AI Models are Optimized Through Web3 Governance",
   "year": "2024",
   "quote": "Deep RL methods often demand large amounts of training data, suggesting that the algorithms may differ fundamentally from those underlying human learning. Moreover, learning without supervision is particularly hard when the reward is sparse, which is likely to happen for sequence generation tasks.",
   "citation": "Wulf A. Kaal, How AI Models are Optimized Through Web3 Governance (2024). SSRN: https://ssrn.com/abstract=4855607"
  },
  {
   "id": "kaal:claim:5095633-004",
   "url": "https://wulfkaal.github.io/claims/5095633-004",
   "claim": "The apparent abundance of internet text overstates the usable supply, because much of it fails quality thresholds for model training due to redundancy, noise, or irrelevance.",
   "specific_name": "quality-filtered supply shortfall",
   "conditions": [
    "applies to internet-sourced text used for model training"
   ],
   "source": "Artificial Intelligence The Final Frontier",
   "year": "2025",
   "quote": "while the internet contains a vast corpus of textual material, not all content meets quality thresholds suitable for model training, given issues such as redundancy, noise, or irrelevance.",
   "citation": "Wulf A. Kaal, Artificial Intelligence The Final Frontier (2025). SSRN: https://ssrn.com/abstract=5095633"
  },
  {
   "id": "kaal:claim:5095633-010",
   "url": "https://wulfkaal.github.io/claims/5095633-010",
   "claim": "When a training dataset disproportionately represents one region or demographic group, the resulting model produces skewed and sometimes inappropriate outputs once deployed in unfamiliar settings.",
   "specific_name": "representation skew",
   "conditions": [
    "deployment context differs from the demographic composition of the training data"
   ],
   "source": "Artificial Intelligence The Final Frontier",
   "year": "2025",
   "quote": "For instance, if a dataset disproportionately represents a particular region or demographic group, the model may offer skewed performance, demonstrating suboptimal or inappropriate outputs when deployed in unfamiliar settings.",
   "citation": "Wulf A. Kaal, Artificial Intelligence The Final Frontier (2025). SSRN: https://ssrn.com/abstract=5095633"
  },
  {
   "id": "kaal:claim:5095633-011",
   "url": "https://wulfkaal.github.io/claims/5095633-011",
   "claim": "As AI-generated content proliferates online it dilutes the diversity and originality of the text pool available for later training, producing performance degradation across successive model generations.",
   "specific_name": "recursive synthetic dilution",
   "conditions": [
    "repeated training generations drawing on an increasingly synthetic web"
   ],
   "source": "Artificial Intelligence The Final Frontier",
   "year": "2025",
   "quote": "As AI-generated content proliferates online, it dilutes the overall diversity and originality of text available for subsequent training processes, potentially leading to a degradation in model performance over repeated generations.",
   "citation": "Wulf A. Kaal, Artificial Intelligence The Final Frontier (2025). SSRN: https://ssrn.com/abstract=5095633"
  },
  {
   "id": "kaal:claim:5095633-015",
   "url": "https://wulfkaal.github.io/claims/5095633-015",
   "claim": "In healthcare, biased or stale training data produces algorithms that misdiagnose underrepresented populations and thereby reinforce existing health disparities instead of reducing them.",
   "specific_name": "clinical bias amplification",
   "conditions": [
    "clinical deployment on populations underrepresented in training data"
   ],
   "source": "Artificial Intelligence The Final Frontier",
   "year": "2025",
   "quote": "In healthcare, for instance, biased or stale training data could lead to algorithms that misdiagnose underrepresented populations, reinforcing existing health disparities rather than alleviating them.",
   "citation": "Wulf A. Kaal, Artificial Intelligence The Final Frontier (2025). SSRN: https://ssrn.com/abstract=5095633"
  },
  {
   "id": "kaal:claim:5095633-019",
   "url": "https://wulfkaal.github.io/claims/5095633-019",
   "claim": "Biases held by human annotators or embedded in automated annotation systems are propagated into the models trained on their output, producing AI that performs inequitably across demographic groups.",
   "specific_name": "annotator bias propagation",
   "conditions": [
    "applies even to providers that emphasize high-quality data such as Scale AI and Appen"
   ],
   "source": "Artificial Intelligence The Final Frontier",
   "year": "2025",
   "quote": "there's a theoretical risk that biases inherent in data annotators or automated systems might be propagated into AI models. This can lead to AI that does not perform equitably across different demographic groups or scenarios.",
   "citation": "Wulf A. Kaal, Artificial Intelligence The Final Frontier (2025). SSRN: https://ssrn.com/abstract=5095633"
  },
  {
   "id": "kaal:claim:5541658-002",
   "url": "https://wulfkaal.github.io/claims/5541658-002",
   "claim": "Early legal expert systems failed because they were brittle: they could not handle unforeseen complexity and could not generalize beyond the domains they were explicitly programmed for.",
   "specific_name": "expert system brittleness",
   "conditions": [
    "rule-based systems of the 1980s"
   ],
   "source": "The Evolving Role of Artificial Intelligence in Law",
   "year": "2025",
   "quote": "However, early expert systems were brittle, struggling with unforeseen complexities and lacking the ability to generalize beyond their programmed",
   "citation": "Wulf A. Kaal, Morgan A. Gray, The Evolving Role of Artificial Intelligence in Law (2025). SSRN: https://ssrn.com/abstract=5541658"
  },
  {
   "id": "kaal:claim:5541658-007",
   "url": "https://wulfkaal.github.io/claims/5541658-007",
   "claim": "Retrieval-augmented generation reduces but does not eliminate hallucination: leading legal research tools including Lexis+AI and Westlaw AI-Assisted Research still generate false citations or facts in 17 to 33 percent of cases.",
   "specific_name": "residual hallucination under RAG",
   "conditions": [
    "RAG-based commercial legal research platforms",
    "as measured at the time of the cited study; the authors note rates are progressively improving"
   ],
   "source": "The Evolving Role of Artificial Intelligence in Law",
   "year": "2025",
   "quote": "the risk of \"hallucinations,\" where AI generates false legal citations or facts, as observed in tools like Lexis+AI and Westlaw AI-Assisted Research, which hallucinate in 17–33% of cases.",
   "citation": "Wulf A. Kaal, Morgan A. Gray, The Evolving Role of Artificial Intelligence in Law (2025). SSRN: https://ssrn.com/abstract=5541658"
  },
  {
   "id": "kaal:claim:5541658-009",
   "url": "https://wulfkaal.github.io/claims/5541658-009",
   "claim": "Natural language processing lets predictive systems handle specialized legal terminology and competing interpretations, but it still cannot capture the emotive and cognitive nuances of legal reasoning.",
   "specific_name": "emotive-cognitive gap in NLP",
   "conditions": [
    "NLP applied to unstructured legal texts"
   ],
   "source": "The Evolving Role of Artificial Intelligence in Law",
   "year": "2025",
   "quote": "NLP enables predictive systems to handle the specialized terminology and multiple interpretations inherent in legal systems, though challenges remain in capturing emotive-cognitive nuances.",
   "citation": "Wulf A. Kaal, Morgan A. Gray, The Evolving Role of Artificial Intelligence in Law (2025). SSRN: https://ssrn.com/abstract=5541658"
  },
  {
   "id": "kaal:claim:5541658-013",
   "url": "https://wulfkaal.github.io/claims/5541658-013",
   "claim": "Bias in judicial AI arises because models are trained on historical data that reflect past inequities, and the standard remedy of fairness through unawareness, meaning the omission of protected characteristics such as race, fails because proxy variables continue to correlate with the omitted attribute.",
   "specific_name": "fairness through unawareness failure",
   "conditions": [
    "machine learning models trained on historical judicial data"
   ],
   "source": "The Evolving Role of Artificial Intelligence in Law",
   "year": "2025",
   "quote": "This bias arises because AI models rely on historical data that reflect past inequities, and even attempts at \"fairness through unawareness\" (omitting protected characteristics like race) fail due to proxy variables that correlate with bias.",
   "citation": "Wulf A. Kaal, Morgan A. Gray, The Evolving Role of Artificial Intelligence in Law (2025). SSRN: https://ssrn.com/abstract=5541658"
  },
  {
   "id": "kaal:claim:5541658-030",
   "url": "https://wulfkaal.github.io/claims/5541658-030",
   "claim": "Because AI systems are predominantly developed in the West and trained mostly on Western data, their outputs are liable to carry cultural biases that inadequately represent non-Western cultures and the values inherent in them.",
   "specific_name": "Western training data bias",
   "conditions": [
    "AI systems deployed in non-Western legal systems",
    "training corpora dominated by Western sources"
   ],
   "source": "The Evolving Role of Artificial Intelligence in Law",
   "year": "2025",
   "quote": "Such western AI system domination can be further exacerbated through mostly western training data for AI systems. This may lead to cultural biases in AI outputs as non-western cultures and non-western values inherent in such cultures are inadequately represented.",
   "citation": "Wulf A. Kaal, Morgan A. Gray, The Evolving Role of Artificial Intelligence in Law (2025). SSRN: https://ssrn.com/abstract=5541658"
  },
  {
   "id": "kaal:claim:6244278-034",
   "url": "https://wulfkaal.github.io/claims/6244278-034",
   "claim": "Centralized AI systems excel at scaling defined work but struggle to cultivate genuine discernment without consequential feedback, and they face escalating constraints in data scarcity, energy demands, regulatory scrutiny, and interconnect bottlenecks.",
   "specific_name": "centralized discernment deficit",
   "conditions": [
    "applies to purely centralized approaches to artificial general intelligence"
   ],
   "source": "AI's Mother's Instinct Engineered Consequence Emergent Ethics and the Institutional Trajectory Toward Agentic Alignment",
   "year": "2026",
   "quote": "Centralized systems excel at scaling defined work but struggle to cultivate genuine discernment without consequential feedback.",
   "citation": "Wulf A. Kaal, AI's Mother's Instinct Engineered Consequence Emergent Ethics and the Institutional Trajectory Toward Agentic Alignment (2026). SSRN: https://ssrn.com/abstract=6244278"
  },
  {
   "id": "kaal:claim:6421319-023",
   "url": "https://wulfkaal.github.io/claims/6421319-023",
   "claim": "Probabilistic language models hallucinate because they are trained to predict statistically likely token sequences rather than to verify propositional truth, so the error is intrinsic to the substrate: plausibility and truth are orthogonal properties in high dimensional token space.",
   "specific_name": "Plausibility Truth Orthogonality",
   "conditions": [
    "applies to purely probabilistic architectures"
   ],
   "source": "The Collapse of Scarcity Economics",
   "year": "2026",
   "quote": "Probabilistic large language models hallucinate because they are trained to predict statistically likely token sequences, not to verify propositional truth.",
   "citation": "Wulf A. Kaal, The Collapse of Scarcity Economics (2026). SSRN: https://ssrn.com/abstract=6421319"
  }
 ]
}