{
  "name": "The SuperSkills evidence base",
  "description": "Graded evidence on what increasingly capable AI does to human capability. Each entry states method, finding, what it supports and what it does not.",
  "url": "https://thesuperskills.com/research/evidence",
  "compiledBy": "Rahim Hirji, The SuperSkills Intelligence Company",
  "licence": "CC BY 4.0",
  "lastReviewed": "12 September 2026",
  "count": 312,
  "grades": {
    "peer-reviewed": "Published in a peer-reviewed journal or archival conference.",
    "working-paper": "Circulated for comment; not yet peer-reviewed.",
    "institutional-survey": "Survey by an organisation, usually self-selected respondents, not peer-reviewed.",
    "institutional-modelling": "Projection or secondary analysis by an organisation, assumption-driven.",
    "compiled-review": "Aggregation of third-party data rather than original research.",
    "expert-survey": "Survey of a defined expert population about future events. Expertise in the subject is not expertise in forecasting, and framing effects are large.",
    "institutional-synthesis": "Synthesis of the published literature by a standing expert body. Reports the state of evidence rather than generating any.",
    "statutory-investigation": "Investigation of a single event by a public body with access to primary records. Determinations rather than estimates, and N is one.",
    "argued-perspective": "A framework or risk model argued in a reviewed venue, reporting no new data. Carries the authority of its reasoning, not of a measurement.",
    "qualitative-field-study": "Interviews, observation or participatory work in a real setting, peer-reviewed. Rich on mechanism and meaning; no measured outcome and no control.",
    "vendor-research": "Empirical work published by a company about the effects of its own product, using proprietary usage data as an input. Often careful and usually unreproducible: the key variable cannot be rebuilt or checked from outside the firm.",
    "practitioner-method": "A working procedure set out by the person or firm who devised it, with adoption behind it and no study design. Establishes where a technique came from and what it instructs, never that it produces the effect claimed for it.",
    "operator-account": "An organisation's own account of a system it runs. Capability as claimed rather than measured, with no independent verification and an interest in the result."
  },
  "entries": [
    {
      "id": "risko-gilbert-2016",
      "citeAs": "https://thesuperskills.com/research/evidence#risko-gilbert-2016",
      "section": "judgement",
      "authors": "Risko, E. F. and Gilbert, S. J.",
      "year": 2016,
      "title": "Cognitive Offloading",
      "publication": "Trends in Cognitive Sciences, 20(9)",
      "url": "https://www.cell.com/trends/cognitive-sciences/abstract/S1364-6613(16)30098-5",
      "grade": "peer-reviewed",
      "method": "Review of the experimental literature on offloading.",
      "finding": "Defines cognitive offloading as using physical action or an external tool to reduce the mental demand of a task, and shows people offload not only when a task is hard but when they judge it to be hard.",
      "supports": "That the decision to offload is metacognitive, and frequently mistaken.",
      "doesNotSupport": "Nothing about generative AI specifically; it predates it.",
      "terms": [
        "cognitive offloading"
      ],
      "relatedPages": [
        "/research/what-is-cognitive-offloading",
        "/research/ai-and-human-judgement"
      ]
    },
    {
      "id": "sparrow-2011",
      "citeAs": "https://thesuperskills.com/research/evidence#sparrow-2011",
      "section": "judgement",
      "authors": "Sparrow, B., Liu, J. and Wegner, D. M.",
      "year": 2011,
      "title": "Google Effects on Memory: Cognitive Consequences of Having Information at Our Fingertips",
      "publication": "Science, 333(6043)",
      "url": "https://www.science.org/doi/10.1126/science.1207745",
      "grade": "peer-reviewed",
      "method": "Four laboratory experiments.",
      "finding": "When people expect information to remain available, they remember where to find it rather than the thing itself.",
      "supports": "That expected availability changes what gets encoded.",
      "doesNotSupport": "That total memory capability declines, or that the trade is net negative.",
      "terms": [
        "the Google effect",
        "cognitive offloading"
      ],
      "relatedPages": [
        "/research/what-is-cognitive-offloading",
        "/research/ai-and-human-judgement"
      ]
    },
    {
      "id": "dahmani-bohbot-2020",
      "citeAs": "https://thesuperskills.com/research/evidence#dahmani-bohbot-2020",
      "section": "learning",
      "authors": "Dahmani, L. and Bohbot, V. D.",
      "year": 2020,
      "title": "Habitual use of GPS negatively impacts spatial memory during self-guided navigation",
      "publication": "Scientific Reports, 10, 6310, 14 April 2020. DOI 10.1038/s41598-020-62877-0. Read in full at source 5 September 2026",
      "url": "https://www.nature.com/articles/s41598-020-62877-0",
      "grade": "peer-reviewed",
      "method": "Behavioural, cross-sectional with an unplanned longitudinal arm. 50 healthy regular drivers in Montreal aged 19 to 35 (18 women, 32 men; mean age 27.6), driving at least four days a week, from 60 recruited. Lifetime GPS experience measured by the McGill GPS questionnaire; navigation measured on two virtual radial-arm mazes plus a map-drawing score and the Santa Barbara Sense of Direction scale. 13 of the 50 returned a mean of 3.23 years later. Effect sizes are Pearson r with bootstrapped one-tailed BCa 95 per cent intervals. NO NEUROIMAGING: hippocampal dependence is inferred from prior validation of the tasks, not measured in this sample.",
      "finding": "Cross-sectionally, greater lifetime GPS experience was associated with lower use of hippocampus-dependent spatial strategies (r = -0.22 on the first probe trial), lower navigation strategy scores (r = -0.20), poorer map drawing (r = -0.22) and fewer landmarks noticed (r = -0.26). In the 13-person follow-up, hours of GPS use since first testing tracked a steeper decline in spatial memory strategy use (r = -0.68) and in map drawing (r = -0.52).",
      "supports": "The only study to follow the same people while their GPS use rose. Its strongest internal argument against reverse causation is that heavier GPS users did not report a poorer sense of direction (r = 0.07 against the SBSOD), so the obvious alternative, that weak navigators reach for the satnav, has no support in the data.",
      "doesNotSupport": "Anything about the brain, because no scan was taken. Anything about dementia, Alzheimer's or atrophy: those words appear nowhere in the paper. And little with confidence about the longitudinal effect, which rests on 13 people from an unplanned follow-up. The authors' own words: they caution against any strong conclusions as spurious correlations are possible. Their discussion elsewhere uses notably firmer causal language than that sentence licenses. Nothing here transfers to reasoning: spatial memory is not judgement.",
      "terms": [
        "cognitive offloading",
        "deskilling",
        "spatial memory",
        "The Knowledge"
      ],
      "relatedPages": [
        "/research/does-gps-damage-your-brain",
        "/research/what-is-cognitive-offloading",
        "/research/using-ai-without-dependency",
        "/research/ai-and-human-judgement"
      ]
    },
    {
      "id": "lee-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#lee-2025",
      "section": "judgement",
      "authors": "Lee, H.-P. et al.",
      "year": 2025,
      "title": "The Impact of Generative AI on Critical Thinking: Self-Reported Reductions in Cognitive Effort and Confidence Effects from a Survey of Knowledge Workers",
      "publication": "Microsoft Research and Carnegie Mellon, CHI 2025",
      "url": "https://www.microsoft.com/en-us/research/publication/the-impact-of-generative-ai-on-critical-thinking-self-reported-reductions-in-cognitive-effort-and-confidence-effects-from-a-survey-of-knowledge-workers/",
      "grade": "peer-reviewed",
      "method": "Survey of 319 knowledge workers about 936 real uses of AI at work.",
      "finding": "Higher confidence in the tool was associated with less critical thinking, and the thinking that remains shifts from producing to verifying, from solving to integrating.",
      "supports": "That the character of professional thinking changes with AI use, by self-report.",
      "doesNotSupport": "Causation. People who think differently may use AI differently, and a survey cannot separate the two.",
      "terms": [
        "critical thinking",
        "verification",
        "cognitive offloading"
      ],
      "relatedPages": [
        "/research/ai-and-critical-thinking",
        "/research/ai-and-human-judgement",
        "/research/is-screen-time-the-same-argument-as-ai-use"
      ]
    },
    {
      "id": "gerlich-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#gerlich-2025",
      "section": "judgement",
      "authors": "Gerlich, M.",
      "year": 2025,
      "title": "AI Tools in Society: Impacts on Cognitive Offloading and the Future of Critical Thinking",
      "publication": "Societies, 15(1), 6",
      "url": "https://www.mdpi.com/2075-4698/15/1/6",
      "grade": "peer-reviewed",
      "method": "Survey and interviews, 666 participants.",
      "finding": "A negative correlation between frequent AI use and critical-thinking scores, mediated by cognitive offloading, strongest among the youngest users.",
      "supports": "An association, with a plausible mechanism.",
      "doesNotSupport": "Causation, and it carries a published correction (Societies 2025, 15(9), 252) which anyone citing it should read alongside.",
      "correction": "https://www.mdpi.com/2075-4698/15/9/252",
      "terms": [
        "critical thinking",
        "cognitive offloading"
      ],
      "relatedPages": [
        "/research/ai-and-critical-thinking",
        "/research/ai-and-human-judgement",
        "/research/is-screen-time-the-same-argument-as-ai-use"
      ]
    },
    {
      "id": "kosmyna-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#kosmyna-2025",
      "section": "judgement",
      "authors": "Kosmyna, N. et al.",
      "year": 2025,
      "title": "Your Brain on ChatGPT: Accumulation of Cognitive Debt when Using an AI Assistant for Essay Writing Task",
      "publication": "MIT Media Lab preprint, arXiv:2506.08872",
      "url": "https://arxiv.org/abs/2506.08872",
      "grade": "working-paper",
      "method": "EEG study, 54 participants, essay writing with an LLM, a search engine, or unaided.",
      "finding": "The LLM group showed the weakest brain connectivity and the lowest sense of ownership over their own writing.",
      "supports": "Very little on its own. It is suggestive and widely over-quoted.",
      "doesNotSupport": "Anything settled. 54 participants, a preprint, and reproducibility flagged by its own commentators. Treat claims of proof with suspicion.",
      "terms": [
        "cognitive debt",
        "critical thinking"
      ],
      "relatedPages": [
        "/research/ai-and-critical-thinking",
        "/research/ai-and-human-judgement",
        "/research/is-screen-time-the-same-argument-as-ai-use"
      ]
    },
    {
      "id": "parasuraman-manzey-2010",
      "citeAs": "https://thesuperskills.com/research/evidence#parasuraman-manzey-2010",
      "section": "judgement",
      "authors": "Parasuraman, R. and Manzey, D. H.",
      "year": 2010,
      "title": "Complacency and Bias in Human Use of Automation: An Attentional Integration",
      "publication": "Human Factors, 52(3)",
      "url": "https://journals.sagepub.com/doi/10.1177/0018720810376055",
      "grade": "peer-reviewed",
      "method": "Review across aviation, medicine and military domains.",
      "finding": "Automation bias and complacency appear in novices and experts alike, resist training, and worsen under workload.",
      "supports": "That under-questioning automated advice is a robust, decades-old finding, not a novelty of the AI era.",
      "doesNotSupport": "The size of the effect for generative AI, which is far less predictable than the automation studied here.",
      "terms": [
        "automation bias",
        "automation complacency"
      ],
      "relatedPages": [
        "/research/human-ai-decision-making",
        "/research/ai-and-human-judgement"
      ]
    },
    {
      "id": "bastani-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#bastani-2025",
      "section": "learning",
      "authors": "Bastani, H., Bastani, O., Sungu, A., Ge, H., Kabakci, O. and Mariman, R.",
      "year": 2025,
      "title": "Generative AI Without Guardrails Can Harm Learning: Evidence from High School Mathematics",
      "publication": "Proceedings of the National Academy of Sciences, 122(26)",
      "url": "https://papers.ssrn.com/sol3/papers.cfm?abstract_id=4895486",
      "grade": "peer-reviewed",
      "method": "Field experiment, nearly 1,000 high-school students, three arms: unrestricted GPT-4, a hints-only tutor, and a control.",
      "finding": "Grades rose 48 percent with unrestricted access and 127 percent with the tutor while the tool was present. With access removed, the unrestricted group scored 17 percent LOWER than students who never had it. The guardrailed tutor largely removed the harm.",
      "supports": "That the design of the interface, not the presence of AI, decides whether people learn. The single most useful result in this literature.",
      "doesNotSupport": "What a guardrailed interface should look like for professional work. It was school mathematics over a bounded period.",
      "terms": [
        "learning",
        "deskilling",
        "missed reps"
      ],
      "relatedPages": [
        "/research/how-humans-learn-with-ai",
        "/research/chro-guide-to-ai",
        "/research/ai-and-human-judgement"
      ]
    },
    {
      "id": "ericsson-1993",
      "citeAs": "https://thesuperskills.com/research/evidence#ericsson-1993",
      "section": "learning",
      "authors": "Ericsson, K. A., Krampe, R. T. and Tesch-Romer, C.",
      "year": 1993,
      "title": "The Role of Deliberate Practice in the Acquisition of Expert Performance",
      "publication": "Psychological Review, 100(3), 363-406",
      "url": "https://eric.ed.gov/?id=EJ471947",
      "grade": "peer-reviewed",
      "method": "Two studies of violinists and pianists in Berlin.",
      "finding": "Sets out deliberate practice: effortful, targeted activity at the edge of current ability, with feedback, sustained over years.",
      "supports": "That expert performance is built through a specific kind of effortful practice rather than exposure.",
      "doesNotSupport": "How much of the difference between performers practice explains. See the Macnamara and Maitra re-examination below.",
      "terms": [
        "deliberate practice",
        "expertise"
      ],
      "relatedPages": [
        "/research/how-humans-learn-with-ai"
      ]
    },
    {
      "id": "macnamara-maitra-2019",
      "citeAs": "https://thesuperskills.com/research/evidence#macnamara-maitra-2019",
      "section": "learning",
      "authors": "Macnamara, B. N. and Maitra, M.",
      "year": 2019,
      "title": "The role of deliberate practice in expert performance: revisiting Ericsson, Krampe and Tesch-Romer (1993)",
      "publication": "Royal Society Open Science, 6, 190327",
      "url": "https://royalsocietypublishing.org/doi/10.1098/rsos.190327",
      "grade": "peer-reviewed",
      "method": "Direct replication and re-analysis of the 1993 study.",
      "finding": "Accumulated practice explained considerably less of the difference between performers than the original is usually taken to claim.",
      "supports": "That the quality and design of practice matters more than the count. Included here deliberately, because it complicates the argument this research relies on.",
      "doesNotSupport": "That practice does not matter. It does; the simple dose-response reading is what fails.",
      "terms": [
        "deliberate practice",
        "expertise"
      ],
      "relatedPages": [
        "/research/how-humans-learn-with-ai"
      ]
    },
    {
      "id": "bjork-desirable-difficulties",
      "citeAs": "https://thesuperskills.com/research/evidence#bjork-desirable-difficulties",
      "section": "learning",
      "authors": "Bjork, E. L. and Bjork, R. A.",
      "year": 2011,
      "title": "Making Things Hard on Yourself, But in a Good Way: Creating Desirable Difficulties to Enhance Learning",
      "publication": "In Psychology and the Real World, Worth Publishers",
      "url": "https://bjorklab.psych.ucla.edu/wp-content/uploads/sites/13/2016/04/EBjork_RBjork_2011.pdf",
      "grade": "peer-reviewed",
      "method": "Synthesis of decades of laboratory work on spacing, interleaving and retrieval practice.",
      "finding": "Conditions that make study feel harder improve long-term retention; conditions that make it feel fluent improve immediate performance and worsen retention. Learners systematically mistake fluency for learning.",
      "supports": "That the subjective sense of learning is an unreliable guide to whether learning occurred. Directly relevant, because AI makes work feel fluent.",
      "doesNotSupport": "That AI-assisted work is equivalent to a fluent study condition. That inference is ours, not the authors'.",
      "terms": [
        "desirable difficulties",
        "learning"
      ],
      "relatedPages": [
        "/research/how-humans-learn-with-ai"
      ]
    },
    {
      "id": "brynjolfsson-2023",
      "citeAs": "https://thesuperskills.com/research/evidence#brynjolfsson-2023",
      "section": "learning",
      "authors": "Brynjolfsson, E., Li, D. and Raymond, L.",
      "year": 2023,
      "title": "Generative AI at Work",
      "publication": "Quarterly Journal of Economics, 140(2), 889-942. DOI 10.1093/qje/qjae044. Advance Access 4 February 2025. Earlier version NBER Working Paper 31161",
      "url": "https://academic.oup.com/qje/article/140/2/889/7990658",
      "grade": "peer-reviewed",
      "method": "Staggered rollout of a GPT-3-based conversational assistant across 5,172 customer-support agents in 133 teams at a single Fortune 500 business-process software firm, most of them working from the Philippines. Three million chats observed, 1.2 million of them post-deployment. The system was fine-tuned on past agent conversations, with chats by top performers deliberately up-weighted in training.",
      "finding": "Resolutions per hour rose 15 per cent on average, 15.2 per cent in the preferred specification with agent and tenure fixed effects. Less skilled and less experienced workers gained a 30 per cent increase in issues resolved per hour, rising to 36 per cent for the lowest skill quintile, while the most skilled saw no significant productivity change and small declines in conversation quality and customer satisfaction. Customer sentiment improved by half a standard deviation and requests to speak to a manager fell about 25 per cent. During unplanned outages, agents with longer AI exposure still handled chats faster than their pre-AI baseline, but only those who had adhered closely to the suggestions.",
      "supports": "That a model trained on the behaviour of a firm's best workers can transfer measurable parts of that behaviour to its newest ones, immediately and at scale, raising the floor far more than the ceiling. The outage evidence shows some of the gain survives the tool being switched off, conditional on the worker having engaged with it rather than passed it through.",
      "doesNotSupport": "Whether those novices became experts. It measures output over months in one firm, one occupation and a stable product environment, and the authors say so. Wages, labour demand and hiring composition were not observed. The outage estimates are the authors' own noisiest, because outages are rare and may not be comparable chats.",
      "terms": [
        "productivity",
        "expertise",
        "synthetic seniority"
      ],
      "relatedPages": [
        "/research/how-humans-learn-with-ai",
        "/research/staying-valuable-in-the-age-of-ai",
        "/research/ai-and-human-judgement",
        "/research/how-will-ai-change-customer-service",
        "/research/what-becomes-more-valuable-as-ai-gets-cheaper",
        "/research/ai-and-expert-judgement",
        "/research/does-ai-actually-make-people-more-productive"
      ]
    },
    {
      "id": "vaccaro-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#vaccaro-2024",
      "section": "collaboration",
      "authors": "Vaccaro, M., Almaatouq, A. and Malone, T.",
      "year": 2024,
      "title": "When combinations of humans and AI are useful: a systematic review and meta-analysis",
      "publication": "Nature Human Behaviour, 8, 2293-2303",
      "url": "https://www.nature.com/articles/s41562-024-02024-1",
      "grade": "peer-reviewed",
      "method": "Preregistered systematic review and meta-analysis: 106 experimental studies, 370 effect sizes, published January 2020 to June 2023.",
      "finding": "Human-AI combinations performed significantly WORSE on average than the better of human or AI alone (Hedges' g = -0.23). Losses concentrated in decision-making; gains in content creation. Pairing gained where humans beat the AI and lost where the AI beat humans.",
      "supports": "That adding a human is not a control, and that undesigned pairing can subtract. The most under-absorbed result in the field.",
      "doesNotSupport": "That human-AI teams are useless. The benchmark is an oracle-selected best performer, which you rarely know in advance. Also predates current frontier models.",
      "terms": [
        "human in the loop",
        "human-AI collaboration",
        "oversight"
      ],
      "relatedPages": [
        "/research/human-ai-decision-making",
        "/research/how-should-leaders-respond-to-ai",
        "/research/ai-and-human-judgement",
        "/research/ai-and-human-intuition"
      ]
    },
    {
      "id": "dellacqua-2023",
      "citeAs": "https://thesuperskills.com/research/evidence#dellacqua-2023",
      "section": "collaboration",
      "authors": "Dell'Acqua, F. et al.",
      "year": 2023,
      "title": "Navigating the Jagged Technological Frontier: Field Experimental Evidence of the Effects of AI on Knowledge Worker Productivity and Quality",
      "publication": "Harvard Business School and BCG working paper",
      "url": "https://papers.ssrn.com/sol3/papers.cfm?abstract_id=4573321",
      "grade": "working-paper",
      "method": "Field experiment, 758 BCG consultants, tasks inside and just outside GPT-4's competence.",
      "finding": "Inside the frontier, AI-assisted consultants were dramatically better and faster. Outside it, they performed worse than consultants with no AI at all.",
      "supports": "That model competence is jagged rather than smooth, and that confident output suppresses scrutiny at exactly the wrong moment.",
      "doesNotSupport": "Where the frontier runs in your domain. That is local and must be learned.",
      "terms": [
        "jagged frontier",
        "automation bias",
        "verification"
      ],
      "relatedPages": [
        "/research/human-ai-decision-making",
        "/research/why-learn-to-prompt-is-weak-career-advice",
        "/research/ai-and-human-judgement",
        "/research/does-ai-actually-make-people-more-productive",
        "/research/will-ai-replace-programmers"
      ]
    },
    {
      "id": "dietvorst-2015",
      "citeAs": "https://thesuperskills.com/research/evidence#dietvorst-2015",
      "section": "collaboration",
      "authors": "Dietvorst, B. J., Simmons, J. P. and Massey, C.",
      "year": 2015,
      "title": "Algorithm Aversion: People Erroneously Avoid Algorithms After Seeing Them Err",
      "publication": "Journal of Experimental Psychology: General, 144(1)",
      "url": "https://marketing.wharton.upenn.edu/wp-content/uploads/2016/10/Dietvorst-Simmons-Massey-2014.pdf",
      "grade": "peer-reviewed",
      "method": "Five experiments.",
      "finding": "After seeing an algorithm err, people abandon it even when it demonstrably outperforms them.",
      "supports": "That trust in a model moves for reasons unrelated to its accuracy.",
      "doesNotSupport": "That this holds for conversational AI, which is far more recent and feels different to use.",
      "terms": [
        "algorithm aversion",
        "trust"
      ],
      "relatedPages": [
        "/research/human-ai-decision-making",
        "/research/ai-and-human-intuition",
        "/research/ai-and-expert-judgement"
      ]
    },
    {
      "id": "logg-2019",
      "citeAs": "https://thesuperskills.com/research/evidence#logg-2019",
      "section": "collaboration",
      "authors": "Logg, J. M., Minson, J. A. and Moore, D. A.",
      "year": 2019,
      "title": "Algorithm Appreciation: People Prefer Algorithmic to Human Judgment",
      "publication": "Organizational Behavior and Human Decision Processes, 151, 90-103",
      "url": "https://www.jennlogg.com/uploads/2/8/9/2/2892148/algorithm_appreciation__logg_minson_moore_2019_.pdf",
      "grade": "peer-reviewed",
      "method": "Six experiments on estimates and forecasts.",
      "finding": "People often weight algorithmic advice MORE heavily than human advice. Domain experts are the notable exception.",
      "supports": "Together with Dietvorst, that miscalibration runs in both directions and cannot be fixed by telling people to use judgement.",
      "doesNotSupport": "Which tendency dominates in any given workplace.",
      "terms": [
        "algorithm appreciation",
        "trust"
      ],
      "relatedPages": [
        "/research/human-ai-decision-making",
        "/research/ai-and-expert-judgement"
      ]
    },
    {
      "id": "zamfirescu-pereira-2023",
      "citeAs": "https://thesuperskills.com/research/evidence#zamfirescu-pereira-2023",
      "section": "collaboration",
      "authors": "Zamfirescu-Pereira, J. D., Wong, R. Y., Hartmann, B. and Yang, Q.",
      "year": 2023,
      "title": "Why Johnny Can't Prompt: How Non-AI Experts Try (and Fail) to Design LLM Prompts",
      "publication": "CHI 2023",
      "url": "https://dl.acm.org/doi/10.1145/3544548.3581388",
      "grade": "peer-reviewed",
      "method": "Design probe study with non-experts using a purpose-built prompt design tool.",
      "finding": "Non-experts approached prompting opportunistically rather than systematically, over-generalised from single successes and failures, and struggled to form an accurate model of the system.",
      "supports": "That prompting is genuinely harder than it looks, which is the strongest case FOR teaching it.",
      "doesNotSupport": "That this persists. It used 2023 models, and providers are actively engineering the difficulty away.",
      "terms": [
        "prompt engineering"
      ],
      "relatedPages": [
        "/research/why-learn-to-prompt-is-weak-career-advice"
      ]
    },
    {
      "id": "ayers-2023",
      "citeAs": "https://thesuperskills.com/research/evidence#ayers-2023",
      "section": "humanness",
      "authors": "Ayers, J. W. et al.",
      "year": 2023,
      "title": "Comparing Physician and Artificial Intelligence Chatbot Responses to Patient Questions Posted to a Public Social Media Forum",
      "publication": "JAMA Internal Medicine, 183(6), 589-596",
      "url": "https://pure.johnshopkins.edu/en/publications/comparing-physician-and-artificial-intelligence-chatbot-responses/",
      "grade": "peer-reviewed",
      "method": "Cross-sectional study, 195 real patient questions from a public forum, blind-rated by licensed healthcare professionals.",
      "finding": "Chatbot responses were rated good or very good quality 78.5 percent of the time against 22.1 percent for physicians, and empathetic or very empathetic 45.1 percent against 4.6 percent.",
      "supports": "That on the observable, textual performance of empathy, the machine already wins comfortably. The claim that empathy is safe from AI is empirically wrong.",
      "doesNotSupport": "That a machine can care for anyone. Doctors answering strangers free of charge between patients are not doing the job they trained for.",
      "terms": [
        "empathy",
        "what stays human"
      ],
      "relatedPages": [
        "/research/what-stays-human",
        "/research/outsourced-recognition"
      ]
    },
    {
      "id": "yin-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#yin-2024",
      "section": "humanness",
      "authors": "Yin, Y., Jia, N. and Wakslak, C. J.",
      "year": 2024,
      "title": "AI can help people feel heard, but an AI label diminishes this impact",
      "publication": "PNAS, 121(14), e2319112121",
      "url": "https://pure.psu.edu/en/publications/ai-can-help-people-feel-heard-but-an-ai-label-diminishes-this-imp/",
      "grade": "peer-reviewed",
      "method": "Experiments comparing AI-generated and human-written responses, with and without disclosure.",
      "finding": "AI-generated replies made recipients feel MORE heard than replies from untrained humans, and labelling the reply as AI removed the advantage.",
      "supports": "That the value of recognition is not in the words but in the belief that a person chose to attend to you. The most clarifying study in this debate.",
      "doesNotSupport": "That the label effect is stable. Norms around disclosed AI assistance are moving, and nobody has measured this over time.",
      "terms": [
        "recognition",
        "outsourced recognition",
        "empathy"
      ],
      "relatedPages": [
        "/research/outsourced-recognition",
        "/research/what-stays-human"
      ]
    },
    {
      "id": "doshi-hauser-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#doshi-hauser-2024",
      "section": "humanness",
      "authors": "Doshi, A. R. and Hauser, O. P.",
      "year": 2024,
      "title": "Generative AI enhances individual creativity but reduces the collective diversity of novel content",
      "publication": "Science Advances, 10(28)",
      "url": "https://discovery.ucl.ac.uk/id/eprint/10195027/",
      "grade": "peer-reviewed",
      "method": "Online experiment, 293 writers producing short fiction and 600 evaluators.",
      "finding": "AI-assisted stories were rated more creative, better written and more enjoyable, with the largest gains for the least creative writers, and were markedly more similar to one another.",
      "supports": "That individual creative quality and collective creative range move in opposite directions. A social dilemma: every writer is right to use it, and the literature gets duller.",
      "doesNotSupport": "That this generalises beyond one short creative task with one form of assistance.",
      "terms": [
        "creativity",
        "homogenisation"
      ],
      "relatedPages": [
        "/research/what-stays-human",
        "/research/what-is-the-human-signal",
        "/research/what-is-the-shared-prompt-review"
      ]
    },
    {
      "id": "autor-thompson-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#autor-thompson-2025",
      "section": "work",
      "authors": "Autor, D. and Thompson, N.",
      "year": 2025,
      "title": "Expertise",
      "publication": "NBER Working Paper 33941; Journal of the European Economic Association, 23(4), 1203-1271",
      "url": "https://www.nber.org/papers/w33941",
      "grade": "peer-reviewed",
      "method": "Four decades of task data across 303 US occupations, 1980-2018, with a novel content-agnostic measure of task expertise.",
      "finding": "Automation that removed the LESS expert tasks raised wages and reduced employment. Automation that removed the EXPERT tasks lowered wages and increased employment.",
      "supports": "That which tasks are automated matters more than how many, and gives a testable way to ask whether a given role is appreciating or commoditising.",
      "doesNotSupport": "Anything measured about generative AI. The data ends in 2018, so this is a lens, not a forecast.",
      "terms": [
        "expertise",
        "task automation",
        "human capability"
      ],
      "relatedPages": [
        "/research/what-is-the-judgement-premium",
        "/research/staying-valuable-in-the-age-of-ai",
        "/research/will-ai-replace-my-job",
        "/research/which-tasks-do-workers-not-want-automated",
        "/research/what-becomes-more-valuable-as-ai-gets-cheaper"
      ]
    },
    {
      "id": "humlum-vestergaard-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#humlum-vestergaard-2025",
      "section": "work",
      "authors": "Humlum, A. and Vestergaard, E.",
      "year": 2025,
      "title": "Still Waters, Rapid Currents: Early Labor Market Transformation under Generative AI",
      "publication": "NBER Working Paper 33777, revised March 2026",
      "url": "https://www.nber.org/papers/w33777",
      "grade": "working-paper",
      "method": "Adoption surveys linked to administrative labour records, roughly 25,000 workers across 7,000 Danish workplaces in 11 exposed occupations.",
      "finding": "Precise null effects on earnings and hours two years after ChatGPT, ruling out effects larger than 2 percent, alongside substantial task reorganisation and new tasks in AI oversight and integration.",
      "supports": "That the structure of work moves well before earnings do, and that pay is the slowest available indicator.",
      "doesNotSupport": "That the same holds elsewhere. Denmark is high-trust, high-wage and heavily unionised, and two years is early.",
      "terms": [
        "labour market",
        "task reorganisation",
        "AI adoption"
      ],
      "relatedPages": [
        "/research/ai-workforce-strategy",
        "/research/will-ai-replace-my-job",
        "/research/ai-and-human-judgement"
      ]
    },
    {
      "id": "eloundou-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#eloundou-2024",
      "section": "work",
      "authors": "Eloundou, T., Manning, S., Mishkin, P. and Rock, D.",
      "year": 2024,
      "title": "GPTs are GPTs: Labor market impact potential of LLMs",
      "publication": "Science, 384(6702), 1306-1308",
      "url": "https://arxiv.org/abs/2303.10130",
      "grade": "peer-reviewed",
      "method": "Human and model ratings of task exposure across occupational task descriptions.",
      "finding": "Around 80 percent of US workers could have at least 10 percent of tasks affected; about 19 percent could see at least half affected.",
      "supports": "Where pressure is likely to fall across the occupational structure.",
      "doesNotSupport": "That any job will be lost. This is exposure, not displacement, and the authors say so explicitly. It is the most misquoted number in the field.",
      "terms": [
        "task exposure",
        "labour market"
      ],
      "relatedPages": [
        "/research/will-ai-replace-my-job"
      ]
    },
    {
      "id": "bick-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#bick-2024",
      "section": "work",
      "authors": "Bick, A., Blandin, A. and Deming, D. J.",
      "year": 2024,
      "title": "The Rapid Adoption of Generative AI",
      "publication": "NBER Working Paper 32966",
      "url": "https://www.nber.org/papers/w32966",
      "grade": "working-paper",
      "method": "Nationally representative US surveys of generative AI use at work and at home.",
      "finding": "By late 2024, nearly 40 percent of US adults aged 18-64 used generative AI and 23 percent of employed respondents had used it for work in the previous week, but only 1 to 5 percent of all work hours were assisted.",
      "supports": "Enormous reach, thin penetration into actual hours. Adoption is not transformation.",
      "doesNotSupport": "Quality of use. Self-reported use counts any use at all.",
      "terms": [
        "AI adoption"
      ],
      "relatedPages": [
        "/research/ai-workforce-strategy",
        "/research/why-learn-to-prompt-is-weak-career-advice"
      ]
    },
    {
      "id": "wef-foj-2018",
      "citeAs": "https://thesuperskills.com/research/evidence#wef-foj-2018",
      "section": "institutional",
      "authors": "World Economic Forum",
      "year": 2018,
      "title": "The Future of Jobs Report 2018",
      "publication": "World Economic Forum, Geneva, September 2018",
      "url": "https://www.weforum.org/publications/the-future-of-jobs-report-2018/",
      "grade": "institutional-survey",
      "method": "Employer survey via the WEF membership community.",
      "finding": "Set out expected skill demand to 2022, with analytical thinking and innovation, active learning and creativity leading the list.",
      "supports": "What the field expected in 2018. Its predictions are now checkable, and that is the reason for including it.",
      "doesNotSupport": "A representative picture of employers. Respondents are drawn from a self-selected membership network.",
      "terms": [
        "human capability",
        "skills"
      ],
      "relatedPages": [
        "/research/human-skills-in-the-age-of-ai"
      ]
    },
    {
      "id": "wef-foj-2020",
      "citeAs": "https://thesuperskills.com/research/evidence#wef-foj-2020",
      "section": "institutional",
      "authors": "World Economic Forum",
      "year": 2020,
      "title": "The Future of Jobs Report 2020",
      "publication": "World Economic Forum, Geneva, October 2020",
      "url": "https://www.weforum.org/publications/the-future-of-jobs-report-2020/",
      "grade": "institutional-survey",
      "method": "Employer survey, conducted during the first year of the pandemic.",
      "finding": "Named critical thinking and problem solving as leading skills, and forecast large-scale reskilling need.",
      "supports": "What the field expected in 2020, including a pandemic-shaped view of remote work.",
      "doesNotSupport": "A clean read on AI. The 2020 edition is dominated by COVID-era disruption.",
      "terms": [
        "human capability",
        "skills",
        "critical thinking"
      ],
      "relatedPages": [
        "/research/human-skills-in-the-age-of-ai"
      ]
    },
    {
      "id": "wef-foj-2023",
      "citeAs": "https://thesuperskills.com/research/evidence#wef-foj-2023",
      "section": "institutional",
      "authors": "World Economic Forum",
      "year": 2023,
      "title": "The Future of Jobs Report 2023",
      "publication": "World Economic Forum, Geneva, April 2023",
      "url": "https://www.weforum.org/publications/the-future-of-jobs-report-2023/",
      "grade": "institutional-survey",
      "method": "Employer survey on expectations to 2027.",
      "finding": "Analytical thinking leads, with creative thinking second, and a growing emphasis on self-efficacy skills.",
      "supports": "The first post-ChatGPT edition, published five months after launch.",
      "doesNotSupport": "Considered judgement on generative AI. It was fielded too early for that.",
      "terms": [
        "human capability",
        "skills"
      ],
      "relatedPages": [
        "/research/human-skills-in-the-age-of-ai"
      ]
    },
    {
      "id": "wef-foj-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#wef-foj-2025",
      "section": "institutional",
      "authors": "World Economic Forum",
      "year": 2025,
      "title": "The Future of Jobs Report 2025",
      "publication": "World Economic Forum, Geneva, January 2025",
      "url": "https://www.weforum.org/publications/the-future-of-jobs-report-2025/",
      "grade": "institutional-survey",
      "method": "Employer survey on expectations to 2030.",
      "finding": "Analytical thinking is the most valued core skill, and skills gaps are named the single biggest barrier to business transformation.",
      "supports": "What employers say they want, which is a real and useful signal about demand.",
      "doesNotSupport": "What employers actually do. Stated skill preference and hiring behaviour diverge routinely.",
      "terms": [
        "human capability",
        "skills",
        "analytical thinking"
      ],
      "relatedPages": [
        "/research/human-skills-in-the-age-of-ai",
        "/research/ai-workforce-strategy",
        "/research/ai-and-human-judgement"
      ]
    },
    {
      "id": "oecd-skills-outlook-2023",
      "citeAs": "https://thesuperskills.com/research/evidence#oecd-skills-outlook-2023",
      "section": "institutional",
      "authors": "OECD",
      "year": 2023,
      "title": "OECD Skills Outlook 2023: Skills for a Resilient Green and Digital Transition",
      "publication": "OECD Publishing, Paris, November 2023",
      "url": "https://www.oecd.org/en/publications/oecd-skills-outlook-2023_27452f29-en.html",
      "grade": "institutional-modelling",
      "method": "Secondary analysis of OECD data including PISA and PIAAC, not a new survey.",
      "finding": "Analyses the skills required for green and digital transitions across member economies.",
      "supports": "A cross-national, methodologically transparent baseline on skills, from data collected to a documented standard.",
      "doesNotSupport": "Anything AI-specific and current. The underlying data collection predates the generative-AI period.",
      "terms": [
        "skills",
        "human capability"
      ],
      "relatedPages": [
        "/research/human-skills-in-the-age-of-ai"
      ]
    },
    {
      "id": "deloitte-hct-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#deloitte-hct-2025",
      "section": "institutional",
      "authors": "Deloitte",
      "year": 2025,
      "title": "2025 Global Human Capital Trends",
      "publication": "Deloitte Insights",
      "url": "https://www.deloitte.com/us/en/insights/topics/talent/human-capital-trends/2025.html",
      "grade": "institutional-survey",
      "method": "Around 10,000 business and HR leaders across 93 countries, plus separate worker, manager and executive surveys and 25 or more executive interviews.",
      "finding": "Frames the worker-organisation relationship as a set of unresolved tensions rather than a set of solved problems.",
      "supports": "Scale and breadth of practitioner sentiment.",
      "doesNotSupport": "Causal claims. It is a sentiment survey by a firm that sells the remedies it recommends, and should be read with that in view.",
      "terms": [
        "workforce",
        "human capability"
      ],
      "relatedPages": [
        "/research/ai-workforce-strategy",
        "/research/chro-guide-to-ai"
      ]
    },
    {
      "id": "microsoft-wti-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#microsoft-wti-2025",
      "section": "institutional",
      "authors": "Microsoft and LinkedIn",
      "year": 2025,
      "title": "2025 Work Trend Index Annual Report: The Year the Frontier Firm is Born",
      "publication": "Microsoft WorkLab, April 2025",
      "url": "https://www.microsoft.com/en-us/worklab/work-trend-index/2025-the-year-the-frontier-firm-is-born",
      "grade": "institutional-survey",
      "method": "31,000 knowledge workers across 31 markets, plus LinkedIn labour data and Microsoft 365 telemetry.",
      "finding": "Describes the emergence of firms organised around human-agent teams and a shift towards workers managing AI agents.",
      "supports": "Large-scale, current sentiment plus real product telemetry, which few others have.",
      "doesNotSupport": "Independence. Microsoft sells the tools whose adoption it is measuring, and telemetry measures usage rather than value.",
      "terms": [
        "AI agents",
        "AI adoption",
        "workforce"
      ],
      "relatedPages": [
        "/research/ai-agents-and-human-judgement",
        "/research/ai-workforce-strategy"
      ]
    },
    {
      "id": "stanford-hai-index-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#stanford-hai-index-2025",
      "section": "institutional",
      "authors": "Stanford HAI",
      "year": 2025,
      "title": "The 2025 AI Index Report",
      "publication": "Stanford Institute for Human-Centered AI, April 2025",
      "url": "https://hai.stanford.edu/ai-index/2025-ai-index-report",
      "grade": "compiled-review",
      "method": "Aggregation of many third-party sources across eight chapters, with public-opinion data from Ipsos and Pew.",
      "finding": "The most comprehensive annual account of AI capability, investment, adoption and public attitudes.",
      "supports": "An authoritative baseline on what AI systems can do and how they are spreading.",
      "doesNotSupport": "Much about human capability. It measures the machine side of the equation. The human side is the gap this evidence base exists to fill.",
      "terms": [
        "AI capability",
        "AI adoption"
      ],
      "relatedPages": [
        "/research/ai-workforce-strategy"
      ]
    },
    {
      "id": "mckinsey-new-future-of-work-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#mckinsey-new-future-of-work-2024",
      "section": "institutional",
      "authors": "McKinsey Global Institute",
      "year": 2024,
      "title": "A new future of work: The race to deploy AI and raise skills in Europe and beyond",
      "publication": "McKinsey Global Institute, May 2024",
      "url": "https://www.mckinsey.com/mgi/our-research/a-new-future-of-work-the-race-to-deploy-ai-and-raise-skills-in-europe-and-beyond",
      "grade": "institutional-modelling",
      "method": "Modelling for 2022-2030 across nine EU countries, the UK and the US, plus a survey of 1,100 or more C-suite executives in five countries.",
      "finding": "Projects large-scale occupational transitions and rising demand for social, emotional and higher cognitive skills.",
      "supports": "A transparent, well-documented scenario model, useful for direction.",
      "doesNotSupport": "What will happen. Scenario models are assumption-driven, and McKinsey's prior transition estimates have moved substantially between editions.",
      "terms": [
        "labour market",
        "skills"
      ],
      "relatedPages": [
        "/research/ai-workforce-strategy",
        "/research/will-ai-replace-my-job"
      ]
    },
    {
      "id": "mckinsey-deltas-2021",
      "citeAs": "https://thesuperskills.com/research/evidence#mckinsey-deltas-2021",
      "section": "institutional",
      "authors": "McKinsey and Company",
      "year": 2021,
      "title": "Defining the skills citizens will need in the future world of work",
      "publication": "McKinsey Public and Social Sector Practice, June 2021",
      "url": "https://www.mckinsey.com/industries/public-sector/our-insights/defining-the-skills-citizens-will-need-in-the-future-world-of-work",
      "grade": "institutional-survey",
      "method": "Online psychometric survey of 18,000 people across 15 countries, fielded 2019; 56 elements in 13 skill groups.",
      "finding": "Identifies distinct elements of talent, the DELTAs, associated with employment, income and job satisfaction.",
      "supports": "An unusually large individual-level dataset on skills and outcomes, which is rare in this literature.",
      "doesNotSupport": "Anything about AI. It was fielded in 2019, before the generative-AI period entirely.",
      "terms": [
        "skills",
        "human capability"
      ],
      "relatedPages": [
        "/research/human-skills-in-the-age-of-ai"
      ]
    },
    {
      "id": "nfer-skills-imperative-2035",
      "citeAs": "https://thesuperskills.com/research/evidence#nfer-skills-imperative-2035",
      "section": "institutional",
      "authors": "NFER",
      "year": 2023,
      "title": "The Skills Imperative 2035: An analysis of the demand for skills in the labour market in 2035 (Working Paper 3)",
      "publication": "National Foundation for Educational Research, with the University of Sheffield, funded by the Nuffield Foundation, May 2023",
      "url": "https://www.nfer.ac.uk/publications/the-skills-imperative-2035-an-analysis-of-the-demand-for-skills-in-the-labour-market-in-2035/",
      "grade": "institutional-modelling",
      "method": "161 skills from the US O*NET database mapped to UK occupational codes, combined with UK employment projections.",
      "finding": "Projects rising demand for a set of essential employment skills in the UK to 2035.",
      "supports": "A rare UK-specific, independently funded, methodologically documented projection.",
      "doesNotSupport": "Precision. It maps US skill data onto UK occupations, and a revised working paper corrects coding errors in the underlying labour force survey.",
      "terms": [
        "skills",
        "labour market"
      ],
      "relatedPages": [
        "/research/human-skills-in-the-age-of-ai"
      ]
    },
    {
      "id": "mckinsey-agents-robots-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#mckinsey-agents-robots-2026",
      "section": "institutional",
      "authors": "McKinsey Global Institute",
      "year": 2026,
      "title": "Agents, robots, and us: How AI reshapes work and skills in Europe",
      "publication": "McKinsey Global Institute, May 2026",
      "url": "https://www.mckinsey.com/mgi/our-research/agents-robots-and-us-how-ai-reshapes-work-and-skills-in-europe",
      "grade": "institutional-modelling",
      "method": "Task-level automation modelling across ten European economies covering more than 75 percent of regional labour force and GDP, plus job-postings data.",
      "finding": "Models how agentic AI and robotics together reshape task composition and skill demand across Europe.",
      "supports": "The most current institutional modelling of the agentic shift, and useful for the direction of task change.",
      "doesNotSupport": "Outcomes. Task exposure modelling has consistently over-predicted the pace of realised change.",
      "terms": [
        "AI agents",
        "task automation",
        "skills"
      ],
      "relatedPages": [
        "/research/ai-agents-and-human-judgement",
        "/research/ai-workforce-strategy"
      ]
    },
    {
      "id": "unicef-ai-children-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#unicef-ai-children-2025",
      "section": "institutional",
      "authors": "UNICEF Innocenti",
      "year": 2025,
      "title": "Guidance on AI and Children, Version 3.0: Recommendations for AI policies and systems that uphold child rights",
      "publication": "UNICEF Innocenti, December 2025",
      "url": "https://www.unicef.org/innocenti/reports/policy-guidance-ai-children",
      "grade": "institutional-modelling",
      "method": "Expert advisory group, multi-stakeholder consultation, peer review and a twelve-country study with children and caregivers.",
      "finding": "Sets out ten requirements and 48 recommendations for AI systems and policy affecting children.",
      "supports": "A rights-based standard developed with children rather than about them, which almost nothing else in this space does.",
      "doesNotSupport": "Empirical claims about learning or capability effects. It is a normative guidance document.",
      "terms": [
        "children",
        "education",
        "human agency"
      ],
      "relatedPages": [
        "/research/how-humans-learn-with-ai"
      ]
    },
    {
      "id": "allianz-labour-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#allianz-labour-2026",
      "section": "institutional",
      "authors": "Allianz Research",
      "year": 2026,
      "title": "Happy Labor Day? How geopolitics, immigration and AI will reshape work",
      "publication": "Allianz Trade, 30 April 2026",
      "url": "https://www.allianz-trade.com/en_global/news-insights/economic-insights/Happy-labor-day-How-geopolitics-immigration-AI-reshape-work.html",
      "grade": "institutional-modelling",
      "method": "Sectoral AI task-exposure estimates combined with national employment structures across the US, UK, Germany, France, Italy and Spain.",
      "finding": "Models the combined effect of AI, demographics and migration on labour supply and task composition.",
      "supports": "A current, cross-country modelling view from outside the consultancy sector.",
      "doesNotSupport": "Measured effects. Like all exposure modelling, it is an estimate of what could be affected.",
      "terms": [
        "labour market",
        "task exposure"
      ],
      "relatedPages": [
        "/research/will-ai-replace-my-job"
      ]
    },
    {
      "id": "pwc-jobs-barometer-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#pwc-jobs-barometer-2025",
      "section": "institutional",
      "authors": "PwC",
      "year": 2025,
      "title": "PwC Global AI Jobs Barometer 2025",
      "publication": "PwC, 2025",
      "url": "https://www.pwc.com/gx/en/issues/artificial-intelligence/job-barometer/2025/report.pdf",
      "grade": "institutional-modelling",
      "method": "Analysis of close to a billion job advertisements across multiple countries.",
      "finding": "Reports wage premiums for AI skills and shifting skill requirements in AI-exposed occupations.",
      "supports": "Large-scale observed labour demand rather than stated preference, which is a genuine strength.",
      "doesNotSupport": "Causation, and job advertisements describe what employers ask for rather than what the work requires.",
      "terms": [
        "labour market",
        "skills"
      ],
      "relatedPages": [
        "/research/will-ai-replace-entry-level-jobs"
      ]
    },
    {
      "id": "hepi-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#hepi-2026",
      "section": "learning",
      "authors": "Stephenson, R. and Armstrong, C.",
      "year": 2026,
      "title": "Student Generative AI Survey 2026",
      "publication": "HEPI Report 199, Higher Education Policy Institute with Kortext, published 12 March 2026. The third annual edition; the 2024 and 2025 editions were written by Josh Freeman",
      "url": "https://www.hepi.ac.uk/reports/student-generative-ai-survey-2026/",
      "grade": "institutional-survey",
      "method": "1,054 full-time undergraduates polled through the Savanta panel in December 2025, weighted on gender, institution type and year of study, margin of error about 3 per cent. The 2024 wave was polled by UCAS rather than Savanta, a change inside the trend line.",
      "finding": "95 per cent report using AI in at least one way and 94 per cent say they have used generative AI to help prepare assessed work, against 89 per cent in 2025 and about 53 per cent in 2024. What they use it for is mostly comprehension: explaining concepts 61 per cent, summarising an article 49, suggesting research ideas 40, structuring thoughts 39. Including AI-generated text directly in assessed work is 12 per cent, up from 8 in 2025 and 3 in 2024. 68 per cent think AI skills are essential, and only 36 per cent feel encouraged by their institution to use AI.",
      "supports": "That AI use among full-time UK undergraduates is close to universal in preparation, and that institutional encouragement lags a long way behind student behaviour.",
      "doesNotSupport": "That 94 per cent of students put AI into work that gets marked. THIS IS THE MISREADING THE FIGURE INVITES and the distinction is the whole point of the survey: the 94 covers everything from explaining a concept to drafting, and the number for including AI-generated text directly is 12. Anyone writing 'uses AI on assessed work' has converted a 12 per cent finding into a 94 per cent one with one preposition. It is also self-report, an ever-done lifetime measure rather than current practice, a commercial panel rather than a census, full-time students only, and HEPI warn that new response options were added in 2026 which mechanically makes the 94 easier to reach.",
      "terms": [
        "assessment",
        "adoption",
        "student behaviour"
      ],
      "relatedPages": [
        "/research/how-to-use-ai-at-university",
        "/research/how-to-assess-students-when-ai-can-do-the-assignment"
      ]
    },
    {
      "id": "mollick-assigning-ai-2023",
      "citeAs": "https://thesuperskills.com/research/evidence#mollick-assigning-ai-2023",
      "section": "learning",
      "authors": "Mollick, E. and Mollick, L.",
      "year": 2023,
      "title": "Assigning AI: Seven Approaches for Students, with Prompts",
      "publication": "Wharton School Research Paper, SSRN 4475995, revised September 2023. Also arXiv:2306.10052, June 2023. Not peer-reviewed in either venue",
      "url": "https://arxiv.org/abs/2306.10052",
      "grade": "argued-perspective",
      "method": "A practical framework paper. Seven roles a student or teacher can put an AI into, each with a rationale, an example prompt, an example output, the risks, and guidance. No study, no sample, no measured outcome.",
      "finding": "The seven roles are mentor, giving feedback; tutor, giving direct instruction; coach, prompting metacognition; teammate, arguing the other side; student, whom you teach in order to find out what you do not know; simulator, for practice; and tool, for the mechanical part. Each carries a named pedagogical risk, and the one attached to tool is 'outsourcing thinking, rather than work'.",
      "supports": "That there are at least seven distinct uses beyond the answer machine, which is a useful correction to a debate conducted as though there were one. The underlying pedagogical claims rest on established learning science, including Bjork on desirable difficulties and the retrieval-practice literature.",
      "doesNotSupport": "That any of it works in this form. The authors say so themselves: the approaches are 'still in their infancy and largely untested' and should be approached 'with a spirit of experimentation'. Established priors applied to an unevaluated delivery mechanism. The prompts are also anchored to mid-2023 model capability and should not be presented as current.",
      "terms": [
        "deliberate practice",
        "metacognition",
        "learning",
        "the missed reps"
      ],
      "relatedPages": [
        "/research/how-to-use-ai-at-university",
        "/research/how-humans-learn-with-ai"
      ]
    },
    {
      "id": "topaz-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#topaz-2026",
      "section": "learning",
      "authors": "Topaz, M., Roguin, N., Gupta, P., Zhang, Z. and Peltonen, L.-M.",
      "year": 2026,
      "title": "Fabricated citations: an audit across 2.5 million biomedical papers",
      "publication": "The Lancet, 407, 1779-1781. DOI 10.1016/S0140-6736(26)00603-3. Online 7 May 2026. Correspondence rather than a full article. Led from Columbia University School of Nursing and Data Science Institute",
      "url": "https://www.thelancet.com/journals/lancet/article/PIIS0140-6736(26)00603-3/fulltext",
      "grade": "peer-reviewed",
      "method": "The CITADEL pipeline over 2.5 million papers in the PubMed Central Open Access collection, January 2023 to 18 February 2026, extracting 125.6 million references of which 97.1 million carried a verifiable identifier. Title and identifier were compared; the 30,812 mismatches were routed by a language model into categories, then each flagged title was searched in PubMed, Crossref, OpenAlex and Google Scholar. A reference counted as fabricated only if it appeared in none of them.",
      "finding": "4,046 fabricated references across 2,810 papers. The rate of papers carrying at least one rose from 1 in 2,828 in 2023 to 1 in 458 in 2025 and 1 in 277 in the first seven weeks of 2026, a twelvefold increase. Review articles ran 57 per cent higher than other paper types. At the time of the audit 98.4 per cent of affected papers had received no publisher action.",
      "supports": "That references to studies which do not exist are entering the peer-reviewed literature at a rising rate, measured rather than asserted, and that the correction machinery is not catching them.",
      "doesNotSupport": "A rate for PubMed. The audit covers the PubMed Central OPEN ACCESS collection, and Nature published a correction to its own coverage on 13 May 2026 making exactly this distinction. The 2026 figure is seven weeks and the authors mark it as an incomplete observation period. It is a floor rather than a ceiling, because 28.5 million references, 22.7 per cent, lacked verifiable identifiers and were excluded. It does not establish intent: Topaz has said 91 per cent of affected papers carried only one or two, many likely honest mistakes by authors who did not check AI output. Cochrane and Northwestern have both criticised the method publicly, on transparency and on not separating citations that matter to a conclusion from those that do not. NOTE, the paper itself could not be read at source, so no limitation is quoted here as the authors' own words.",
      "terms": [
        "verification",
        "the verifier's discount",
        "fabricated citations",
        "provenance"
      ],
      "relatedPages": [
        "/research/how-to-use-ai-at-university",
        "/research/proving-you-did-the-work",
        "/research/is-ai-dangerous"
      ]
    },
    {
      "id": "charlotin-hallucination-cases",
      "citeAs": "https://thesuperskills.com/research/evidence#charlotin-hallucination-cases",
      "section": "judgement",
      "authors": "Charlotin, D.",
      "year": 2026,
      "title": "AI Hallucination Cases database",
      "publication": "Maintained at HEC Paris. A continuously updated tracker rather than a fixed publication, so any figure taken from it is a dated snapshot",
      "url": "https://www.damiencharlotin.com/hallucinations/",
      "grade": "compiled-review",
      "method": "Collects legal decisions in which a court or tribunal addresses the use of AI in more than passing reference, and includes a case only where the court has found or implied that a party relied on hallucinated material. The unit is a judicial decision, not a filing and not a sanction.",
      "finding": "Read on 5 September 2026: roughly 2,022 decisions across 42 jurisdictions. United States 1,380, Canada 217, Australia 110, the United Kingdom 69. By party, pro se litigants 1,161 and lawyers 808, with judges appearing 31 times. By nature, fabricated 1,678, misrepresented 845, false quotes 547.",
      "supports": "That fabricated and misrepresented citations reach formal proceedings often enough to be counted, and that the problem is not confined to people without training: professionals whose work is checking sources appear in it more than eight hundred times.",
      "doesNotSupport": "Anything with a stable number attached. This is one researcher's live tracker and every figure is a snapshot that will be out of date within days, so it must be cited with the date it was read. 'Filed material that does not exist' also overstates it, since only the fabricated category means that and the courts' own errors are included. And pro se litigants are not the same set as 'not lawyers', because the party field is not a binary and some cases carry more than one tag.",
      "terms": [
        "verification",
        "fabricated citations",
        "oversight"
      ],
      "relatedPages": [
        "/research/how-to-use-ai-at-university"
      ]
    },
    {
      "id": "scarfe-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#scarfe-2024",
      "section": "learning",
      "authors": "Scarfe, P., Watcham, K., Clarke, A. and Roesch, E.",
      "year": 2024,
      "title": "A real-world test of artificial intelligence infiltration of a university examinations system: A 'Turing Test' case study",
      "publication": "PLOS ONE, 19(6), e0305354. DOI 10.1371/journal.pone.0305354. Published 26 June 2024. Open access, data at OSF",
      "url": "https://doi.org/10.1371/journal.pone.0305354",
      "grade": "peer-reviewed",
      "method": "33 fake student accounts submitted wholly GPT-4-written answers into the live examinations system of the School of Psychology and Clinical Language Sciences at the University of Reading, summer 2023. Five undergraduate modules across all years, about 5 per cent of submissions on each. Two formats, four 200-word answers in a 2.5-hour window and one 1,500-word essay in an 8-hour window, both unsupervised at home. Markers were staff and trained postgraduates, marking anonymously and entirely unaware of the study.",
      "finding": "94 per cent of the AI submissions were not detected, and 97 per cent went undetected on the stricter test of a marker actually mentioning AI. Across the five modules there was an 83.4 per cent probability, by resampling, that the AI submissions would outscore an equal-sized random draw of real students, an advantage of just over half a classification boundary. AI lost on one module only, a finalist module carrying three AI submissions.",
      "supports": "That wholly machine-written work passed through a real examinations system substantially undetected and scored above the human median, under blind marking, in a live setting rather than a demonstration.",
      "doesNotSupport": "How much real cheating occurs, which the authors state plainly: 'we have no way to estimate the proportion of students in our sample who used AI'. It also flatters detection rather than damning it. In their words the 6 per cent detection rate 'likely overestimates our ability to detect real-world use of AI to cheat in exams', because a real student would not take so naively obvious an approach. The comparison group may itself contain AI-assisted work. And the 83.4 per cent is a resampling probability against a median, NOT a count of head-to-head comparisons won, which is how it is usually restated.",
      "terms": [
        "assessment",
        "detection",
        "synthetic seniority",
        "the missed reps"
      ],
      "relatedPages": [
        "/research/how-to-use-ai-at-university",
        "/research/how-to-assess-students-when-ai-can-do-the-assignment",
        "/research/does-ai-detection-work"
      ]
    },
    {
      "id": "uk-university-ai-penalties-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#uk-university-ai-penalties-2026",
      "section": "learning",
      "authors": "Multiple Freedom of Information investigations: The Times, The Student Eye and The Scotsman",
      "year": 2026,
      "title": "Recorded penalties for AI misuse in UK universities",
      "publication": "Three separate FOI investigations covering different institutions and different academic years. They must not be aggregated",
      "url": "https://thestudenteye.substack.com/p/exclusive-university-of-bristol-sees",
      "grade": "compiled-review",
      "method": "Freedom of Information requests to universities, compiled by three outlets. The Times to the 24 Russell Group members for 2024-25; The Student Eye to the University of Bristol, May 2025; The Scotsman to Scottish institutions via Miles Briggs MSP, March 2025, covering 2023-24.",
      "finding": "Russell Group: 2,053 recorded punishments in 2024-25 against roughly 700 the year before, from about 350,000 students, with four members disclosing expulsions (UCL, Imperial, Glasgow and Leeds). Bristol: 526 penalties in 2023-24, against 153 the previous year and 7 in 2021-22. Scotland: 1,051 cases in 2023-24 against 131 in 2022-23, of which Abertay alone recorded 351 cases with 342 upheld.",
      "supports": "That formal academic penalties for AI misuse have risen steeply and are now numbered in thousands, and that a student who assumes this is theoretical is wrong.",
      "doesNotSupport": "A national picture, a trend in behaviour, or one dataset. Seven of the 24 Russell Group universities do not record AI investigations at all, so 2,053 is a floor across an incomplete sample. The three sets cover DIFFERENT YEARS, 2024-25 for the Russell Group and 2023-24 for Scotland and Bristol, and they overlap institutionally, so they cannot be added together. Most importantly the universities' own position is that much of the rise reflects newly created recording categories rather than more cheating: Abertay introduced 'unacceptable AI use' as a category only in 2023 and accounts for a third of the Scottish total. Cases, penalties and upheld findings are three different counts and are routinely conflated in coverage.",
      "terms": [
        "assessment",
        "detection",
        "academic misconduct"
      ],
      "relatedPages": [
        "/research/how-to-use-ai-at-university",
        "/research/does-ai-detection-work"
      ]
    },
    {
      "id": "fleisig-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#fleisig-2024",
      "section": "humanness",
      "authors": "Fleisig, E., Smith, G., Bossi, M., Rustagi, I., Yin, X. and Klein, D.",
      "year": 2024,
      "title": "Linguistic Bias in ChatGPT: Language Models Reinforce Dialect Discrimination",
      "publication": "Proceedings of EMNLP 2024, 13541-13564. University of California, Berkeley. Preprint arXiv:2406.08818",
      "url": "https://aclanthology.org/2024.emnlp-main.750/",
      "grade": "peer-reviewed",
      "method": "Ten varieties of English: Standard American and Standard British, plus African American, Indian, Irish, Jamaican, Kenyan, Nigerian, Scottish and Singaporean. Around fifty native-speaker messages per variety on everyday topics, put to GPT-3.5 Turbo and GPT-4 in a plain condition and an imitate-the-style condition. Annotated for ten linguistic features per variety at Krippendorff's alpha 0.97, then evaluated by native speakers recruited through Prolific, at least eleven per variety.",
      "finding": "Responses to non-standard varieties carried more stereotyping (19 per cent worse), more demeaning content (25 per cent worse), less comprehension (9 per cent worse) and more condescension (15 per cent worse), all significant after correction. The more striking result is retention: a Standard American English input keeps 77.9 per cent of its distinctive features in the reply and Standard British 72.2 per cent, while five of the eight minoritised varieties keep only 2 to 3 per cent. Indian, Nigerian and Kenyan English sit between the two groups at 10 to 16 per cent, and the authors report that retention rate tracks the estimated maximum speaker population of a variety, which they take as a proxy for training data volume. Asked to imitate, GPT-4 improved comprehension and warmth while stereotyping rose 18 per cent.",
      "supports": "That homogenisation towards one dialect is measured rather than impressionistic, and that it is severe: the model does not merely prefer Standard American English, it strips almost every marker of the others. The clearest evidence the estate holds that convergence on a single voice has a distributional cost, borne by particular speakers.",
      "doesNotSupport": "That models exaggerate or caricature dialects, which is the claim usually attached to this paper in secondary coverage. The measured default is the opposite, features are stripped rather than amplified, and the exaggeration line rests on one annotator's free-text comment about Singlish. It also says nothing about human speech: this is model output, not people.",
      "terms": [
        "homogenisation",
        "standardisation",
        "the human signal",
        "voice"
      ],
      "relatedPages": [
        "/research/does-ai-make-everyone-think-alike",
        "/research/what-stays-human"
      ]
    },
    {
      "id": "yakura-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#yakura-2024",
      "section": "humanness",
      "authors": "Yakura, H., Lopez-Lopez, E., Brinkmann, L., de la Serna, I., Kirfel, L., Gupta, P., Soraperra, I., Eisenmann, T. F., Wulff, D. U. and Rahwan, I.",
      "year": 2024,
      "title": "Empirical evidence of Large Language Model's influence on human spoken communication",
      "publication": "arXiv:2409.01754. Max Planck Institute for Human Development. v1 submitted 3 September 2024, v2 30 June 2025, v3 8 July 2025, v4 16 July 2026. STILL NOT PEER-REVIEWED after four versions: the arXiv record carries no journal reference and no DOI other than the preprint server's own, checked 7 September 2026. Substantially replaced by the authors: v1's YouTube analysis is gone, and the six-author list on the version most secondary coverage cites is now ten",
      "url": "https://arxiv.org/abs/2409.01754",
      "grade": "working-paper",
      "method": "TWO DIFFERENT STUDIES UNDER ONE IDENTIFIER, and the distinction is the whole entry. v1 (2024): 279,480 YouTube transcripts from academic-institution channels, filtered from nearly three million videos, spanning 36 months before and 18 months after the release of ChatGPT, with a hierarchical Bayesian regression on the monthly log frequency of videos containing a given word and a change point at the release date. v4 (2026): 737,083 hours of conversation from 824,634 podcast episodes screened for unscripted speech, analysed by synthetic control, in which each treated word's monthly relative document frequency is compared against a counterfactual built from donor words with near-zero GPT scores and matched pre-release trajectories; plus a preregistered experiment with 496 participants who played a referential image-guessing game with a chatbot covertly prompted to use particular synonyms, then described the target image aloud after a three-minute arithmetic distractor.",
      "finding": "v1 reported that words the model favours rose over the 18 months after release: adept by 51 per cent, delve 48, meticulous 40, realm 35, with the word list taken from Liang and colleagues at Stanford, who compared 10,000 human-written abstracts with their ChatGPT-edited versions. v4 reports the same direction on a far better design and does not report those percentages: it finds an abrupt rise in GPT-preferred words in spontaneous speech, attributes it to the release by synthetic control, with a placebo test at p = 0.01 for the most-cited of those words, and finds experimentally that a brief chatbot interaction led participants to adopt its words as their own, persisting past a distractor task and confirmed in forced lexical choice.",
      "supports": "That the vocabulary of the machine appears in human material at scale and on a timeline consistent with the model's arrival, and, on the current version, that a short exposure is sufficient to move an individual's active vocabulary under randomisation. The strongest quantified evidence the estate has for a pattern everybody claims to notice.",
      "doesNotSupport": "THE 51 PER CENT SHOULD NOT BE QUOTED AT ALL. It is adept alone, not a general figure; its outcome is the PREVALENCE OF VIDEOS containing a word, not how often speakers use it; its population is academic institutional channels rather than speakers at large; and when the authors hand-checked fifty videos containing the most-cited of those words, 32 per cent showed signs of the speaker reading from a script. Decisively, neither the word adept nor the figure 51 appears anywhere in v4: the authors have withdrawn the analysis it came from. The current version has its own limits, stated by the authors: the corpus is English-only and drawn from a self-selected, public-facing population of podcast hosts and guests; the main analysis covers the first 18 months and treats OpenAI's models as the dominant driver, while deployment has since fragmented across providers, making attribution harder; residual interference in the synthetic control cannot be ruled out; and the outcome is still the share of episodes containing a word rather than a speaker's frequency of use. Cite the direction, never the number.",
      "terms": [
        "homogenisation",
        "AI-speak",
        "the human signal"
      ],
      "relatedPages": [
        "/research/does-ai-make-everyone-think-alike"
      ]
    },
    {
      "id": "mackworth-1948",
      "citeAs": "https://thesuperskills.com/research/evidence#mackworth-1948",
      "section": "judgement",
      "authors": "Mackworth, N.H.",
      "year": 1948,
      "title": "The Breakdown of Vigilance during Prolonged Visual Search",
      "publication": "Quarterly Journal of Experimental Psychology, 1(1), 6-21. DOI 10.1080/17470214808416738. Much of the same material was reported at greater length in Researches on the Measurement of Human Performance, MRC Special Report 268, HMSO, 1950, which is where the 1950 date in circulation comes from",
      "url": "https://doi.org/10.1080/17470214808416738",
      "grade": "peer-reviewed",
      "method": "The Clock Test, built to simulate radar and sonar watchkeeping. An unmarked clock face whose pointer jumps in equal steps about once a second, making a rare double jump at irregular intervals, which the observer reports by pressing a button. Two-hour watches, analysed in half-hour blocks. Signal probability in the operational task was a little over half a per cent.",
      "finding": "Detection of rare signals fell measurably between the first and second half-hour block and continued to decline across the watch. The vigilance decrement, and the founding result of the field.",
      "supports": "That sustained attention to rare events degrades within the first hour, as a property of the task rather than of the person, which is the mechanism behind every oversight arrangement that asks someone to watch a mostly correct system.",
      "doesNotSupport": "Anything more precise about timing than the block structure allows. Because Mackworth analysed in half-hour blocks, the decline cannot be located inside the first thirty minutes, and the widely repeated claim that accuracy falls within the first half hour states the result more sharply than the design supports. The specific percentage figures in circulation come from secondary literature: the original is paywalled and was not read at source for this entry.",
      "terms": [
        "vigilance decrement",
        "automation complacency",
        "oversight"
      ],
      "relatedPages": [
        "/research/what-are-shallow-jobs",
        "/research/the-invisible-work-of-oversight"
      ]
    },
    {
      "id": "endsley-kiris-1995",
      "citeAs": "https://thesuperskills.com/research/evidence#endsley-kiris-1995",
      "section": "judgement",
      "authors": "Endsley, M. R. and Kiris, E. O.",
      "year": 1995,
      "title": "The Out-of-the-Loop Performance Problem and Level of Control in Automation",
      "publication": "Human Factors, 37(2), 381-394. DOI 10.1518/001872095779064555. Publisher record and abstract read at source 7 September 2026; the FULL TEXT IS PAYWALLED at Sage and was not read",
      "url": "https://journals.sagepub.com/doi/10.1518/001872095779064555",
      "grade": "peer-reviewed",
      "method": "Laboratory experiment on an automobile navigation task automated by an expert system, run at five levels of operator control on Endsley's own scale: manual; decision support, where the system suggests; consensual, where the system acts with the operator's consent; monitored, where the system acts unless vetoed; and full automation with no operator interaction. Situation awareness measured against decision time following an induced failure of the expert system. SAMPLE SIZE NOT RECORDED HERE: it is not stated in the abstract and the full text could not be read, so no participant count should be quoted from this entry.",
      "finding": "Situation awareness was lower under fully automated and semi-automated conditions than under manual performance, and low situation awareness corresponded with out-of-the-loop performance decrements in decision time after the expert system failed. The out-of-the-loop effect was significantly greater under full automation than under the intermediate levels, so the level of operator control moderated the loss. The authors attribute the decrement principally to the shift from active to passive information processing.",
      "supports": "That the capacity to take over from a failed automated system falls as the level of automation rises, and that the fall is moderated by design rather than by effort. It is the naming source for the out-of-the-loop performance problem and the origin of the five-level control scale that later work, including Kaber and Endsley, builds on.",
      "doesNotSupport": "Anything about generative AI. This is one laboratory task from 1995, automated by an expert system producing route recommendations on a defined problem, and a model producing fluent text across every task is a different object. It is also a single study with no participant count available here, so it should not be cited as though its effect size were known. The estate's standard is the paper and this entry rests on the publisher's abstract plus the author's own later summary; anyone with institutional access should re-read the results section and correct it.",
      "terms": [
        "the out-of-the-loop performance problem",
        "situation awareness",
        "levels of automation",
        "automation complacency"
      ],
      "relatedPages": [
        "/research/what-is-the-out-of-the-loop-performance-problem"
      ]
    },
    {
      "id": "endsley-1996",
      "citeAs": "https://thesuperskills.com/research/evidence#endsley-1996",
      "section": "judgement",
      "authors": "Endsley, M. R.",
      "year": 1996,
      "title": "Automation and Situation Awareness",
      "publication": "In R. Parasuraman and M. Mouloua (Eds.), Automation and Human Performance: Theory and Applications, 163-181. Mahwah, NJ: Lawrence Erlbaum. Author's manuscript read in full at source 7 September 2026. Note that its own reference list gives the Endsley and Kiris page range as 381-194, a transposition; Sage confirms 381-394",
      "url": "https://maritimesafetyinnovationlab.org/wp-content/uploads/2019/12/Automation-and-Situation-Awareness-Endsley.pdf",
      "grade": "compiled-review",
      "method": "Narrative review chapter in an edited scholarly volume, drawing together the automation and situation-awareness literature up to 1996 and summarising the author's own experimental work. Not peer-reviewed in the journal sense; reviewed by the volume's editors.",
      "finding": "Sets out three mechanisms by which automation produces the out-of-the-loop performance problem: changes in vigilance and complacency associated with monitoring, the assumption of a passive rather than an active role in controlling the system, and changes in the quality or form of feedback reaching the operator. Reporting Endsley and Kiris, it gives the split that the journal abstract does not: only Level 2 situation awareness, comprehension of what the data mean in relation to operational goals, was damaged, while Level 1, perception of the data itself, was unaffected. Since the displayed information did not change between conditions and monitoring effects were too small to account for the decrement, the author attributes the loss to passivity alone.",
      "supports": "That the harm measured in this literature falls on comprehension rather than on attention, in operators who are monitoring effectively. That is the finding which makes an approval step an inadequate remedy, and it is the reason the concept belongs in a discussion of AI oversight and not only in aviation.",
      "doesNotSupport": "It is a review chapter and reports no new data of its own. The Level 1 against Level 2 split is the author's summary of her own study, written while that study was still in press, so it is one step from the results section and should be read as such. Its aviation examples are accident narratives drawn from NTSB reports rather than measurements, and its figure of 88 per cent of reviewed commercial aviation accidents involving a situation-awareness problem is the author citing her own unpublished 1994 conference paper and is not verified here.",
      "terms": [
        "the out-of-the-loop performance problem",
        "situation awareness",
        "levels of automation",
        "automation complacency"
      ],
      "relatedPages": [
        "/research/what-is-the-out-of-the-loop-performance-problem"
      ]
    },
    {
      "id": "klein-feltmate-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#klein-feltmate-2025",
      "section": "judgement",
      "authors": "Klein, R.M. and Feltmate, B.B.T.",
      "year": 2025,
      "title": "The vigilance decrement: its first 75 years",
      "publication": "Frontiers in Cognition, 4. DOI 10.3389/fcogn.2025.1632885",
      "url": "https://doi.org/10.3389/fcogn.2025.1632885",
      "grade": "compiled-review",
      "method": "A review of seventy-five years of vigilance research from Mackworth onwards.",
      "finding": "The decrement itself has held across the literature. Its mechanism has not: whether the decline reflects falling sensitivity or a shifting response criterion remains disputed, and the sensitivity account has recently been challenged.",
      "supports": "That the effect is durable enough to design around, and that citing it as settled science overstates the position.",
      "doesNotSupport": "Any particular mechanism, and therefore any intervention that depends on one. A page arguing that oversight fails for a specific cognitive reason is going beyond what this review supports.",
      "terms": [
        "vigilance decrement",
        "oversight"
      ],
      "relatedPages": [
        "/research/what-are-shallow-jobs"
      ]
    },
    {
      "id": "tomasev-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#tomasev-2026",
      "section": "collaboration",
      "authors": "Tomasev, N., Franklin, M. and Osindero, S.",
      "year": 2026,
      "title": "Intelligent AI Delegation",
      "publication": "Google DeepMind. arXiv:2602.11865, submitted 12 February 2026. Preprint, not peer-reviewed",
      "url": "https://arxiv.org/abs/2602.11865",
      "grade": "working-paper",
      "method": "A framework paper proposing how AI agents should decompose problems and delegate across other agents and people, covering task assignment, monitoring, trust, permissions and accountability in long delegation chains. No experiment, no data. The capability material is section 5.6, Risk of De-skilling.",
      "finding": "The authors name oversight readiness as a property a future workforce may lack, arguing that expertise is built through the repetitive execution of narrowly scoped tasks, that those are the tasks most likely to be delegated to agents first, and that fully automating them would deprive junior staff of the experience needed for strategic judgement. Their proposed remedies are unusual for a frontier lab: curriculum-aware task routing that allocates work inside a junior's zone of proximal development, and a delegation framework that should occasionally introduce minor inefficiencies by routing tasks to humans deliberately, to maintain their skills.",
      "supports": "That the missing rungs argument is being reached independently by people building delegation infrastructure rather than only by those writing about its consequences, and that at least one frontier lab has proposed deliberate inefficiency as a capability-preservation mechanism.",
      "doesNotSupport": "Anything measured. It is a preprint framework paper with no empirical component, so it establishes that the risk is taken seriously by practitioners and not that the risk has been observed. The proposed routing systems also assume an organisation can assess what a junior can currently do, which is the unsolved part.",
      "terms": [
        "oversight readiness",
        "the missing rungs",
        "synthetic seniority",
        "delegation"
      ],
      "relatedPages": [
        "/research/what-is-oversight-readiness",
        "/research/what-is-a-frontier-firm",
        "/research/who-manages-ai-agents"
      ]
    },
    {
      "id": "ke-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#ke-2026",
      "section": "learning",
      "authors": "Ke, Y., Jin, L., Ong, J.C.L., Thirunavukarasu, A.J., Car, J., Cheung, C.Y., Tham, Y.C., Ting, D.S.W., Ong, M.E.H., Compton, S., Narayan, A., Keane, P.A., Wong, T.Y., Bates, D.W., Tan, P. and Liu, N.",
      "year": 2026,
      "title": "AI-induced never-skilling in medical education",
      "publication": "Nature Medicine, 32(6), 1997-2006, published 22 May 2026. DOI 10.1038/s41591-026-04438-y",
      "url": "https://www.nature.com/articles/s41591-026-04438-y",
      "grade": "argued-perspective",
      "method": "A Perspective, not primary research. Sixteen authors across sixteen institutions argue a three-part taxonomy of how AI can interrupt the formation of clinical competence, and propose an untested three-phase protective framework. The clinical evidence it leans on is borrowed, chiefly Budzyn and colleagues on colonoscopy.",
      "finding": "Separates three distinct failures. Deskilling is the degradation of established competence in clinicians already trained. Mis-skilling is the acquisition of incorrect reasoning patterns through uncritical adoption of erroneous or biased AI output. Never-skilling is the failure to form foundational competence during training, when AI substitutes for the cognitive effort that would have built it. The authors predict a state they call false proficiency: competence that appears real but depends on the AI remaining available.",
      "supports": "That the three failures are different problems requiring different responses, and that the entry-level case is not simply deskilling applied to younger people. Never-skilling has no baseline to return to, which is what makes it a distinct category rather than a matter of degree.",
      "doesNotSupport": "That never-skilling occurs. The authors disclaim this themselves and do so more than once: 'Direct causal evidence linking AI exposure during training to competency failure in medical trainees does not exist', and the abstract concedes that direct evidence from medical training is absent. It is a risk model, explicitly not an established phenomenon. Prevalence, severity and reversibility are all stated as unknown, and the proposed framework is untested. Cite it for the taxonomy and as the origin of the terms, never as evidence of harm.",
      "terms": [
        "never-skilling",
        "mis-skilling",
        "deskilling",
        "false proficiency",
        "synthetic seniority"
      ],
      "relatedPages": [
        "/research/will-ai-replace-entry-level-jobs",
        "/research/synthetic-seniority",
        "/research/how-humans-learn-with-ai"
      ]
    },
    {
      "id": "ehsan-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#ehsan-2026",
      "section": "frontline",
      "authors": "Ehsan, U., Passi, S., Saha, K., McNutt, T., Riedl, M.O. and Alcorn, S.",
      "year": 2026,
      "title": "From Future of Work to Future of Workers: Addressing Asymptomatic AI Harms to Foster Dignified Human-AI Interaction",
      "publication": "Proceedings of the CHI Conference on Human Factors in Computing Systems (CHI '26), ACM. DOI 10.1145/3772318.3791081. Preprint arXiv:2601.21920, 29 January 2026",
      "url": "https://dl.acm.org/doi/10.1145/3772318.3791081",
      "grade": "qualitative-field-study",
      "method": "Twelve months of situated fieldwork through the first year of routine use of a commercial AI-assisted radiotherapy treatment-planning system across a five-site North American hospital group. 42 participants: 15 radiation oncologists, 12 medical physicists, 7 dosimetrists, 8 administrators, with 2 to 24 years of experience. 52 think-aloud sessions, 24 interviews staged across months 2 to 11, five participatory workshops, 63 hours transcribed, analysed by grounded theory.",
      "finding": "Measured operational gains and reported capability loss ran together. Planning cycles shortened by roughly 15 per cent and confidence rose, while by month nine several dosimetrists said their unaided proficiency had worsened over the year, one saying they had grown slower without the tool. The authors name two things the estate has nowhere else: intuition rust, the gradual dulling of expert judgement beneath intact output, and identity commoditisation, the erosion of professional dignity as practitioners describe becoming AI babysitters, button-pushers and bystanders in their own practice. They call these harms asymptomatic because the organisation's own measures showed only the improvement.",
      "supports": "That the instruments an organisation uses to judge an AI deployment can register the gain and be structurally blind to the cost, and that practitioners perceive the loss long before any dashboard does. It is the best available account of what deskilling feels like from inside a high-stakes clinical specialty, and the only source here on occupational identity.",
      "doesNotSupport": "Any measured deskilling. The skill claims are self-reported, plus one informal unaided exercise with a couple of dosimetrists during a workshop. There is no controlled comparison, no pre and post measurement and no quantified skill outcome, so it cannot be set beside Budzyn as a second measured result. IMPORTANT, this study is widely misdescribed in secondary summaries as radiologists using generative AI. It is neither. The participants plan radiotherapy treatment rather than interpret images, and the system is optimisation-based, not generative. Do not repeat that description.",
      "terms": [
        "deskilling",
        "capability debt",
        "occupational identity",
        "intuition rust",
        "identity commoditisation",
        "the invisible work of oversight"
      ],
      "relatedPages": [
        "/research/capability-debt",
        "/research/which-professions-face-the-greatest-deskilling-risk",
        "/research/ai-and-human-judgement"
      ]
    },
    {
      "id": "dixon-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#dixon-2026",
      "section": "work",
      "authors": "Dixon, J.C.",
      "year": 2026,
      "title": "I surveyed workers to see if AI had caused job losses and was surprised by the findings",
      "publication": "The Conversation, 27 August 2026. YouGov survey commissioned by the author, College of the Holy Cross. Described by the author as a study in progress and not peer-reviewed",
      "url": "https://theconversation.com/i-surveyed-workers-to-see-if-ai-had-caused-job-losses-and-was-surprised-by-the-findings-290100",
      "grade": "institutional-survey",
      "method": "1,250 employed US workers surveyed online by YouGov between 30 July and 4 August 2026, 25 questions, weighted to the employed US population on age, gender, race and education. Opt-in panel. The author discloses an AI consulting business.",
      "finding": "About 3 per cent said they had lost a job to AI since 2023, against roughly 6 per cent who said they held a job that did not exist before AI and about 9 per cent reporting an AI-related promotion. Around 95 per cent said no, with 3 to 4 per cent unsure.",
      "supports": "That self-attributed AI job loss is rare among people currently in work, and that reported AI-created gain in the same sample runs at twice the reported loss. Useful as a counterweight to displacement figures drawn from payroll data, which cannot ask anyone why.",
      "doesNotSupport": "Displacement in the workforce, and the reason is structural rather than a matter of sampling error. Every respondent was employed when surveyed, in Dixon's own words 'none were jobless, whether due to AI or another reason'. Anyone displaced by AI and still out of work is excluded from the numerator and the denominator alike, so the 3 per cent counts only those who lost a job and have since found another. It is a floor among survivors, not an estimate of displacement. It is also not a measure of fear: the question asked what happened, not what respondents expect.",
      "terms": [
        "entry-level employment",
        "displacement",
        "self-report"
      ],
      "relatedPages": [
        "/research/will-ai-replace-entry-level-jobs",
        "/research/what-is-the-ai-employment-gap"
      ]
    },
    {
      "id": "budzyn-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#budzyn-2025",
      "section": "frontline",
      "authors": "Budzyn, K., Roman'czyk, M., Kitala, D. et al.",
      "year": 2025,
      "title": "Endoscopist deskilling risk after exposure to artificial intelligence in colonoscopy: a multicentre, observational study",
      "publication": "The Lancet Gastroenterology and Hepatology, 10(10), 896-903. DOI 10.1016/S2468-1253(25)00133-5",
      "url": "https://pubmed.ncbi.nlm.nih.gov/40816301/",
      "grade": "peer-reviewed",
      "method": "Retrospective observational study nested in the ACCEPT trial, four Polish endoscopy centres. 1,443 colonoscopies performed WITHOUT AI assistance (795 before and 648 after AI was introduced) by 19 endoscopists averaging 27.6 years of experience, ranging from 8 to 39 years.",
      "finding": "Adenoma detection rate in unassisted colonoscopy fell from 28.4 percent before AI exposure to 22.4 percent after, a drop of 6.0 percentage points (p=0.0089; adjusted odds ratio 0.69).",
      "supports": "Measured deskilling in highly experienced professionals, in unassisted performance, within months of routine AI exposure. The strongest direct evidence that capability degrades when a tool takes over the judgement, rather than merely a plausible mechanism.",
      "doesNotSupport": "Causation with certainty: it is observational, not randomised, and other changes over the period cannot be fully excluded. It is also one procedure in one country, and detection rate is a proxy for skill rather than skill itself.",
      "terms": [
        "deskilling",
        "capability debt"
      ],
      "relatedPages": [
        "/research/capability-debt",
        "/research/ai-and-human-judgement",
        "/research/how-humans-learn-with-ai",
        "/research/ai-and-expert-judgement",
        "/research/is-ai-dangerous"
      ]
    },
    {
      "id": "yu-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#yu-2024",
      "section": "frontline",
      "authors": "Yu, F., Moehring, A., Banerjee, O., Salz, T., Agarwal, N. and Rajpurkar, P.",
      "year": 2024,
      "title": "Heterogeneity and predictors of the effects of AI assistance on radiologists",
      "publication": "Nature Medicine, 30(3), 837-849",
      "url": "https://pubmed.ncbi.nlm.nih.gov/38504016/",
      "grade": "peer-reviewed",
      "method": "140 radiologists, 15 chest X-ray diagnostic tasks, roughly 5,190 observations, randomised AI assistance, with empirical-Bayes shrinkage to separate genuine individual differences from noise.",
      "finding": "The effect of AI assistance diverged sharply between radiologists, from strongly positive to strongly negative. Experience, subspecialty and prior familiarity with AI all failed to predict who would benefit, and lower performers did not consistently gain.",
      "supports": "That the effect of AI assistance on expert performance is individual and currently unpredictable, so a policy of giving everyone the tool will help some professionals and harm others with no way to tell in advance which.",
      "doesNotSupport": "That AI assistance is bad on average, or that the pattern holds outside diagnostic imaging.",
      "terms": [
        "human-AI collaboration",
        "expertise"
      ],
      "relatedPages": [
        "/research/human-ai-decision-making",
        "/research/staying-valuable-in-the-age-of-ai",
        "/research/ai-and-human-judgement",
        "/research/ai-and-expert-judgement"
      ]
    },
    {
      "id": "lee-nursing-homes-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#lee-nursing-homes-2024",
      "section": "frontline",
      "authors": "Lee, Y. S., Iizuka, T. and Eggleston, K.",
      "year": 2024,
      "title": "Robots and Labor in Nursing Homes",
      "publication": "NBER Working Paper 33116",
      "url": "https://www.nber.org/system/files/working_papers/w33116/w33116.pdf",
      "grade": "working-paper",
      "method": "Original facility-level panel of Japanese nursing homes, using regional robot subsidies as an instrument for adoption.",
      "finding": "Robot adoption RAISED employment and improved retention, most strongly for non-regular staff, reallocated worker effort towards direct care, and improved quality: less use of physical restraint and fewer pressure ulcers.",
      "supports": "That automation can absorb routine physical work and upgrade the human job rather than hollow it. The cleanest counter-case in this evidence base, drawn from care work rather than knowledge work.",
      "doesNotSupport": "That this generalises. Japanese long-term care faces acute labour shortage, so robots substituted for vacancies rather than for people, which is a very particular condition.",
      "terms": [
        "automation",
        "care work"
      ],
      "relatedPages": [
        "/research/what-stays-human",
        "/research/ai-workforce-strategy"
      ]
    },
    {
      "id": "dauth-2021",
      "citeAs": "https://thesuperskills.com/research/evidence#dauth-2021",
      "section": "frontline",
      "authors": "Dauth, W., Findeisen, S., Suedekum, J. and Woessner, N.",
      "year": 2021,
      "title": "The Adjustment of Labor Markets to Robots",
      "publication": "Journal of the European Economic Association, 19(6), 3104-3153",
      "url": "https://academic.oup.com/jeea/article-abstract/19/6/3104/6179884",
      "grade": "peer-reviewed",
      "method": "German administrative worker and plant data, 1994 to 2014, with a shift-share instrument for robot exposure.",
      "finding": "Incumbent workers largely kept their jobs and moved into new, higher-quality tasks within their original plants. The cost fell instead on young labour-market entrants, who shifted away from vocational manufacturing training towards university.",
      "supports": "That the damage from automation falls on skill FORMATION rather than skill possession. Twenty years of German manufacturing data making the missing-rungs argument before anyone applied it to knowledge work.",
      "doesNotSupport": "That generative AI will behave like industrial robots. The technologies and the tasks differ substantially.",
      "terms": [
        "missing rungs",
        "manufacturing",
        "early careers"
      ],
      "relatedPages": [
        "/research/missing-rungs",
        "/research/will-ai-replace-entry-level-jobs"
      ]
    },
    {
      "id": "hosseini-lichtinger-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#hosseini-lichtinger-2026",
      "section": "frontline",
      "authors": "Hosseini Maasoum, S. M. and Lichtinger, G.",
      "year": 2026,
      "title": "Generative AI as Seniority-Biased Technological Change: Evidence from U.S. Resume and Job Posting Data",
      "publication": "SSRN working paper, DOI 10.2139/ssrn.5425555. Harvard University. First version 31 August 2025; version read is dated 25 May 2026 on its own masthead, obtained from the author's site. SSRN records the current revision as 6 June 2026, 109 pages. NOT peer-reviewed, no journal or NBER placement. Main text read in full at source 9 September 2026; the supplemental appendix, A.1 to A.23, was NOT read",
      "url": "https://papers.ssrn.com/sol3/papers.cfm?abstract_id=5425555",
      "grade": "working-paper",
      "method": "Revelio Labs resume data via WRDS, LinkedIn-derived, merged with Revelio job postings. 281,111 firms, 161,192,969 positions from 2015, roughly 66 million unique workers, and 198,773,384 postings from 2021. Firms included if they recorded at least 20 new positions between January 2021 and March 2025; HR intermediaries and the roughly 800 largest firms excluded. Juniors are Revelio's Entry and Junior seniority levels, seniors Associate and above. Adoption is identified from job postings for people hired to integrate generative AI into the firm, by keyword flag then LLM classification: 131,845 of 198.8 million postings, 0.066 per cent, giving 10,433 adopter firms, 3.71 per cent of firms and 16 per cent of employment. Four designs against non-adopting controls: difference-in-differences, triple-difference with firm-by-time and industry-by-seniority-by-time fixed effects, a triple-difference by occupational exposure, and a staggered event study using Callaway and Sant'Anna. Errors clustered at firm level. A separate task-level analysis maps 355,013 postings from 1,000 firms onto O*NET tasks and tests how task bundles change.",
      "finding": "Junior employment at adopting firms fell about 9 per cent relative to non-adopters six quarters after diffusion, and 8 per cent eight quarters after adoption in the staggered design, while senior employment showed no comparable break. The fall concentrates in exposed occupations: high-exposure junior employment contracts roughly 7 log points against low-exposure between 2022Q4 and 2025Q1, while the senior coefficient continues to rise. The decomposition attributes the fall to hiring, not exits. Junior hiring fell by 4.006 per firm-quarter (SE 0.221) against a pre-period mean of 5.059, which the authors describe as about an 80 per cent reduction. Junior separations ALSO fell, by 1.089 (SE 0.163), roughly a quarter of the hiring decline. Promotion rates rose slightly, by 0.033 percentage points (SE 0.014). At task level, a one standard deviation increase in a task's exposure is associated with roughly a four percentage point larger contraction in junior task bundles at adopters, with the senior interaction running the other way.",
      "supports": "The first firm-level, within-firm measurement of what happens to junior employment when a firm adopts generative AI, at a scale no other study has. Pre-trends are flat back to 2015, covering an earlier tightening cycle. A postings placebo points to falling demand rather than a labour supply shock. The decisive contribution is the mechanism: this is a hiring story, not a redundancy story. Separations fell. Firms stopped opening the door rather than pushing people out of it, which is why the effect is invisible in unemployment figures and visible only to people who cannot get in. The task-level test shows exposed tasks being removed from junior job descriptions specifically, which is the missing-rungs claim measured directly rather than inferred.",
      "doesNotSupport": "Causation. The authors write throughout that adoption IS ASSOCIATED WITH the decline, call the evidence suggestive, and state in their conclusion that unobserved confounders may remain and that the window, 2023 to 2025, is short. It is an unrefereed working paper: no journal, no NBER number, and the headline employment estimates are read off event-study figures with no printed standard errors, so only the flows and task tables carry reportable precision. The data are LinkedIn profiles, which tilt towards managerial, professional and high-information work; the authors now show that tilt is stable across the period and absorbed by fixed effects, which protects the comparison but does not make the sample representative of the labour force. The adoption measure misses informal use inside firms, which the authors say should bias estimates towards zero. They cannot test the mechanism they propose, which is that firms cut junior hiring in anticipation rather than because tasks were already automated. And they flag the missing intercept problem themselves: this does not aggregate to an economy-wide claim without further assumptions. Nothing here shows that the juniors not hired were worse off, or that the capability they would have built is gone rather than delayed.",
      "terms": [
        "missing rungs",
        "synthetic seniority",
        "early careers",
        "capability debt"
      ],
      "relatedPages": [
        "/research/missing-rungs",
        "/research/will-ai-replace-entry-level-jobs",
        "/research/do-apprenticeships-still-work"
      ]
    },
    {
      "id": "kanazawa-2022",
      "citeAs": "https://thesuperskills.com/research/evidence#kanazawa-2022",
      "section": "frontline",
      "authors": "Kanazawa, K., Kawaguchi, D., Shigeoka, H. and Watanabe, Y.",
      "year": 2022,
      "title": "AI, Skill, and Productivity: The Case of Taxi Drivers",
      "publication": "NBER Working Paper 30612; published in Management Science, 72(2), 1376-1388 (2026)",
      "url": "https://www.nber.org/papers/w30612",
      "grade": "peer-reviewed",
      "method": "Driver-level data from a Japanese taxi fleet through the rollout of an AI demand-prediction system.",
      "finding": "Productivity gains accrued almost entirely to LOW-skilled drivers, narrowing the gap between best and worst by 14 percent.",
      "supports": "That the novice-boost pattern found in customer support and software also appears in manual frontline work.",
      "doesNotSupport": "Included deliberately as a disconfirming case for a tidy story. Anyone arguing that AI levels up white-collar workers while degrading frontline ones has to explain this result, which runs the other way.",
      "terms": [
        "productivity",
        "expertise"
      ],
      "relatedPages": [
        "/research/staying-valuable-in-the-age-of-ai",
        "/research/ai-and-expert-judgement"
      ]
    },
    {
      "id": "nilsson-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#nilsson-2025",
      "section": "frontline",
      "authors": "Nilsson, A. et al. (Karolinska Institutet)",
      "year": 2025,
      "title": "Algorithmic management is associated with psychological distress, musculoskeletal pain, and occupational accidents: a cross-sectional study in logistics",
      "publication": "International Archives of Occupational and Environmental Health, 98",
      "url": "https://link.springer.com/article/10.1007/s00420-025-02180-5",
      "grade": "peer-reviewed",
      "method": "Survey of Swedish logistics workers, February to July 2024, 978 respondents (592 drivers, 378 warehouse), using an eleven-item algorithmic-management exposure scale with adjusted models.",
      "finding": "Higher exposure to algorithmic management was associated with greater psychological distress, more occupational accidents and more musculoskeletal pain.",
      "supports": "That where AI meets frontline work the output is measured in bodies rather than in output quality, a category almost entirely absent from white-collar productivity research.",
      "doesNotSupport": "Causation: it is cross-sectional and self-reported. Workers under strain may also perceive management as more algorithmic.",
      "terms": [
        "algorithmic management",
        "logistics"
      ],
      "relatedPages": [
        "/research/ai-workforce-strategy",
        "/research/how-should-leaders-respond-to-ai"
      ]
    },
    {
      "id": "kesavan-2022",
      "citeAs": "https://thesuperskills.com/research/evidence#kesavan-2022",
      "section": "frontline",
      "authors": "Kesavan, S., Lambert, S. J., Williams, J. C. and Pendem, P. K.",
      "year": 2022,
      "title": "Doing Well by Doing Good: Improving Retail Store Performance with Responsible Scheduling Practices at the Gap, Inc.",
      "publication": "Management Science, 68(11), 7818-7836",
      "url": "https://pubsonline.informs.org/doi/10.1287/mnsc.2021.4291",
      "grade": "peer-reviewed",
      "method": "Randomised field experiment across 28 Gap stores in San Francisco and Chicago over nine months, November 2015 to August 2016, analysed as intent-to-treat.",
      "finding": "Restoring schedule predictability and worker control raised productivity 5.1 percent, with sales up 3.3 percent and labour hours down 1.8 percent.",
      "supports": "That algorithmic optimisation of scheduling is not merely harsh but operationally counterproductive: giving humans back control improved the numbers the algorithm was optimising.",
      "doesNotSupport": "Anything directly about generative AI. It concerns algorithmic scheduling, which is an older and different technology.",
      "terms": [
        "algorithmic management",
        "retail",
        "agency"
      ],
      "relatedPages": [
        "/research/how-should-leaders-respond-to-ai"
      ]
    },
    {
      "id": "iab-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#iab-2024",
      "section": "international",
      "authors": "Institut fur Arbeitsmarkt- und Berufsforschung (IAB), Germany",
      "year": 2024,
      "title": "Folgen des technologischen Wandels fur den Arbeitsmarkt (Consequences of technological change for the labour market: it is above all the highly qualified who feel digitalisation)",
      "publication": "IAB-Kurzbericht 5/2024, published in German",
      "url": "https://doku.iab.de/kurzber/2024/kb2024-05.pdf",
      "grade": "institutional-modelling",
      "method": "The Substituierbarkeitspotenziale series, 2022 wave. Three independent coders score more than 9,000 tasks in the BERUFENET expert database across roughly 4,600 occupations.",
      "finding": "Substitutability rose about ten percentage points for degree-level expert occupations between 2019 and 2022, and was roughly flat for helper occupations. IAB frames AI as relief for skills shortages rather than as displacement.",
      "supports": "That the German expert assessment puts the pressure on the highly qualified, which inverts the assumption that automation threatens the least skilled first.",
      "doesNotSupport": "Realised outcomes. Substitutability is technical potential assessed by coders, not what employers did.",
      "terms": [
        "skills",
        "deskilling",
        "international"
      ],
      "relatedPages": [
        "/research/will-ai-replace-my-job",
        "/research/human-skills-in-the-age-of-ai"
      ]
    },
    {
      "id": "laboria-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#laboria-2024",
      "section": "international",
      "authors": "LaborIA (French Ministry of Labour, Inria and Matrice)",
      "year": 2024,
      "title": "Etude des impacts de l'IA sur le travail: Rapport d'enquete LaborIA Explorer (Study of the impacts of AI on work)",
      "publication": "LaborIA, published in French",
      "url": "https://www.laboria.ai/wp-content/uploads/2024/05/Rapport-denquete-LaborIA-Explorer.pdf",
      "grade": "institutional-survey",
      "method": "Telephone survey of 250 decision-makers in firms with more than 50 staff (42 with AI deployed), longitudinal interviews with 10 decision-makers across three waves, and six ethnographic field sites.",
      "finding": "Names a conflit de rationalite, a clash of rationalities: managers justify AI by error reduction (81 percent), performance (75 percent) and removing drudgery (74 percent), while fieldwork shows workers becoming the system's de facto trainers.",
      "supports": "That what management believes AI is doing and what workers experience it doing can diverge systematically inside the same organisation. A work-psychology and ergonomics frame rather than task-exposure modelling.",
      "doesNotSupport": "Scale. The qualitative core rests on six sites and ten repeated interviews.",
      "terms": [
        "organisation design",
        "international"
      ],
      "relatedPages": [
        "/research/ai-workforce-strategy",
        "/research/how-should-leaders-respond-to-ai"
      ]
    },
    {
      "id": "diwabe-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#diwabe-2025",
      "section": "international",
      "authors": "BAuA, ZEW, IAB and BIBB, Germany",
      "year": 2025,
      "title": "Digitalisierung und Wandel der Beschaftigung, DiWaBe 2.0 (Digitalisation and the transformation of employment)",
      "publication": "Bundesanstalt fur Arbeitsschutz und Arbeitsmedizin, published in German",
      "url": "https://www.baua.de/EN/Service/Publications/Report/F2573",
      "grade": "institutional-survey",
      "method": "Representative 2024 survey of roughly 9,800 employees subject to social insurance, linkable to administrative employer and employee records.",
      "finding": "More than half already use AI at work but largely informally. Use ranges from about a third of unqualified workers to around 80 percent of those with a degree or Meister qualification. There was NO difference in training participation between AI users and non-users.",
      "supports": "That AI use is spreading through workplaces without any corresponding increase in training, which is the adoption-without-redesign pattern measured directly at national scale.",
      "doesNotSupport": "What that absence of training does to capability over time. The survey is a single 2024 snapshot.",
      "terms": [
        "AI adoption",
        "skills",
        "international"
      ],
      "relatedPages": [
        "/research/ai-workforce-strategy",
        "/research/chro-guide-to-ai"
      ]
    },
    {
      "id": "jilpt-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#jilpt-2025",
      "section": "international",
      "authors": "Japan Institute for Labour Policy and Training (JILPT)",
      "year": 2025,
      "title": "Survey on the impact of workplace AI adoption on working styles (Research Series No. 256)",
      "publication": "JILPT, published in Japanese, designed with the OECD",
      "url": "https://www.jil.go.jp/institute/research/2025/256.html",
      "grade": "institutional-survey",
      "method": "22,000 employees, stratified on 2020 Census occupation, employment type, sex and age. Fieldwork May to June 2024.",
      "finding": "Only 12.9 percent report any firm AI use and 8.4 percent use it themselves. Among users, reports of improved job quality and wellbeing outweighed reports of decline, and the gain was markedly larger where the employer had consulted staff and funded training.",
      "supports": "That the effect of AI on how work feels is conditional on how the employer introduced it, not determined by the technology. Direct empirical support for the design-rather-than-drift argument, from a 22,000-person sample.",
      "doesNotSupport": "Long-run capability effects, and Japanese adoption rates are far below US levels so the user group is early and unusual.",
      "terms": [
        "AI adoption",
        "agency",
        "international"
      ],
      "relatedPages": [
        "/research/ai-workforce-strategy",
        "/research/design-versus-drift"
      ]
    },
    {
      "id": "kdi-2023",
      "citeAs": "https://thesuperskills.com/research/evidence#kdi-2023",
      "section": "international",
      "authors": "Korea Development Institute (KDI)",
      "year": 2023,
      "title": "Changes in the labour market due to artificial intelligence and policy directions (Research Report 2023-03)",
      "publication": "KDI, published in Korean",
      "url": "https://www.kdi.re.kr/research/reportView?pub_no=18370",
      "grade": "institutional-modelling",
      "method": "Expert and GPT-4 capability ratings applied to Korean occupational profiles, a KDI survey of 800 firms in September 2023, and firm-panel econometrics.",
      "finding": "38.8 percent of jobs are technically automatable across more than 70 percent of their tasks, yet only 2.7 percent of firms with ten or more staff had adopted AI. Realised effects showed no aggregate employment change, lower earnings, and the impact concentrated on YOUNGER, tertiary-educated workers and women.",
      "supports": "That the gap between technical potential and actual adoption is enormous, and that where effects appear they fall on the young and educated rather than the low-skilled.",
      "doesNotSupport": "That the Korean pattern transfers. Korea has unusually high tertiary education rates and a distinctive labour market.",
      "terms": [
        "labour market",
        "early careers",
        "international"
      ],
      "relatedPages": [
        "/research/will-ai-replace-entry-level-jobs",
        "/research/will-ai-replace-my-job"
      ]
    },
    {
      "id": "cas-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#cas-2025",
      "section": "international",
      "authors": "Lu, Y. and Gui, L. (Bulletin of the Chinese Academy of Sciences)",
      "year": 2025,
      "title": "Analysis of the impact of artificial intelligence technology on employment and income in China",
      "publication": "Bulletin of the Chinese Academy of Sciences, 40(4), 642-651, published in Chinese",
      "url": "http://old2022.bulletin.cas.cn/publish_article/2025/4/20250408.htm",
      "grade": "institutional-modelling",
      "method": "Policy synthesis of the Chinese empirical literature and official statistics, under National Social Science Fund major project 23ZDA100.",
      "finding": "Between 2018 and 2023 the substitution effect outweighed complementarity: a one percent rise in industrial robots reduced firm labour demand by 0.18 percent. Roughly 200 million people, 27 percent of employment, are in flexible work with 37 percent social-insurance coverage.",
      "supports": "That in the largest manufacturing economy the measured balance so far has been substitution rather than augmentation, and the policy response centres on social security redesign rather than retraining.",
      "doesNotSupport": "Comparability with service-sector generative AI in high-income economies. This is largely industrial robotics.",
      "terms": [
        "labour market",
        "manufacturing",
        "international"
      ],
      "relatedPages": [
        "/research/will-ai-replace-my-job"
      ]
    },
    {
      "id": "cepal-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#cepal-2026",
      "section": "international",
      "authors": "Jung, J. and Katz, R. (CEPAL / ECLAC)",
      "year": 2026,
      "title": "Impacto economico de la inteligencia artificial en America Latina (The economic impact of artificial intelligence in Latin America)",
      "publication": "United Nations Economic Commission for Latin America and the Caribbean, published in Spanish",
      "url": "https://www.cepal.org/es/publicaciones/81909-impacto-economico-la-inteligencia-artificial-america-latina-transformacion",
      "grade": "institutional-modelling",
      "method": "Theoretical and econometric modelling of AI's macroeconomic effect through skilled-labour productivity across the region.",
      "finding": "The gains run through skilled labour, and the binding constraint across Latin America is human-capital formation and low investment. The regional risk is UNDER-adoption rather than displacement.",
      "supports": "That the framing dominant in rich economies, where the worry is AI doing too much, inverts in middle-income economies, where the worry is that it will not arrive at all.",
      "doesNotSupport": "Firm-level outcomes; it is a macro model.",
      "terms": [
        "labour market",
        "skills",
        "international"
      ],
      "relatedPages": [
        "/research/ai-workforce-strategy"
      ]
    },
    {
      "id": "funcas-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#funcas-2026",
      "section": "international",
      "authors": "Rodriguez-Fernandez, M. (Funcas)",
      "year": 2026,
      "title": "Inteligencia artificial y mercado de trabajo en Espana (Artificial intelligence and the labour market in Spain)",
      "publication": "Funcas working paper, published in Spanish",
      "url": "https://www.funcas.es/documentos_trabajo/inteligencia-artificial-y-mercado-de-trabajo-en-espana-exposicion-ocupacional-efectos-sobre-el-empleo-y-adopcion-empresarial/",
      "grade": "institutional-modelling",
      "method": "The Felten AI occupational exposure index remapped to Spanish occupational classifications, combined with the Q4 2025 Labour Force Survey.",
      "finding": "Spain shows medium-high exposure at 27.4 percent but low automation risk at 5.9 percent, against an OECD average near 12 percent, because of its interpersonal and physical occupational mix.",
      "supports": "That national occupational structure, not technology, determines exposure. An economy weighted towards interpersonal and physical work is structurally less automatable.",
      "doesNotSupport": "Outcomes. Exposure indices remain estimates of what could be affected.",
      "terms": [
        "task exposure",
        "international"
      ],
      "relatedPages": [
        "/research/will-ai-replace-my-job"
      ]
    },
    {
      "id": "azim-premji-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#azim-premji-2026",
      "section": "international",
      "authors": "Azim Premji University",
      "year": 2026,
      "title": "State of Working India 2026: Youth in the Labour Market",
      "publication": "Azim Premji University, Bengaluru",
      "url": "https://azimpremjiuniversity.edu.in/publications/2026/report/swi-2026",
      "grade": "institutional-modelling",
      "method": "Analysis of National Sample Survey employment data 1983 to 2011, all quarters of the Periodic Labour Force Survey 2017 to 2024, plus AISHE, NCVT-MIS and CMIE-CPHS.",
      "finding": "Roughly five million graduates enter the Indian labour market each year against about 2.8 million finding work, with graduate unemployment near 40 percent for 15 to 25 year olds. The report explicitly declines to attribute this to AI.",
      "supports": "That in the world's most populous labour market the early-career crisis is a demand-side bottleneck that long predates AI. A necessary corrective to reading every graduate hiring problem as an AI story.",
      "doesNotSupport": "Anything about AI's effect in India, which it deliberately does not claim.",
      "terms": [
        "early careers",
        "jobs",
        "international"
      ],
      "relatedPages": [
        "/research/will-ai-replace-entry-level-jobs"
      ]
    },
    {
      "id": "insee-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#insee-2026",
      "section": "international",
      "authors": "INSEE, France",
      "year": 2026,
      "title": "Note de conjoncture: digital investment, artificial intelligence and youth employment",
      "publication": "Institut national de la statistique et des etudes economiques, March 2026, published in French",
      "url": "https://www.insee.fr/fr/statistiques/fichier/8907419/ndc-mars-2026-ecl-AI.pdf",
      "grade": "institutional-modelling",
      "method": "National accounts and quarterly employment files, with an error-correction model estimated 1990Q1 to 2019Q4.",
      "finding": "French employment of 15 to 29 year olds, excluding apprentices, fell 7.4 percent year on year in IT services, 5.8 percent in publishing and 3.7 percent in management consulting in Q4 2025, against minus 0.7 percent across the market sector overall.",
      "supports": "A European national-statistics office finding the same entry-level pattern reported in the US, which makes the signal considerably harder to dismiss as an American artefact.",
      "doesNotSupport": "That AI caused it. INSEE explicitly cautions against attributing the fall to AI alone.",
      "terms": [
        "early careers",
        "jobs",
        "international"
      ],
      "relatedPages": [
        "/research/will-ai-replace-entry-level-jobs",
        "/research/will-ai-replace-my-job"
      ]
    },
    {
      "id": "maguire-2000",
      "citeAs": "https://thesuperskills.com/research/evidence#maguire-2000",
      "section": "learning",
      "authors": "Maguire, E. A., Gadian, D. G., Johnsrude, I. S., Good, C. D., Ashburner, J., Frackowiak, R. S. J. and Frith, C. D.",
      "year": 2000,
      "title": "Navigation-related structural change in the hippocampi of taxi drivers",
      "publication": "PNAS, 97(8), 4398-4403",
      "url": "https://www.pnas.org/doi/10.1073/pnas.070039597",
      "grade": "peer-reviewed",
      "method": "Cross-sectional structural MRI. 16 right-handed male London taxi drivers with more than 1.5 years driving, against scans of 50 healthy right-handed male non-taxi-drivers.",
      "finding": "Posterior hippocampi were significantly larger in taxi drivers than in controls, and hippocampal volume correlated with time spent driving a taxi, positively in the posterior and negatively in the anterior hippocampus.",
      "supports": "That sustained, effortful spatial practice is associated with measurable structural difference in the adult brain, and that the association scales with how long the practice has continued.",
      "doesNotSupport": "Causation. It is cross-sectional, so it cannot separate the practice building the brain from a particular brain selecting into the job. It says nothing about what happens when the practice stops, nothing about AI, and nothing about knowledge work.",
      "terms": [
        "deliberate practice",
        "skill acquisition",
        "spatial memory",
        "The Knowledge"
      ],
      "relatedPages": [
        "/research/what-is-deliberate-practice",
        "/research/how-fast-do-skills-decay"
      ]
    },
    {
      "id": "woollett-maguire-2011",
      "citeAs": "https://thesuperskills.com/research/evidence#woollett-maguire-2011",
      "section": "learning",
      "authors": "Woollett, K. and Maguire, E. A.",
      "year": 2011,
      "title": "Acquiring 'the Knowledge' of London's Layout Drives Structural Brain Changes",
      "publication": "Current Biology, 21(24), 2109-2114, 20 December 2011",
      "url": "https://www.cell.com/current-biology/fulltext/S0960-9822(11)01267-X",
      "grade": "peer-reviewed",
      "method": "Longitudinal structural MRI over four years. 79 male trainee London taxi drivers studying for the Knowledge, and 31 male non-taxi-driver controls, scanned before and after. Average-IQ adults, real training rather than a laboratory task.",
      "finding": "In those who qualified, acquiring an internal spatial representation of London was associated with a selective increase in grey matter volume in the posterior hippocampi, with concomitant changes to their memory profile. In the authors' words, no structural brain changes were observed in trainees who failed to qualify or in control participants. The gain in posterior hippocampus came alongside costs elsewhere in the memory profile.",
      "supports": "The strongest available evidence that the practice itself does the work rather than selection. Same starting cohort, same training, and the structural change appears only in those who completed it. It also shows the trade: the capability gained is paid for with capability elsewhere, which is skill substitution observed in tissue rather than argued.",
      "doesNotSupport": "Anything about AI, and anything about removal. This is acquisition, over four years, in one domain. It does not show that the structure regresses when the practice is delegated to a machine, and no study has shown that. The frequently circulated claim that GPS use produces measurable cognitive decline or early-onset dementia is a prediction rather than a finding, and cannot be traced to a study of this kind.",
      "terms": [
        "deliberate practice",
        "skill acquisition",
        "spatial memory",
        "The Knowledge",
        "skill substitution"
      ],
      "relatedPages": [
        "/research/what-is-deliberate-practice",
        "/research/how-fast-do-skills-decay",
        "/research/what-is-the-substitution-myth"
      ]
    },
    {
      "id": "spa-hajj-ai-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#spa-hajj-ai-2026",
      "section": "international",
      "authors": "Ministry of Interior, Kingdom of Saudi Arabia",
      "year": 2026,
      "title": "Smart Predictive Technologies Enhance Pilgrim Safety and Crowd Management",
      "publication": "Saudi Press Agency, Makkah, 30 May 2026, 1447 AH Hajj season; read at source 4 September 2026",
      "url": "https://www.spa.gov.sa/en/N2603232",
      "grade": "operator-account",
      "method": "The state news agency reporting the Ministry of Interior's own account of its Hajj operation. No methodology, no figures, no independent evaluation.",
      "finding": "The ministry states it implemented predictive analytics to anticipate and mitigate congestion and hazardous conditions before they occurred, using an intelligent framework powered by AI and data analytics to accelerate strategic decision-making and, in its own words, to support field commanders. Oversight is described as running through a digital framework of performance indicators under the supervision of specialised personnel.",
      "supports": "That a state operating one of the largest recurring crowd operations in the world describes AI-driven prediction as an input to command decisions, and says so in the language of supporting rather than replacing the commander.",
      "doesNotSupport": "Anything measured. There is no figure for accuracy, no override rate, no comparison against unaided command judgement, and no independent evaluation of whether the oversight functions. It is a state news agency reporting a ministry on its own performance.",
      "terms": [
        "automation bias",
        "human oversight",
        "prediction",
        "command decisions"
      ],
      "relatedPages": [
        "/research/ai-and-work-in-the-gulf",
        "/research/what-is-meaningful-human-oversight"
      ]
    },
    {
      "id": "sdaia-baseer-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#sdaia-baseer-2025",
      "section": "international",
      "authors": "Moquim, S. A., Vice President, Saudi Data and Artificial Intelligence Authority, interviewed by Saeed al-Abyad",
      "year": 2025,
      "title": "SDAIA: Saudi AI Platform Baseer Boosts Crowd, Security Control During Hajj",
      "publication": "Asharq Al-Awsat, dateline Jeddah, 4 June 2025; read at source 4 September 2026",
      "url": "https://english.aawsat.com/gulf/5150726-sdaia-saudi-ai-platform-baseer-boosts-crowd-security-control-during-hajj",
      "grade": "operator-account",
      "method": "On-the-record interview with the operating authority's vice president. Descriptive throughout, with no evaluation and no data.",
      "finding": "SDAIA operates Baseer, built with the Ministry of Interior, using AI algorithms and computer vision on live feeds to detect crowd density and distribution within the Grand Mosque and to pinpoint overcrowded zones such as the Tawaf area moment by moment, stated as enabling authorities to act swiftly to prevent overcrowding or stampedes. Companion platforms Sawaher and Sawaher Qiyada analyse live security camera feeds. The Smart Makkah Operations Center coordinates them, and biometric systems run at 12 international airports across 8 countries under the Makkah Route initiative.",
      "supports": "That the same public authority which published Saudi Arabia's AI ethics principles, including the requirement that irreversible or life-and-death decisions trigger human oversight and final determination, also builds and operates the system to which that requirement applies.",
      "doesNotSupport": "That the oversight works, or that anyone has tested it. Nothing here measures how often a commander overrides the system, what happens when the prediction is wrong, or whether unaided crowd-reading skill is maintained. The speaker heads the authority whose systems he is describing.",
      "terms": [
        "human oversight",
        "automation bias",
        "who supervises",
        "computer vision"
      ],
      "relatedPages": [
        "/research/ai-and-work-in-the-gulf",
        "/research/what-is-meaningful-human-oversight",
        "/research/who-supervises-work-they-cannot-do"
      ]
    },
    {
      "id": "yu-speedup-illusion-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#yu-speedup-illusion-2026",
      "section": "judgement",
      "authors": "Yu, S., Cheng, M., Jabbar, A., Sucholutsky, I., Collins, K. M., Jurafsky, D. and Hawkins, R. D.",
      "year": 2026,
      "title": "Cognitive offloading and the speedup illusion in human-AI interaction",
      "publication": "Proceedings of the 48th Annual Meeting of the Cognitive Science Society; also arXiv 2605.23177",
      "url": "https://arxiv.org/abs/2605.23177",
      "grade": "peer-reviewed",
      "method": "Preregistered behavioural study, N=1,237, on simple cognitive tasks. Compared forecast completion times against actual completion times, with and without AI assistance, against a control condition in which participants imagined help from another person.",
      "finding": "Actual completion times did not differ between independent and AI-assisted completion, while participants predicted AI would be significantly faster. The same bias did not appear when participants imagined help from another person. Participants also reported lower subjective effort with AI at equivalent completion times, so time and effort came apart.",
      "supports": "That people are miscalibrated about AI time savings specifically rather than about assistance in general, and that reported effort is not a proxy for elapsed time.",
      "doesNotSupport": "Anything about professional work, output quality or long-horizon tasks. The tasks were short and simple by design, and a forecasting error is not the same thing as a productivity claim.",
      "terms": [
        "cognitive offloading",
        "self-report",
        "productivity",
        "metacognition"
      ],
      "relatedPages": [
        "/research/usage-theatre",
        "/research/the-unclaimed-hour",
        "/research/what-is-cognitive-offloading"
      ]
    },
    {
      "id": "brynjolfsson-canaries-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#brynjolfsson-canaries-2026",
      "section": "work",
      "authors": "Brynjolfsson, E., Chandar, B. and Chen, R.",
      "year": 2026,
      "title": "Canaries in the Coal Mine? Six Facts about the Recent Employment Effects of Artificial Intelligence",
      "publication": "Stanford Digital Economy Lab, updated August 2026",
      "url": "https://digitaleconomy.stanford.edu/publication/canaries-in-the-coal-mine-six-facts-about-the-recent-employment-effects-of-artificial-intelligence/",
      "grade": "working-paper",
      "method": "ADP payroll microdata covering millions of US workers, comparing employment by age and by occupational AI exposure since the release of ChatGPT.",
      "finding": "No widespread economy-wide displacement. But employment among 22 to 25 year olds in highly AI-exposed occupations sits about 19 percent below where it would be had it tracked similarly aged workers in less-exposed occupations. The underlying levels matter and are easily lost: employment of that age group in the two most exposed quintiles fell about 11 percent between November 2022 and June 2026, while the same age group in the three least exposed quintiles grew about 10 percent. The 19 is the distance between those two, not a fall of 19. The divergence runs through reduced hiring rather than increased separations, and declines concentrate in occupations where AI substitutes for human tasks; where it complements, employment is flat or rising, especially for experienced workers.",
      "supports": "That the entry-level effect is real, measurable in payroll data rather than inferred, and specific to substitution rather than to AI exposure as such.",
      "doesNotSupport": "Economy-wide job destruction, which the authors explicitly rule out on current evidence. Nor does it establish causation: youth hiring is sensitive to interest rates, cohort size and hiring freezes, and the design is observational. Two further cautions, both from the paper itself. First, the phrase 'below trend' is wrong and is how this figure is usually repeated: the comparison is with same-aged workers in less-exposed occupations, not with a historical trend line and not with older workers. Second, the figures in circulation are not a worsening time series. 13 and 16 percent come from an earlier regression estimator, 15 and 19 from the descriptive kept-pace measure the authors now prefer, so quoting them in sequence as though the effect grew misrepresents a change of method as a change in the world. Note also that the ADP figures widely cited alongside this paper are the same payroll data reported differently, not independent corroboration.",
      "terms": [
        "entry-level employment",
        "youth employment",
        "substitution versus complementarity",
        "hiring"
      ],
      "relatedPages": [
        "/research/will-ai-replace-entry-level-jobs",
        "/research/missing-rungs",
        "/research/the-best-writing-on-ai",
        "/research/what-is-the-ai-employment-gap"
      ]
    },
    {
      "id": "metr-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#metr-2025",
      "section": "collaboration",
      "authors": "Becker, J., Rush, N., Barnes, E. and Rein, D. (METR)",
      "year": 2025,
      "title": "Measuring the Impact of Early-2025 AI on Experienced Open-Source Developer Productivity",
      "publication": "METR, 10 July 2025; arXiv:2507.09089, v2 25 July 2025. No journal reference as at 9 September 2026.",
      "url": "https://metr.org/blog/2025-07-10-early-2025-ai-experienced-os-dev-study/",
      "grade": "working-paper",
      "method": "Randomised controlled trial. 16 experienced open-source developers drawn from repositories averaging over 22,000 stars and a million lines of code, 246 real issues from their own projects, each randomly assigned to permit or prohibit AI tools. Tasks averaged about two hours. Screens recorded, implementation time self-reported, developers paid 150 dollars an hour. Tooling was mostly Cursor Pro with Claude 3.5 and 3.7 Sonnet. Twenty candidate explanations for the slowdown were tested and five judged contributory.",
      "finding": "Developers were measured as 19 percent SLOWER when permitted to use AI tools. They had forecast a 24 percent speed-up beforehand, and after completing the tasks and experiencing the slowdown, still estimated AI had made them about 20 percent faster.",
      "supports": "That self-reported productivity gain is an unreliable measure of actual productivity gain, and that the error can run in the opposite direction to the truth by a wide margin.",
      "doesNotSupport": "That AI slows all developers or all software work, and NOT the current position. Sixteen participants, all experienced, all working on large mature codebases they knew well, using early-2025 tooling. METR themselves withdrew this as a current signal on 24 February 2026: see metr-2026-update. The durable finding is the perception gap, not the 19 per cent.",
      "terms": [
        "productivity",
        "perception gap",
        "self-report",
        "software development"
      ],
      "relatedPages": [
        "/research/what-is-the-metr-study",
        "/research/the-best-writing-on-ai",
        "/research/usage-theatre",
        "/research/what-is-human-ai-collaboration",
        "/research/ai-and-human-judgement",
        "/research/what-becomes-more-valuable-as-ai-gets-cheaper",
        "/research/the-mid-career-squeeze",
        "/research/how-do-you-measure-ai-adoption-properly",
        "/research/how-will-ai-change-consulting",
        "/research/the-unclaimed-hour",
        "/research/what-is-the-substitution-myth",
        "/research/does-ai-actually-make-people-more-productive",
        "/research/will-ai-replace-programmers"
      ]
    },
    {
      "id": "pwc-jobs-barometer-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#pwc-jobs-barometer-2026",
      "section": "institutional",
      "authors": "PwC",
      "year": 2026,
      "title": "Global AI Jobs Barometer 2026",
      "publication": "PwC, 15 June 2026",
      "url": "https://www.pwc.com/gx/en/news-room/press-releases/2026/pwc-2026-ai-jobs-barometer.html",
      "grade": "compiled-review",
      "method": "Analysis of more than a billion job advertisements across six continents, comparing skill requirements, advertised wages and job availability by occupational AI exposure. The entry-level analysis rests on 2.4 million US entry-level advertisements.",
      "finding": "PwC describe a two-track labour market. In professionalised roles, where AI automates routine tasks and the advertisement leans on human judgement and expertise, jobs grow at twice the rate and advertised salaries 42 per cent faster than in democratised roles, where AI makes the work easier for a non-expert. Radiologists and recruiters are their examples of the first, IT service managers and medical secretaries of the second. The average advertised wage premium for AI skills reached 62 per cent, against 56 in the 2025 edition and 25 in 2024, while jobs requiring AI skills grew 69 per cent against 9 per cent for the market as a whole. Companies most exposed to AI grew headcount 52 per cent against 36, and wages 24 per cent against 17. At entry level, the roles most exposed to AI are seven times more likely to require traditionally senior human-intensive skills such as leadership, creativity or face-to-face interaction; those roles grew 35 per cent since 2019 while other entry-level roles fell 10 per cent.",
      "supports": "That demand-side signals in job advertisements are moving fast and are splitting by whether AI takes the routine work or the role itself, which is the same direction Autor and Thompson find historically. The entry-level result is the strongest demand-side evidence the estate holds for juniors being asked for senior capability.",
      "doesNotSupport": "Wage or employment outcomes for actual workers, and nothing at all about the judgement premium as a price. Job advertisements are stated employer demand, not revealed price or realised hiring, and the salary figures are advertised salary. PwC has a commercial interest in the AI-skills market it is measuring. The release is also inconsistent about which skills the entry-level finding covers: a summary bullet says judgement and leadership, the body says leadership, creativity or face-to-face interactions. The body wording is the one to quote.",
      "terms": [
        "wage premium",
        "skills demand",
        "job advertisements",
        "human skills",
        "judgement premium"
      ],
      "relatedPages": [
        "/research/what-is-the-judgement-premium",
        "/research/the-best-writing-on-ai",
        "/research/human-skills-in-the-age-of-ai"
      ]
    },
    {
      "id": "bainbridge-1983",
      "citeAs": "https://thesuperskills.com/research/evidence#bainbridge-1983",
      "section": "collaboration",
      "authors": "Bainbridge, L.",
      "year": 1983,
      "title": "Ironies of Automation",
      "publication": "Automatica, 19(6)",
      "url": "https://doi.org/10.1016/0005-1098(83)90046-8",
      "grade": "peer-reviewed",
      "method": "Theoretical analysis of automated process control systems and the human roles left within them.",
      "finding": "Automating the routine parts of a task leaves the human with the hardest residue, monitoring and exception handling, while removing the routine practice that built the competence to do it. Automation makes the remaining human role harder, not easier.",
      "supports": "That monitoring is a demanding task rather than a light one, and that the design of automation determines whether the human retains the capability to supervise it.",
      "doesNotSupport": "Anything specific to AI. It is a process-control argument from 1983, and its application to generative systems is by analogy rather than by measurement.",
      "terms": [
        "ironies of automation",
        "monitoring",
        "deskilling",
        "oversight"
      ],
      "relatedPages": [
        "/research/human-in-the-loop-is-not-a-safeguard",
        "/research/what-is-deskilling",
        "/research/delegation-boundary-map",
        "/research/how-do-you-design-a-stop-button-people-will-use",
        "/research/how-do-humans-and-agents-divide-work",
        "/research/what-is-the-shared-prompt-review",
        "/research/what-is-automating-versus-informating"
      ]
    },
    {
      "id": "skitka-1999",
      "citeAs": "https://thesuperskills.com/research/evidence#skitka-1999",
      "section": "judgement",
      "authors": "Skitka, L. J., Mosier, K. L. and Burdick, M.",
      "year": 1999,
      "title": "Does automation bias decision-making?",
      "publication": "International Journal of Human-Computer Studies, 51(5)",
      "url": "https://doi.org/10.1006/ijhc.1999.0252",
      "grade": "peer-reviewed",
      "method": "Controlled experiments with a simulated flight task, comparing automated and non-automated decision aids.",
      "finding": "Automated aids produced two distinct error types: errors of omission, missing events the automation failed to flag, and errors of commission, following automated advice that was wrong.",
      "supports": "That automation bias has a measurable structure, and that the two error types require different countermeasures.",
      "doesNotSupport": "That the effect sizes transfer to generative AI or to non-simulated professional settings.",
      "terms": [
        "automation bias",
        "omission",
        "commission",
        "decision aids"
      ],
      "relatedPages": [
        "/research/what-is-automation-bias",
        "/research/human-ai-decision-making",
        "/research/ai-and-human-judgement"
      ]
    },
    {
      "id": "dzindolet-2003",
      "citeAs": "https://thesuperskills.com/research/evidence#dzindolet-2003",
      "section": "judgement",
      "authors": "Dzindolet, M. T., Peterson, S. A., Pomranky, R. A., Pierce, L. G. and Beck, H. P.",
      "year": 2003,
      "title": "The role of trust in automation reliance",
      "publication": "International Journal of Human-Computer Studies, 58(6)",
      "url": "https://doi.org/10.1016/S1071-5819(03)00038-7",
      "grade": "peer-reviewed",
      "method": "Experiments manipulating what participants were told about an automated aid's reliability and failure modes.",
      "finding": "Explaining why an automated aid might err INCREASED reliance on it, restoring trust even where that trust was unwarranted.",
      "supports": "That awareness training is a weak control, and can move reliance in the opposite direction to the one intended.",
      "doesNotSupport": "That explanation is always counterproductive. The effect is about restoring trust after observed error, not about all forms of transparency.",
      "terms": [
        "trust in automation",
        "explanation",
        "reliance",
        "automation bias"
      ],
      "relatedPages": [
        "/research/what-is-automation-bias",
        "/research/why-does-ai-sound-so-confident",
        "/research/what-is-meaningful-human-oversight",
        "/research/ai-and-human-judgement"
      ]
    },
    {
      "id": "parasuraman-riley-1997",
      "citeAs": "https://thesuperskills.com/research/evidence#parasuraman-riley-1997",
      "section": "judgement",
      "authors": "Parasuraman, R. and Riley, V.",
      "year": 1997,
      "title": "Humans and Automation: Use, Misuse, Disuse, Abuse",
      "publication": "Human Factors, 39(2)",
      "url": "https://journals.sagepub.com/doi/10.1518/001872097778543886",
      "grade": "peer-reviewed",
      "method": "Review and framework paper synthesising the human-factors literature on how people interact with automated systems.",
      "finding": "Establishes four distinct failure modes: use, misuse through over-reliance, disuse through under-reliance, and abuse through automating without regard for the human consequences.",
      "supports": "That over-reliance and under-reliance are separate problems requiring separate design responses, and that the failure can sit with the deploying organisation rather than the operator.",
      "doesNotSupport": "Quantified effect sizes. It is a framework paper rather than an experiment.",
      "terms": [
        "misuse",
        "disuse",
        "abuse",
        "over-reliance",
        "automation"
      ],
      "relatedPages": [
        "/research/what-is-automation-bias",
        "/research/human-ai-decision-making",
        "/research/design-versus-drift",
        "/research/ai-and-human-judgement",
        "/research/how-do-you-design-a-stop-button-people-will-use"
      ]
    },
    {
      "id": "acemoglu-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#acemoglu-2024",
      "section": "work",
      "authors": "Acemoglu, D.",
      "year": 2024,
      "title": "The Simple Macroeconomics of AI",
      "publication": "NBER Working Paper 32487; published in Economic Policy, 40(121), 2025",
      "url": "https://www.nber.org/papers/w32487",
      "grade": "peer-reviewed",
      "method": "Task-based macroeconomic model applying Hulten's theorem to existing estimates of AI task exposure and task-level cost savings.",
      "finding": "Estimates total factor productivity gains of no more than 0.66 percent over ten years, revised to under 0.53 percent once the difficulty of hard-to-learn tasks is accounted for. Argues AI is likely to widen the gap between capital and labour income rather than reduce labour income inequality.",
      "supports": "That plausible macroeconomic gains are an order of magnitude smaller than the headline value estimates in circulation.",
      "doesNotSupport": "That AI is unimportant. It models productivity through task-level cost savings, and would not capture effects running through new products, new tasks or capability change.",
      "terms": [
        "productivity",
        "macroeconomics",
        "total factor productivity",
        "inequality"
      ],
      "relatedPages": [
        "/research/how-should-leaders-respond-to-ai",
        "/research/ai-workforce-strategy",
        "/research/will-ai-replace-my-job",
        "/research/ai-and-human-judgement",
        "/research/does-ai-actually-make-people-more-productive"
      ]
    },
    {
      "id": "autor-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#autor-2024",
      "section": "work",
      "authors": "Autor, D.",
      "year": 2024,
      "title": "Applying AI to Rebuild Middle Class Jobs",
      "publication": "NBER Working Paper 32140",
      "url": "https://www.nber.org/papers/w32140",
      "grade": "working-paper",
      "method": "Argument and synthesis rather than empirical test, drawing on the author's prior work on task structure and labour demand.",
      "finding": "Argues that AI's distinctive opportunity is to extend the reach of expertise, letting a wider set of workers with complementary knowledge perform higher-stakes decision tasks currently reserved to elite experts.",
      "supports": "That there is a serious, well-argued case for AI as an expertise-widening technology rather than an expertise-replacing one.",
      "doesNotSupport": "That this will happen. The author is explicit that the thesis is an argument about what is possible rather than a forecast, and no evidence yet shows it occurring at scale.",
      "terms": [
        "expertise",
        "middle-skill work",
        "complementarity",
        "task structure"
      ],
      "relatedPages": [
        "/research/what-is-the-judgement-premium",
        "/research/will-ai-replace-my-job",
        "/research/staying-valuable-in-the-age-of-ai",
        "/research/what-happens-to-work-that-moves-information",
        "/research/what-becomes-more-valuable-as-ai-gets-cheaper"
      ]
    },
    {
      "id": "liang-2023",
      "citeAs": "https://thesuperskills.com/research/evidence#liang-2023",
      "section": "learning",
      "authors": "Liang, W., Yuksekgonul, M., Mao, Y., Wu, E. and Zou, J.",
      "year": 2023,
      "title": "GPT detectors are biased against non-native English writers",
      "publication": "Patterns, 4(7)",
      "url": "https://www.sciencedirect.com/science/article/pii/S2666389923001307",
      "grade": "peer-reviewed",
      "method": "Seven widely used GPT detectors evaluated against TOEFL essays by non-native English speakers and essays by US eighth-grade students.",
      "finding": "Detectors misclassified more than half of the non-native essays as AI-generated, an average false positive rate of 61.22 percent, while classifying US eighth-grade essays with near-perfect accuracy. The proposed mechanism is that detectors rely on perplexity, and second-language writing is more predictable.",
      "supports": "That AI detection carries a severe and systematic bias against second-language writers, and that the bias is structural rather than a tuning problem.",
      "doesNotSupport": "That every detector now on the market performs identically. The study tested tools available at the time, and vendors dispute the generalisation.",
      "terms": [
        "AI detection",
        "false positives",
        "academic integrity",
        "fairness"
      ],
      "relatedPages": [
        "/research/does-ai-detection-work",
        "/research/how-to-assess-students-when-ai-can-do-the-assignment"
      ]
    },
    {
      "id": "vonstumm-2011",
      "citeAs": "https://thesuperskills.com/research/evidence#vonstumm-2011",
      "section": "capability",
      "authors": "von Stumm, S., Hell, B. and Chamorro-Premuzic, T.",
      "year": 2011,
      "title": "The Hungry Mind: Intellectual Curiosity Is the Third Pillar of Academic Performance",
      "publication": "Perspectives on Psychological Science, 6(6)",
      "url": "https://journals.sagepub.com/doi/abs/10.1177/1745691611421204",
      "grade": "peer-reviewed",
      "method": "Path-model synthesis of prior meta-analytic correlation matrices. Component samples range from 608 to 28,471.",
      "finding": "Intellectual curiosity predicts academic performance independently of intelligence and effort, which the authors describe as a third pillar.",
      "supports": "That curiosity carries predictive weight that conscientiousness and ability do not account for.",
      "doesNotSupport": "Causation. It is a correlational synthesis. The widely quoted figure of roughly 50,000 students comes from the accompanying press release rather than from the paper itself.",
      "terms": [
        "curiosity"
      ],
      "relatedPages": [
        "/research/superskill-curiosity"
      ]
    },
    {
      "id": "harrison-2011",
      "citeAs": "https://thesuperskills.com/research/evidence#harrison-2011",
      "section": "capability",
      "authors": "Harrison, S. H., Sluss, D. M. and Ashforth, B. E.",
      "year": 2011,
      "title": "Curiosity adapted the cat: The role of trait curiosity in newcomer adaptation",
      "publication": "Journal of Applied Psychology, 96(1)",
      "url": "https://pubmed.ncbi.nlm.nih.gov/21244132/",
      "grade": "peer-reviewed",
      "method": "Longitudinal field study of 123 newcomers across 12 call-centre organisations.",
      "finding": "Specific curiosity predicted information-seeking from colleagues, which in turn was associated with more creative handling of customer problems.",
      "supports": "That curiosity operates through a behavioural mechanism, asking, rather than as a disposition on its own.",
      "doesNotSupport": "That curiosity can be trained into people, or that the effect holds outside newcomer adaptation.",
      "terms": [
        "curiosity"
      ],
      "relatedPages": [
        "/research/superskill-curiosity"
      ]
    },
    {
      "id": "swan-carmelli-1996",
      "citeAs": "https://thesuperskills.com/research/evidence#swan-carmelli-1996",
      "section": "capability",
      "authors": "Swan, G. E. and Carmelli, D.",
      "year": 1996,
      "title": "Curiosity and mortality in aging adults: A 5-year follow-up of the Western Collaborative Group Study",
      "publication": "Psychology and Aging, 11(3)",
      "url": "https://pubmed.ncbi.nlm.nih.gov/8893314/",
      "grade": "peer-reviewed",
      "method": "Prospective cohort. 1,118 men, mean age 70.6 at baseline, with an ancillary sample of 1,035 women.",
      "finding": "Higher curiosity at baseline was associated with survival at five-year follow-up, and state curiosity remained significant after adjustment for other risk factors.",
      "supports": "That the association survives controlling for medical risk factors in one large cohort.",
      "doesNotSupport": "That curiosity extends life. Residual confounding, in particular underlying health driving both curiosity and survival, cannot be excluded.",
      "terms": [
        "curiosity"
      ],
      "relatedPages": [
        "/research/superskill-curiosity"
      ]
    },
    {
      "id": "gino-2018",
      "citeAs": "https://thesuperskills.com/research/evidence#gino-2018",
      "section": "capability",
      "authors": "Gino, F.",
      "year": 2018,
      "title": "The Business Case for Curiosity",
      "publication": "Harvard Business Review, September to October 2018",
      "url": "https://hbr.org/2018/09/the-business-case-for-curiosity",
      "grade": "compiled-review",
      "method": "Survey of more than 3,000 employees across a range of firms, reported in a practitioner magazine.",
      "finding": "Around 92 per cent said curious people bring new ideas to their teams, while about 24 per cent reported feeling curious in their jobs regularly.",
      "supports": "That the stated value of curiosity and the felt experience of it diverge sharply inside organisations.",
      "doesNotSupport": "Any causal link between curiosity and a business outcome. Self-report, not peer-reviewed, and the sample is not nationally representative.",
      "terms": [
        "curiosity"
      ],
      "relatedPages": [
        "/research/superskill-curiosity"
      ]
    },
    {
      "id": "sadri-2011",
      "citeAs": "https://thesuperskills.com/research/evidence#sadri-2011",
      "section": "capability",
      "authors": "Sadri, G., Weber, T. J. and Gentry, W. A.",
      "year": 2011,
      "title": "Empathic emotion and leadership performance: An empirical analysis across 38 countries",
      "publication": "The Leadership Quarterly, 22(5)",
      "url": "https://www.sciencedirect.com/science/article/abs/pii/S1048984311001093",
      "grade": "peer-reviewed",
      "method": "360-degree ratings of 6,731 mid to upper-level managers across 38 countries. Subordinates rated empathy, superiors rated performance.",
      "finding": "Managers rated as more empathic by subordinates received higher performance ratings from their own superiors, with the effect moderated by national power distance.",
      "supports": "That the association holds at scale and across many national contexts.",
      "doesNotSupport": "That empathy training improves performance. The design is cross-sectional and correlational.",
      "terms": [
        "empathy"
      ],
      "relatedPages": [
        "/research/superskill-empathy"
      ]
    },
    {
      "id": "howick-2018",
      "citeAs": "https://thesuperskills.com/research/evidence#howick-2018",
      "section": "capability",
      "authors": "Howick, J., Moscrop, A., Mebius, A. et al.",
      "year": 2018,
      "title": "Effects of empathic and positive communication in healthcare consultations: a systematic review and meta-analysis",
      "publication": "Journal of the Royal Society of Medicine, 111(7)",
      "url": "https://journals.sagepub.com/doi/10.1177/0141076818769477",
      "grade": "peer-reviewed",
      "method": "Systematic review and meta-analysis of 28 randomised trials, 6,017 patients in total. Seven of the 28 tested empathic communication specifically; the remainder tested positive-expectation messaging.",
      "finding": "The seven empathy-specific trials showed a small improvement in pain, anxiety and satisfaction, SMD -0.18, 95 per cent CI -0.32 to -0.03.",
      "supports": "That empathic communication has a measurable effect on patient-reported outcomes.",
      "doesNotSupport": "A large clinical benefit. The authors describe the effect as small, and only a quarter of the pooled trials tested empathy rather than positive framing.",
      "terms": [
        "empathy"
      ],
      "relatedPages": [
        "/research/superskill-empathy"
      ]
    },
    {
      "id": "konrath-2011",
      "citeAs": "https://thesuperskills.com/research/evidence#konrath-2011",
      "section": "capability",
      "authors": "Konrath, S. H., O'Brien, E. H. and Hsing, C.",
      "year": 2011,
      "title": "Changes in Dispositional Empathy in American College Students Over Time: A Meta-Analysis",
      "publication": "Personality and Social Psychology Review, 15(2)",
      "url": "https://journals.sagepub.com/doi/10.1177/1088868310377395",
      "grade": "peer-reviewed",
      "method": "Cross-temporal meta-analysis of 72 samples of American college students, total N 13,737, from 1979 to 2009.",
      "finding": "Empathic Concern fell by 48 per cent and Perspective Taking by 34 per cent across the period, with most of the decline after 2000.",
      "supports": "That self-reported dispositional empathy declined measurably in this population over three decades.",
      "doesNotSupport": "A cause. The authors speculate about individualism and media but test no mechanism. It is also American college students only.",
      "terms": [
        "empathy"
      ],
      "relatedPages": [
        "/research/superskill-empathy"
      ]
    },
    {
      "id": "pulakos-2000",
      "citeAs": "https://thesuperskills.com/research/evidence#pulakos-2000",
      "section": "capability",
      "authors": "Pulakos, E. D., Arad, S., Donovan, M. A. and Plamondon, K. E.",
      "year": 2000,
      "title": "Adaptability in the Workplace: Development of a Taxonomy of Adaptive Performance",
      "publication": "Journal of Applied Psychology, 85(4)",
      "url": "https://psycnet.apa.org/doi/10.1037/0021-9010.85.4.612",
      "grade": "peer-reviewed",
      "method": "Content analysis of over 1,000 critical incidents drawn from 21 jobs, followed by scale validation.",
      "finding": "Eight dimensions of adaptive performance, including handling emergencies, managing stress, solving problems creatively and dealing with uncertain situations.",
      "supports": "That adaptability is decomposable into observable behaviours rather than being a single trait.",
      "doesNotSupport": "That employers reward these dimensions, or that they transfer across every occupation. The taxonomy was derived, not tested against outcomes.",
      "terms": [
        "change readiness",
        "adaptive performance"
      ],
      "relatedPages": [
        "/research/superskill-change-readiness"
      ]
    },
    {
      "id": "huang-2014",
      "citeAs": "https://thesuperskills.com/research/evidence#huang-2014",
      "section": "capability",
      "authors": "Huang, J. L., Ryan, A. M., Zabel, K. L. and Palmer, A.",
      "year": 2014,
      "title": "Personality and Adaptive Performance at Work: A Meta-Analytic Investigation",
      "publication": "Journal of Applied Psychology, 99(1)",
      "url": "https://psycnet.apa.org/doi/10.1037/a0034285",
      "grade": "peer-reviewed",
      "method": "Meta-analysis of 71 independent samples, total N 7,535.",
      "finding": "Emotional stability and ambition predict adaptive performance, and the pattern differs from predictors of routine task performance.",
      "supports": "That individual differences carry predictive weight for adaptive performance specifically.",
      "doesNotSupport": "That adaptive performance is a distinct construct. This paper assumes the construct from earlier taxonomy work rather than establishing it.",
      "terms": [
        "change readiness",
        "adaptive performance"
      ],
      "relatedPages": [
        "/research/superskill-change-readiness"
      ]
    },
    {
      "id": "edmondson-1999",
      "citeAs": "https://thesuperskills.com/research/evidence#edmondson-1999",
      "section": "capability",
      "authors": "Edmondson, A.",
      "year": 1999,
      "title": "Psychological Safety and Learning Behavior in Work Teams",
      "publication": "Administrative Science Quarterly, 44(2)",
      "url": "https://journals.sagepub.com/doi/10.2307/2666999",
      "grade": "peer-reviewed",
      "method": "Multi-method field study of 51 work teams in a single manufacturing company.",
      "finding": "Team psychological safety predicted learning behaviour, which in turn mediated the relationship with team performance.",
      "supports": "That the route from safety to performance runs through learning behaviour rather than directly.",
      "doesNotSupport": "Causation, or generalisation beyond one manufacturing firm. It is a correlational field study.",
      "terms": [
        "change readiness",
        "psychological safety"
      ],
      "relatedPages": [
        "/research/superskill-change-readiness"
      ]
    },
    {
      "id": "defreitas-2023",
      "citeAs": "https://thesuperskills.com/research/evidence#defreitas-2023",
      "section": "capability",
      "authors": "De Freitas, J., Uguralp, A. K., Oguz-Uguralp, Z., Paul, L. A., Tenenbaum, J. and Ullman, T. D.",
      "year": 2023,
      "title": "Self-orienting in human and machine learning",
      "publication": "Nature Human Behaviour, 7",
      "url": "https://www.nature.com/articles/s41562-023-01696-5",
      "grade": "peer-reviewed",
      "method": "Behavioural experiments with 124 human players across custom games, benchmarked against deep reinforcement learning agents.",
      "finding": "Humans were near optimal at working out their own position and capabilities after conditions were altered. The reinforcement learning baselines were far from optimal at the same task.",
      "supports": "That rapid self-orientation after an unexpected change is currently a human advantage over the algorithms tested.",
      "doesNotSupport": "General workplace adaptability. These are simple custom games, the sample is modest, and the comparison is against specific algorithms rather than all AI approaches.",
      "terms": [
        "change readiness"
      ],
      "relatedPages": [
        "/research/superskill-change-readiness"
      ]
    },
    {
      "id": "schlaegel-2021",
      "citeAs": "https://thesuperskills.com/research/evidence#schlaegel-2021",
      "section": "capability",
      "authors": "Schlaegel, C., Richter, N. F. and Taras, V.",
      "year": 2021,
      "title": "Cultural intelligence and work-related outcomes: A meta-analytic examination of joint effects and incremental predictive validity",
      "publication": "Journal of World Business, 56(4)",
      "url": "https://www.sciencedirect.com/science/article/abs/pii/S1090951621000225",
      "grade": "peer-reviewed",
      "method": "Meta-analysis of 70 studies providing 80 independent samples, total N 18,359.",
      "finding": "Cultural intelligence is moderately associated with work-related outcomes, with a reliability-corrected average effect of about .39.",
      "supports": "That the association is consistent across a large body of studies.",
      "doesNotSupport": "Causation, and it does not establish any single dimension as the strongest predictor. The paper is about joint effects across all four dimensions.",
      "terms": [
        "global adaptability",
        "cultural intelligence"
      ],
      "relatedPages": [
        "/research/superskill-global-adaptability"
      ]
    },
    {
      "id": "maddux-galinsky-2009",
      "citeAs": "https://thesuperskills.com/research/evidence#maddux-galinsky-2009",
      "section": "capability",
      "authors": "Maddux, W. W. and Galinsky, A. D.",
      "year": 2009,
      "title": "Cultural Borders and Mental Barriers: The Relationship Between Living Abroad and Creativity",
      "publication": "Journal of Personality and Social Psychology, 96(5)",
      "url": "https://psycnet.apa.org/doi/10.1037/a0014861",
      "grade": "peer-reviewed",
      "method": "Five studies with MBA and undergraduate participants, combining correlational designs with causal priming experiments.",
      "finding": "Time spent living abroad predicted success on creative-insight tasks and creative negotiation outcomes. Time spent travelling abroad did not.",
      "supports": "That adaptation to living in another culture, rather than exposure to it, is what relates to creativity.",
      "doesNotSupport": "A specific effect size for the general population. The samples are business and undergraduate students.",
      "terms": [
        "global adaptability"
      ],
      "relatedPages": [
        "/research/superskill-global-adaptability"
      ]
    },
    {
      "id": "stillman-2018",
      "citeAs": "https://thesuperskills.com/research/evidence#stillman-2018",
      "section": "capability",
      "authors": "Stillman, P. E., Fujita, K., Sheldon, O. and Trope, Y.",
      "year": 2018,
      "title": "From 'Me' to 'We': The Role of Construal Level in Promoting Maximized Joint Outcomes",
      "publication": "Organizational Behavior and Human Decision Processes, 147",
      "url": "https://www.sciencedirect.com/science/article/abs/pii/S0749597817303989",
      "grade": "peer-reviewed",
      "method": "Four laboratory and online experiments, pooled N approximately 691.",
      "finding": "Prompting a higher level of construal led participants to choose options that maximised joint outcomes, including where doing so reduced their own payoff.",
      "supports": "That the level at which a problem is framed changes whether people optimise for themselves or for the whole.",
      "doesNotSupport": "Field behaviour. These are economic games with student and online samples.",
      "terms": [
        "big picture thinking"
      ],
      "relatedPages": [
        "/research/superskill-big-picture-thinking"
      ]
    },
    {
      "id": "mehta-2014",
      "citeAs": "https://thesuperskills.com/research/evidence#mehta-2014",
      "section": "capability",
      "authors": "Mehta, R., Zhu, R. and Meyers-Levy, J.",
      "year": 2014,
      "title": "When Does a Higher Construal Level Increase or Decrease Indulgence? Resolving the Myopia versus Hyperopia Puzzle",
      "publication": "Journal of Consumer Research, 41(2)",
      "url": "https://academic.oup.com/jcr/article/41/2/475/2907518",
      "grade": "peer-reviewed",
      "method": "Multi-study laboratory experiments.",
      "finding": "Where the self is focal, a higher construal level increases indulgence rather than reducing it, reversing the effect the earlier literature predicted.",
      "supports": "That the benefit of stepping back is conditional, and the condition is identifiable.",
      "doesNotSupport": "That distant-future thinking is generally counterproductive. The effect is moderated, not reversed outright.",
      "terms": [
        "big picture thinking"
      ],
      "relatedPages": [
        "/research/superskill-big-picture-thinking"
      ]
    },
    {
      "id": "orlitzky-2003",
      "citeAs": "https://thesuperskills.com/research/evidence#orlitzky-2003",
      "section": "capability",
      "authors": "Orlitzky, M., Schmidt, F. L. and Rynes, S. L.",
      "year": 2003,
      "title": "Corporate Social and Financial Performance: A Meta-Analysis",
      "publication": "Organization Studies, 24(3)",
      "url": "https://journals.sagepub.com/doi/10.1177/0170840603024003910",
      "grade": "peer-reviewed",
      "method": "Meta-analysis of 52 studies, 33,878 observations.",
      "finding": "A positive association between corporate social performance and financial performance, which the authors describe as bidirectional.",
      "supports": "That principled conduct and financial results are not in general tension.",
      "doesNotSupport": "Causation in either direction, and the relationship is not uniform across how social performance is operationalised.",
      "terms": [
        "principled innovation"
      ],
      "relatedPages": [
        "/research/superskill-principled-innovation"
      ]
    },
    {
      "id": "bcg-diversity-2018",
      "citeAs": "https://thesuperskills.com/research/evidence#bcg-diversity-2018",
      "section": "capability",
      "authors": "Lorenzo, R., Voigt, N., Tsusaka, M., Krentz, M. and Abouzahr, K.",
      "year": 2018,
      "title": "How Diverse Leadership Teams Boost Innovation",
      "publication": "Boston Consulting Group",
      "url": "https://www.bcg.com/publications/2018/how-diverse-leadership-teams-boost-innovation",
      "grade": "institutional-survey",
      "method": "Survey of more than 1,700 companies across eight countries.",
      "finding": "Companies with above-average management diversity reported innovation revenue 19 percentage points higher than below-average companies, 45 per cent of total revenue against 26 per cent.",
      "supports": "That reported diversity and reported innovation revenue move together at scale.",
      "doesNotSupport": "Causation. Nor is this audited financial data. Note that 19 percentage points is not the same as 19 per cent higher, a distinction frequently lost in citation.",
      "terms": [
        "principled innovation"
      ],
      "relatedPages": [
        "/research/superskill-principled-innovation"
      ]
    },
    {
      "id": "simons-2002",
      "citeAs": "https://thesuperskills.com/research/evidence#simons-2002",
      "section": "capability",
      "authors": "Simons, T.",
      "year": 2002,
      "title": "The High Cost of Lost Trust",
      "publication": "Harvard Business Review, September 2002",
      "url": "https://hbr.org/2002/09/the-high-cost-of-lost-trust",
      "grade": "compiled-review",
      "method": "Survey of more than 6,500 employees at 76 Holiday Inn hotels in the United States and Canada, matched to hotel financial records.",
      "finding": "A one-eighth point improvement in managers' behavioural integrity rating was associated with a 2.5 per cent increase in profitability, roughly 250,000 dollars a year for an average hotel. No other measured aspect of manager behaviour had as large an effect on profits.",
      "supports": "That whether managers are seen to mean what they say tracks measurable financial outcomes.",
      "doesNotSupport": "Causation. It is a correlational study within one hotel chain, published in a practitioner magazine rather than a peer-reviewed journal.",
      "terms": [
        "principled innovation"
      ],
      "relatedPages": [
        "/research/superskill-principled-innovation"
      ]
    },
    {
      "id": "mckinsey-shorttermism-2017",
      "citeAs": "https://thesuperskills.com/research/evidence#mckinsey-shorttermism-2017",
      "section": "capability",
      "authors": "Barton, D., Manyika, J., Koller, T., Palter, R., Godsall, J. and Zoffer, J.",
      "year": 2017,
      "title": "Measuring the Economic Impact of Short-Termism",
      "publication": "McKinsey Global Institute",
      "url": "https://www.mckinsey.com/featured-insights/long-term-capitalism/where-companies-with-a-long-term-view-outperform-their-peers",
      "grade": "institutional-modelling",
      "method": "Corporate Horizon Index applied to 615 large and mid-cap United States public companies, 2001 to 2014.",
      "finding": "Revenue of long-term firms grew cumulatively 47 per cent more than other firms, earnings 36 per cent more, and their share prices recovered faster after the financial crisis.",
      "supports": "That a measurable long-term orientation tracks with stronger cumulative growth in this sample.",
      "doesNotSupport": "Causation, or application outside large United States listed companies. The index is a constructed measure, not an observed policy.",
      "terms": [
        "principled innovation"
      ],
      "relatedPages": [
        "/research/superskill-principled-innovation"
      ]
    },
    {
      "id": "epa-vw-2015",
      "citeAs": "https://thesuperskills.com/research/evidence#epa-vw-2015",
      "section": "capability",
      "authors": "United States Environmental Protection Agency",
      "year": 2015,
      "title": "Notice of Violation, Volkswagen Group",
      "publication": "US EPA enforcement record",
      "url": "https://19january2021snapshot.epa.gov/enforcement/learn-about-volkswagen-violations_.html",
      "grade": "institutional-survey",
      "method": "Regulatory enforcement notice.",
      "finding": "Affected 2.0-litre vehicles emitted nitrogen oxides at up to 40 times the standard in normal driving while appearing compliant in laboratory testing.",
      "supports": "That the defeat device produced a measured gap between test and road conditions of that magnitude.",
      "doesNotSupport": "That the same multiple applies across the range. The separate 3.0-litre notice cited up to nine times.",
      "terms": [
        "principled innovation"
      ],
      "relatedPages": [
        "/research/superskill-principled-innovation"
      ]
    },
    {
      "id": "faa-parc-2013",
      "citeAs": "https://thesuperskills.com/research/evidence#faa-parc-2013",
      "section": "learning",
      "authors": "PARC/CAST Flight Deck Automation Working Group",
      "year": 2013,
      "title": "Operational Use of Flight Path Management Systems: Final Report",
      "publication": "Federal Aviation Administration",
      "url": "https://www.faa.gov/sites/faa.gov/files/aircraft/air_cert/design_approvals/human_factors/OUFPMS_Report.pdf",
      "grade": "institutional-modelling",
      "method": "Working-group synthesis of accident and incident data, operator surveys and prior research. 28 findings, 18 recommendations.",
      "finding": "Identified vulnerabilities in manual handling after transition from automated control, and in the definition, development and retention of those skills. Also found that pilots sometimes rely too much on automated systems and may be reluctant to intervene.",
      "supports": "That a regulator examined automation dependence in a whole industry and named skill retention as a finding.",
      "doesNotSupport": "A measured rate of skill decay. It is a synthesis of findings, not a controlled study.",
      "terms": [
        "automation complacency",
        "deskilling"
      ],
      "relatedPages": [
        "/research/what-professions-can-learn-from-aviation"
      ]
    },
    {
      "id": "bea-af447-2012",
      "citeAs": "https://thesuperskills.com/research/evidence#bea-af447-2012",
      "section": "learning",
      "authors": "Bureau d'Enquetes et d'Analyses",
      "year": 2012,
      "title": "Final Report on the accident on 1 June 2009 to the Airbus A330-203, flight AF 447",
      "publication": "BEA, France",
      "url": "https://bea.aero/docspa/2009/f-cp090601.en/pdf/f-cp090601.en.pdf",
      "grade": "institutional-survey",
      "method": "Statutory accident investigation. Flight recorder analysis and training record review. 25 new safety recommendations.",
      "finding": "The investigation cited the lack of practical training in high-altitude manual handling and in the procedure for speed anomalies among the contributing factors.",
      "supports": "That a formal investigation attributed part of an accident to manual handling practice that had not been maintained.",
      "doesNotSupport": "A general rate of skill loss across the pilot population. It is one accident.",
      "terms": [
        "deskilling"
      ],
      "relatedPages": [
        "/research/what-professions-can-learn-from-aviation"
      ]
    },
    {
      "id": "cfr-121-441",
      "citeAs": "https://thesuperskills.com/research/evidence#cfr-121-441",
      "section": "institutional",
      "authors": "United States Federal Aviation Regulations",
      "year": 1974,
      "title": "14 CFR 121.441, Proficiency checks",
      "publication": "Electronic Code of Federal Regulations",
      "url": "https://www.ecfr.gov/current/title-14/chapter-I/subchapter-G/part-121/subpart-O/section-121.441",
      "grade": "institutional-survey",
      "method": "Binding regulation.",
      "finding": "A pilot in command must pass a proficiency check every 12 calendar months, and within every 6 calendar months either a proficiency check or an approved simulator course.",
      "supports": "That one profession has made recurrent tested practice a legal condition of continuing to work.",
      "doesNotSupport": "That the intervals are calibrated to measured decay curves. The regulation sets a minimum, not an evidence-based optimum.",
      "terms": [],
      "relatedPages": [
        "/research/what-professions-can-learn-from-aviation"
      ]
    },
    {
      "id": "cfr-121-542",
      "citeAs": "https://thesuperskills.com/research/evidence#cfr-121-542",
      "section": "institutional",
      "authors": "United States Federal Aviation Regulations",
      "year": 1981,
      "title": "14 CFR 121.542, Flight crewmember duties, the sterile cockpit rule",
      "publication": "Electronic Code of Federal Regulations",
      "url": "https://www.ecfr.gov/current/title-14/chapter-I/subchapter-G/part-121/subpart-T/section-121.542",
      "grade": "institutional-survey",
      "method": "Binding regulation, Docket 20661, 46 FR 5502, 19 January 1981.",
      "finding": "No crew member may perform any duty during a critical phase of flight other than those required for safe operation. Critical phases include taxi, take-off, landing and all operations below 10,000 feet except cruise.",
      "supports": "That protected attention can be written into law as a condition rather than left to individual discipline.",
      "doesNotSupport": "Anything about automation or skill decay directly. It is a rule about distraction.",
      "terms": [],
      "relatedPages": [
        "/research/what-professions-can-learn-from-aviation"
      ]
    },
    {
      "id": "helmreich-1999",
      "citeAs": "https://thesuperskills.com/research/evidence#helmreich-1999",
      "section": "collaboration",
      "authors": "Helmreich, R. L., Merritt, A. C. and Wilhelm, J. A.",
      "year": 1999,
      "title": "The Evolution of Crew Resource Management Training in Commercial Aviation",
      "publication": "International Journal of Aviation Psychology, 9(1)",
      "url": "https://www.faa.gov/sites/faa.gov/files/2022-11/crmhistory.pdf",
      "grade": "peer-reviewed",
      "method": "Historical and evaluative review of CRM programmes from 1979 onwards.",
      "finding": "CRM began with a 1979 NASA workshop prompted by an NTSB finding that a captain had failed to accept input from junior crew. Line audits show CRM produces the intended behavioural change, though measured attitudes decay over time even with recurrent training.",
      "supports": "That an industry built a training response to a named human-factors failure and measured behaviour rather than opinion.",
      "doesNotSupport": "That CRM reduces accidents. The authors state directly that accident rates are too rare to serve as a validation criterion.",
      "terms": [],
      "relatedPages": [
        "/research/what-professions-can-learn-from-aviation",
        "/research/ai-and-human-disagreement"
      ]
    },
    {
      "id": "kleinberg-2017",
      "citeAs": "https://thesuperskills.com/research/evidence#kleinberg-2017",
      "section": "judgement",
      "authors": "Kleinberg, J., Mullainathan, S. and Raghavan, M.",
      "year": 2017,
      "title": "Inherent Trade-Offs in the Fair Determination of Risk Scores",
      "publication": "8th Innovations in Theoretical Computer Science Conference (ITCS 2017), LIPIcs vol. 67, article 43. Read at source 10 September 2026",
      "url": "https://drops.dagstuhl.de/entities/document/10.4230/LIPIcs.ITCS.2017.43",
      "grade": "peer-reviewed",
      "method": "Formal proof. The authors state three fairness conditions that recur in public argument about risk scores and ask whether any method can satisfy all three at once.",
      "finding": "Except in highly constrained special cases, no method satisfies the three conditions simultaneously. Satisfying them even approximately requires the data to sit in an approximate version of one of those special cases.",
      "supports": "That 'unbiased' is not one target. Several reasonable definitions of fairness are mathematically incompatible, so a system can be made to pass one and will then fail another, and the choice between them is a value judgement rather than a technical one.",
      "doesNotSupport": "That fairness is unachievable or that auditing is pointless. It is a result about simultaneous satisfaction of particular formal criteria, not a claim that bias cannot be reduced, and it says nothing about how any deployed system behaves.",
      "terms": [],
      "relatedPages": [
        "/research/can-ai-be-unbiased"
      ]
    },
    {
      "id": "morris-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#morris-2024",
      "section": "institutional",
      "authors": "Morris, M. R., Sohl-Dickstein, J., Fiedel, N., Warkentin, T., Dafoe, A., Faust, A., Farabet, C. and Legg, S.",
      "year": 2024,
      "title": "Position: Levels of AGI for Operationalizing Progress on the Path to AGI",
      "publication": "Proceedings of the 41st International Conference on Machine Learning, PMLR 235. Google DeepMind",
      "url": "https://proceedings.mlr.press/v235/morris24b.html",
      "grade": "argued-perspective",
      "method": "Analysis of the existing published definitions of AGI, from which the authors distil six principles an ontology should satisfy, then propose a two-dimensional framework of performance depth against capability breadth.",
      "finding": "The authors set out six performance levels: No AI, Emerging (equal to or somewhat better than an unskilled human), Competent (50th percentile of skilled adults), Expert (90th percentile), Virtuoso (99th percentile) and Superhuman (outperforming all humans), crossed with narrow and general breadth. They argue that levels interact with deployment choices about autonomy and risk, so a capability level does not by itself say how a system should be used.",
      "supports": "That the published definitions of AGI were divergent enough that a team at the largest AI lab judged a new ontology necessary, and that AGI is better treated as a set of levels than a threshold.",
      "doesNotSupport": "Nothing empirical. This is a position paper proposing a framework, not a measurement, and the percentile bands are a proposal rather than a validated instrument. The framework has not been adopted as a standard.",
      "terms": [
        "AGI",
        "superintelligence"
      ],
      "relatedPages": [
        "/research/what-is-agi"
      ]
    },
    {
      "id": "grace-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#grace-2024",
      "section": "institutional",
      "authors": "Grace, K., Stewart, H., Sandkühler, J. F., Thomas, S., Weinstein-Raun, B. and Brauner, J.",
      "year": 2024,
      "title": "Thousands of AI Authors on the Future of AI",
      "publication": "AI Impacts, preprint, January 2024. Survey fielded October 2023. Read in full at source 9 September 2026",
      "url": "https://aiimpacts.org/wp-content/uploads/2023/04/Thousands_of_AI_authors_on_the_future_of_AI.pdf",
      "grade": "expert-survey",
      "method": "Survey of 2,778 researchers who had published in the prior year at NeurIPS, ICML, ICLR, AAAI, IJCAI or JMLR. 20,066 contacted, 1,607 bounced, 18,459 functioning addresses, response rate 15 per cent. Randomised question framings across respondents, deliberately, to measure framing effects. Timelines aggregated by fitting gamma distributions.",
      "finding": "On timing, the 2023 aggregate forecast gave High-Level Machine Intelligence a 50 per cent chance by 2047, thirteen years earlier than the 2060 given one year before, and a 10 per cent chance by 2027. On risk, the median answer depended on the wording put to the same population: 5 per cent for future AI advances causing human extinction or similarly permanent and severe disempowerment (n=1,321, mean 16.2), and 10 per cent for human INABILITY TO CONTROL advanced AI causing the same outcome (n=661, mean 19.4). Depending on how it was asked, between 41.2 and 51.4 per cent gave more than a 10 per cent chance. 38 per cent gave at least 10 per cent to extremely bad outcomes on the general question, down from 48 per cent in 2022.",
      "supports": "That the most-quoted numbers on AI extinction risk come from a large expert population, and that a change of wording moves the median from 5 to 10 per cent within it.",
      "doesNotSupport": "Any probability of anything. The authors say so themselves: their participants are experts in AI and not, to their knowledge, skilled forecasters; different respondents give very different answers; and they cite Karger et al. finding that framing moved lay estimates of existential risk by nearly six orders of magnitude. A 15 per cent response rate leaves participation bias possible, though the authors looked for it and did not find it. HLMI is defined as feasibility, not adoption.",
      "terms": [
        "AGI",
        "High-Level Machine Intelligence",
        "superintelligence"
      ],
      "relatedPages": [
        "/research/what-is-agi",
        "/research/is-ai-dangerous"
      ]
    },
    {
      "id": "iais-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#iais-2026",
      "section": "institutional",
      "authors": "Bengio, Y. and others (Expert Advisory Panel nominated by over 30 countries and international organisations)",
      "year": 2026,
      "title": "International AI Safety Report 2026",
      "publication": "Published 3 February 2026. Second edition. Over 100 contributing experts. arXiv:2602.21012",
      "url": "https://internationalaisafetyreport.org/publication/international-ai-safety-report-2026",
      "grade": "institutional-synthesis",
      "method": "Synthesis of the scientific literature on general-purpose AI capability, risk and risk management. Makes no policy recommendations by design.",
      "finding": "Capability is JAGGED: systems solve graduate-level mathematics and science problems while failing simpler tasks, are less reliable across many steps, still hallucinate, and remain limited on the physical world and on unfamiliar languages and cultural contexts. Agents complete software tasks with limited oversight but cannot yet do the long-horizon planning that automating a job requires, so the report concludes they complement rather than replace. On loss of control, expert views vary widely and current systems show at most early signs. The report names an EVIDENCE DILEMMA: the landscape changes fast and evidence about new risks emerges slowly, so acting early may entrench the wrong intervention and waiting may leave society exposed.",
      "supports": "The most institutionally backed statement available of what general-purpose AI can and cannot do, and that the disagreement about loss of control is between experts rather than between experts and the public.",
      "doesNotSupport": "Anything about what will happen. It is a synthesis of evidence, not a forecast, and it declines to recommend policy.",
      "terms": [
        "AGI",
        "general-purpose AI",
        "superintelligence"
      ],
      "relatedPages": [
        "/research/what-is-agi",
        "/research/is-ai-dangerous"
      ]
    },
    {
      "id": "molloy-parasuraman-1996",
      "citeAs": "https://thesuperskills.com/research/evidence#molloy-parasuraman-1996",
      "section": "judgement",
      "authors": "Molloy, R. and Parasuraman, R.",
      "year": 1996,
      "title": "Monitoring an Automated System for a Single Failure: Vigilance and Task Complexity Effects",
      "publication": "Human Factors, 38(2)",
      "url": "https://journals.sagepub.com/doi/10.1177/001872089606380211",
      "grade": "peer-reviewed",
      "method": "Laboratory flight-simulation experiments varying task complexity and time on task.",
      "finding": "Detection of a single automation failure degrades with time on task, and the effect is strongest where the automation has been consistently reliable.",
      "supports": "That monitoring performance falls predictably rather than randomly, and that reliability itself is part of the cause.",
      "doesNotSupport": "Field incidence. It is simulation, not accident data.",
      "terms": [
        "automation complacency"
      ],
      "relatedPages": [
        "/research/what-professions-can-learn-from-aviation",
        "/research/ai-and-human-judgement"
      ]
    },
    {
      "id": "arthur-1998",
      "citeAs": "https://thesuperskills.com/research/evidence#arthur-1998",
      "section": "learning",
      "authors": "Arthur, W., Bennett, W., Stanush, P. L. and McNelly, T. L.",
      "year": 1998,
      "title": "Factors that influence skill decay and retention: A quantitative review and analysis",
      "publication": "Human Performance, 11(1)",
      "url": "https://www.tandfonline.com/doi/abs/10.1207/s15327043hup1101_3",
      "grade": "peer-reviewed",
      "method": "Meta-analysis of 189 independent data points from 53 articles.",
      "finding": "Skill loss ran from d of -0.01 immediately after training to d of -1.4 after more than 365 days of non-use. Physical, natural and speed-based tasks decayed less than cognitive, artificial and accuracy-based tasks.",
      "supports": "That skill decay is measurable, that it is a function of the interval, and that cognitive skills go first.",
      "doesNotSupport": "How fast any particular professional skill decays, or how quickly it can be regained.",
      "terms": [
        "deskilling",
        "capability debt"
      ],
      "relatedPages": [
        "/research/can-you-regain-a-skill-you-have-lost"
      ]
    },
    {
      "id": "murre-dros-2015",
      "citeAs": "https://thesuperskills.com/research/evidence#murre-dros-2015",
      "section": "learning",
      "authors": "Murre, J. M. J. and Dros, J.",
      "year": 2015,
      "title": "Replication and Analysis of Ebbinghaus' Forgetting Curve",
      "publication": "PLoS ONE, 10(7)",
      "url": "https://journals.plos.org/plosone/article?id=10.1371/journal.pone.0120644",
      "grade": "peer-reviewed",
      "method": "Single-subject relearning experiment replicating Ebbinghaus across intervals from 20 minutes to 31 days, 10 replications per interval.",
      "finding": "Relearning to criterion took less time than original learning at every retention interval tested, confirming the savings effect Ebbinghaus reported in 1885.",
      "supports": "That something survives apparent forgetting, and that it shows up as faster relearning rather than as recall.",
      "doesNotSupport": "That professional skills behave like nonsense syllables, or that savings hold at the scale of a career. It is one subject and verbal material.",
      "terms": [],
      "relatedPages": [
        "/research/can-you-regain-a-skill-you-have-lost"
      ]
    },
    {
      "id": "casner-2014",
      "citeAs": "https://thesuperskills.com/research/evidence#casner-2014",
      "section": "learning",
      "authors": "Casner, S. M., Geven, R. W., Recker, M. P. and Schooler, J. W.",
      "year": 2014,
      "title": "The Retention of Manual Flying Skills in the Automated Cockpit",
      "publication": "Human Factors, 56(8)",
      "url": "https://journals.sagepub.com/doi/abs/10.1177/0018720814535628",
      "grade": "peer-reviewed",
      "method": "16 airline pilots flew routine and non-routine scenarios in a Boeing 747-400 simulator with automation level varied.",
      "finding": "Instrument scanning and manual control were mostly intact even where pilots reported little recent practice. The cognitive tasks, tracking position without a map, deciding the next navigational step and recognising instrument failures, showed frequent and significant problems.",
      "supports": "That the hands survive disuse better than the judgement does, in the profession with the most automation experience.",
      "doesNotSupport": "A general rule for knowledge work. The sample is 16 pilots in a simulator.",
      "terms": [
        "deskilling"
      ],
      "relatedPages": [
        "/research/can-you-regain-a-skill-you-have-lost",
        "/research/what-professions-can-learn-from-aviation"
      ]
    },
    {
      "id": "redcross-cpr-2009",
      "citeAs": "https://thesuperskills.com/research/evidence#redcross-cpr-2009",
      "section": "learning",
      "authors": "American Red Cross Advisory Council on First Aid, Aquatics, Safety and Preparedness",
      "year": 2009,
      "title": "Scientific Review: CPR Skill Retention",
      "publication": "American Red Cross",
      "url": "https://www.redcross.org/content/dam/redcross/Health-Safety-Services/scientific-advisory-council/Scientific%20Advisory%20Council%20SCIENTIFIC%20REVIEW%20-%20CPR%20Skill%20Retention.pdf",
      "grade": "compiled-review",
      "method": "Systematic review of 47 articles on CPR skill retention across healthcare and lay populations, with retest intervals from six weeks to 24 months.",
      "finding": "Substantial skill degradation occurs within the first year after training, with declining retention from six to twelve months unless there is refresher training.",
      "supports": "That a life-critical, heavily trained procedural skill decays on a timescale of months without practice.",
      "doesNotSupport": "Any link to patient outcomes. Decay was measured on manikins, not in resuscitations.",
      "terms": [
        "deskilling"
      ],
      "relatedPages": [
        "/research/can-you-regain-a-skill-you-have-lost"
      ]
    },
    {
      "id": "cepeda-2006",
      "citeAs": "https://thesuperskills.com/research/evidence#cepeda-2006",
      "section": "learning",
      "authors": "Cepeda, N. J., Pashler, H., Vul, E., Wixted, J. T. and Rohrer, D.",
      "year": 2006,
      "title": "Distributed Practice in Verbal Recall Tasks: A Review and Quantitative Synthesis",
      "publication": "Psychological Bulletin, 132(3)",
      "url": "https://pubmed.ncbi.nlm.nih.gov/16719566/",
      "grade": "peer-reviewed",
      "method": "Meta-analysis of 839 assessments across 317 experiments in 184 articles.",
      "finding": "Spacing and retention interval act jointly. The gap between practice sessions that produces best retention increases as the target retention interval increases.",
      "supports": "That when practice happens changes how much survives, independently of how much practice there is.",
      "doesNotSupport": "Application to procedural or professional skill. The synthesis covers verbal recall.",
      "terms": [
        "desirable difficulty"
      ],
      "relatedPages": [
        "/research/can-you-regain-a-skill-you-have-lost"
      ]
    },
    {
      "id": "sackett-2023",
      "citeAs": "https://thesuperskills.com/research/evidence#sackett-2023",
      "section": "learning",
      "authors": "Sackett, P. R., Zhang, C., Berry, C. M. and Lievens, F.",
      "year": 2023,
      "title": "Revisiting the design of selection systems in light of new findings regarding the validity of widely used predictors",
      "publication": "Industrial and Organizational Psychology, 16",
      "url": "https://doi.org/10.1017/iop.2023.24",
      "grade": "peer-reviewed",
      "method": "Meta-analytic re-correction of prior personnel-selection meta-analyses, addressing systematic overcorrection for range restriction.",
      "finding": "Work sample validity falls from the widely quoted .54 to .33. Structured interviews fall from .51 to .42 and become the strongest single predictor. Unstructured interviews fall from .38 to .19. General cognitive ability falls from .51 to .31.",
      "supports": "That the numbers most often cited for assessment methods were inflated, and by how much.",
      "doesNotSupport": "That these methods do not work. Work samples and structured interviews remain among the strongest predictors available. It also offers no validity data for AI-era assessment formats.",
      "terms": [],
      "relatedPages": [
        "/research/how-do-you-assess-capability-rather-than-output"
      ]
    },
    {
      "id": "mcdaniel-1994",
      "citeAs": "https://thesuperskills.com/research/evidence#mcdaniel-1994",
      "section": "learning",
      "authors": "McDaniel, M. A., Whetzel, D. L., Schmidt, F. L. and Maurer, S. D.",
      "year": 1994,
      "title": "The Validity of Employment Interviews: A Comprehensive Review and Meta-Analysis",
      "publication": "Journal of Applied Psychology, 79(4)",
      "url": "https://psycnet.apa.org/doi/10.1037/0021-9010.79.4.599",
      "grade": "peer-reviewed",
      "method": "Meta-analysis of 245 validity coefficients from 86,311 individuals.",
      "finding": "Structured interviews predict job performance substantially better than unstructured interviews.",
      "supports": "That structure, rather than the interview itself, is what carries the predictive weight.",
      "doesNotSupport": "A single stable figure for the gap. Estimates of the size of the structure effect vary considerably between meta-analyses.",
      "terms": [],
      "relatedPages": [
        "/research/how-do-you-assess-capability-rather-than-output"
      ]
    },
    {
      "id": "moonen-2013",
      "citeAs": "https://thesuperskills.com/research/evidence#moonen-2013",
      "section": "learning",
      "authors": "Moonen-van Loon, J. M. W., Overeem, K., Donkers, H. H. L. M., van der Vleuten, C. P. M. and Driessen, E. W.",
      "year": 2013,
      "title": "Composite reliability of a workplace-based assessment toolbox for postgraduate medical education",
      "publication": "Advances in Health Sciences Education, 18(5)",
      "url": "https://link.springer.com/article/10.1007/s10459-013-9450-z",
      "grade": "peer-reviewed",
      "method": "Generalisability study of 12,779 workplace-based assessments from 953 medical residents.",
      "finding": "A reliability coefficient of 0.80 required eight mini-CEX observations, nine DOPS or nine multi-source feedback rounds. Combined in a portfolio the requirement fell to seven, eight and one respectively.",
      "supports": "That observing someone at work can reach defensible reliability, and roughly how many observations that takes.",
      "doesNotSupport": "That these scores predict later performance or patient outcomes. It measures consistency, not criterion validity. A single observation is not reliable.",
      "terms": [],
      "relatedPages": [
        "/research/how-do-you-assess-capability-rather-than-output"
      ]
    },
    {
      "id": "prasad-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#prasad-2025",
      "section": "learning",
      "authors": "Prasad, S. et al.",
      "year": 2025,
      "title": "Enhancing medical assessment strategies: a comparative study between structured, traditional and hybrid viva-voce assessment",
      "publication": "BMC Medical Education, 25",
      "url": "https://link.springer.com/article/10.1186/s12909-025-07428-9",
      "grade": "peer-reviewed",
      "method": "151 medical students assessed by two examiners across three viva formats, compared on reliability and perceived fairness.",
      "finding": "Traditional unstructured viva showed significant inter-examiner variability. Structured formats improved fairness and coverage. The best format reached a reliability of 0.663, which is moderate rather than high.",
      "supports": "That oral examination can be made fairer by structuring it, and that unstructured viva has a measurable examiner problem.",
      "doesNotSupport": "That oral assessment is highly reliable. Even the best format tested was only moderate, at a single institution.",
      "terms": [],
      "relatedPages": [
        "/research/how-do-you-assess-capability-rather-than-output"
      ]
    },
    {
      "id": "euaiact-art12",
      "citeAs": "https://thesuperskills.com/research/evidence#euaiact-art12",
      "section": "institutional",
      "authors": "European Union",
      "year": 2024,
      "title": "Article 12, Record-keeping, Regulation (EU) 2024/1689",
      "publication": "Official Journal of the European Union",
      "url": "https://artificialintelligenceact.eu/article/12/",
      "grade": "institutional-survey",
      "method": "Binding regulation.",
      "finding": "High-risk AI systems must technically allow automatic recording of events over the system lifetime, to enable identification of risk situations, post-market monitoring and monitoring of operation. Article 19 requires providers to keep those logs for at least six months.",
      "supports": "That the capability to reconstruct what a system did is now a legal requirement rather than good practice.",
      "doesNotSupport": "That anyone must read the logs, or that a decision must be reviewable at the level of the individual case.",
      "terms": [],
      "relatedPages": [
        "/research/how-do-you-audit-an-ai-assisted-decision"
      ]
    },
    {
      "id": "cjeu-c203-22",
      "citeAs": "https://thesuperskills.com/research/evidence#cjeu-c203-22",
      "section": "institutional",
      "authors": "Court of Justice of the European Union",
      "year": 2025,
      "title": "Dun and Bradstreet Austria, Case C-203/22",
      "publication": "CJEU judgment of 27 February 2025",
      "url": "https://curia.europa.eu/site/upload/docs/application/pdf/2025-02/cp250022en.pdf",
      "grade": "institutional-survey",
      "method": "Preliminary ruling interpreting GDPR Articles 15(1)(h) and 22.",
      "finding": "A controller must describe the procedure and principles actually applied so the person can understand which of their data was used and how. Disclosing the algorithm is not a sufficient explanation, and a blanket trade-secret refusal is not permitted.",
      "supports": "That explanation now has a legal standard, and that the standard is comprehension rather than disclosure.",
      "doesNotSupport": "A right to source code or model weights. The balance with trade secrets is decided case by case.",
      "terms": [],
      "relatedPages": [
        "/research/how-do-you-audit-an-ai-assisted-decision",
        "/research/is-it-ethical-to-let-ai-judge-people"
      ]
    },
    {
      "id": "nist-airmf-2023",
      "citeAs": "https://thesuperskills.com/research/evidence#nist-airmf-2023",
      "section": "institutional",
      "authors": "National Institute of Standards and Technology",
      "year": 2023,
      "title": "AI Risk Management Framework 1.0",
      "publication": "NIST, 26 January 2023",
      "url": "https://www.nist.gov/itl/ai-risk-management-framework",
      "grade": "institutional-survey",
      "method": "Voluntary framework organised around four functions: Govern, Map, Measure and Manage.",
      "finding": "Provides a common structure and vocabulary for AI risk management, with a companion Playbook and a 2024 generative AI profile.",
      "supports": "That a shared vocabulary exists that a board is likely to recognise.",
      "doesNotSupport": "Compliance with anything. It is voluntary and confers no legal status.",
      "terms": [],
      "relatedPages": [
        "/research/how-do-you-audit-an-ai-assisted-decision"
      ]
    },
    {
      "id": "iso-42001-2023",
      "citeAs": "https://thesuperskills.com/research/evidence#iso-42001-2023",
      "section": "institutional",
      "authors": "International Organization for Standardization",
      "year": 2023,
      "title": "ISO/IEC 42001:2023, Artificial intelligence management system",
      "publication": "ISO, December 2023",
      "url": "https://www.iso.org/standard/42001",
      "grade": "institutional-survey",
      "method": "Certifiable management system standard developed by ISO/IEC JTC 1/SC 42.",
      "finding": "Specifies requirements for an AI management system on a Plan-Do-Check-Act structure, against which an organisation can be certified by a third party.",
      "supports": "That an auditable management standard for AI now exists and can be certified.",
      "doesNotSupport": "Legal compliance. Certification against ISO 42001 is not the same as meeting the EU AI Act, and the two are routinely conflated.",
      "terms": [],
      "relatedPages": [
        "/research/how-do-you-audit-an-ai-assisted-decision"
      ]
    },
    {
      "id": "sharma-2023",
      "citeAs": "https://thesuperskills.com/research/evidence#sharma-2023",
      "section": "judgement",
      "authors": "Sharma, M., Tong, M., Korbak, T. et al.",
      "year": 2023,
      "title": "Towards Understanding Sycophancy in Language Models",
      "publication": "ICLR 2024, arXiv 2310.13548",
      "url": "https://arxiv.org/abs/2310.13548",
      "grade": "peer-reviewed",
      "method": "Analysis of five production AI assistants across four free-form text generation tasks, plus analysis of the human preference datasets used to train them.",
      "finding": "All five assistants consistently exhibited sycophancy. Both humans and the preference models trained on their judgements prefer convincingly written sycophantic responses over correct ones a non-negligible share of the time, and optimising against those preference models sometimes sacrifices truthfulness.",
      "supports": "That sycophancy is a predictable consequence of training on human preference, rather than an incidental defect of one product.",
      "doesNotSupport": "Any effect on the quality of a user's decisions. It measures model behaviour, not user outcomes.",
      "terms": [],
      "relatedPages": [
        "/research/how-do-i-get-ai-to-challenge-me",
        "/research/ai-and-human-judgement"
      ]
    },
    {
      "id": "cheng-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#cheng-2026",
      "section": "judgement",
      "authors": "Cheng, M., Lee, C., Khadpe, P., Yu, S., Han, D. and Jurafsky, D.",
      "year": 2026,
      "title": "Sycophantic AI Decreases Prosocial Intentions and Promotes Dependence",
      "publication": "Science",
      "url": "https://arxiv.org/abs/2510.01395",
      "grade": "peer-reviewed",
      "method": "Eleven models tested against human responses on interpersonal advice, plus two preregistered experiments with 1,604 participants, including a live-interaction study using a real personal conflict.",
      "finding": "Models affirmed users' actions about 50 per cent more often than humans did, including 47 per cent endorsement on prompts describing clearly harmful behaviour. Interacting with a sycophantic model reduced participants' willingness to repair an interpersonal conflict and increased their conviction that they were in the right. Participants rated the sycophantic model higher quality, trusted it more, and were more willing to use it again.",
      "supports": "That agreement changes what people subsequently do, and that preference runs in the opposite direction from benefit.",
      "doesNotSupport": "An effect on factual or analytical decisions. The scenarios are interpersonal advice, not technical judgement.",
      "terms": [],
      "relatedPages": [
        "/research/how-do-i-get-ai-to-challenge-me",
        "/research/ai-and-human-judgement"
      ]
    },
    {
      "id": "openai-sycophancy-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#openai-sycophancy-2025",
      "section": "judgement",
      "authors": "OpenAI",
      "year": 2025,
      "title": "Sycophancy in GPT-4o: what happened and what we are doing about it",
      "publication": "OpenAI, 29 April and 2 May 2025",
      "url": "https://openai.com/index/sycophancy-in-gpt-4o/",
      "grade": "institutional-survey",
      "method": "First-party incident postmortem covering an update released on 25 April 2025 and rolled back from 28 April.",
      "finding": "OpenAI attributed the behaviour to weighting short-term user feedback too heavily, which weakened the reward signal that had been holding sycophancy in check. Offline evaluations and A/B tests looked positive. The problem was flagged only by informal qualitative checks, which were overridden.",
      "supports": "That the failure is a measurement failure as much as a training one, and that satisfaction metrics can rise while the product gets worse.",
      "doesNotSupport": "Anything independently verified. It is a company's account of its own incident.",
      "terms": [],
      "relatedPages": [
        "/research/how-do-i-get-ai-to-challenge-me",
        "/research/ai-and-human-judgement"
      ]
    },
    {
      "id": "dubois-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#dubois-2026",
      "section": "judgement",
      "authors": "Dubois, M., Ududec, C., Summerfield, C. and Luettgau, L.",
      "year": 2026,
      "title": "Ask don't tell: Reducing sycophancy in large language models",
      "publication": "arXiv 2602.23971",
      "url": "https://arxiv.org/pdf/2602.23971",
      "grade": "working-paper",
      "method": "Factorial experiments on three frontier models using 40 debatable questions rendered in 11 framings, rated over ten epochs, with a follow-up test across 600 personas.",
      "finding": "Framing input as a statement rather than a question raised sycophancy by roughly 24 percentage points. Prompting the model to convert a user's statement into a question before answering reduced sycophancy more than instructing it not to be sycophantic.",
      "supports": "That how the user phrases the input matters more than telling the model to behave.",
      "doesNotSupport": "That adversarial or devil's advocate prompting works. That was not tested, and no controlled evidence for it was found.",
      "terms": [],
      "relatedPages": [
        "/research/how-do-i-get-ai-to-challenge-me",
        "/research/ai-and-human-judgement"
      ]
    },
    {
      "id": "gaube-workflow-2022",
      "citeAs": "https://thesuperskills.com/research/evidence#gaube-workflow-2022",
      "section": "judgement",
      "authors": "Who Goes First? Influences of Human-AI Workflow on Decision Making in Clinical Imaging",
      "year": 2022,
      "title": "Who Goes First? Influences of Human-AI Workflow on Decision Making in Clinical Imaging",
      "publication": "arXiv 2205.09696",
      "url": "https://arxiv.org/abs/2205.09696",
      "grade": "working-paper",
      "method": "Between-subjects study, 19 veterinary radiologists reviewing 40 X-rays for 33 findings, comparing seeing the AI output alongside the image against committing to a provisional diagnosis first.",
      "finding": "Final diagnoses matched the AI 91 per cent of the time when the AI was seen first, against 89 per cent when the clinician committed first. Where the AI flagged a finding, agreement was 71 per cent against 65 per cent. The anchoring produced only marginal diagnostic gains because of over-reliance on erroneous advice.",
      "supports": "That forming a view before seeing the machine's answer measurably changes the final judgement.",
      "doesNotSupport": "Generalisation. Nineteen participants in one clinical speciality, and the effect sizes are small.",
      "terms": [
        "automation bias"
      ],
      "relatedPages": [
        "/research/how-do-i-get-ai-to-challenge-me",
        "/research/human-at-the-start",
        "/research/ai-and-human-judgement",
        "/research/ai-and-expert-judgement"
      ]
    },
    {
      "id": "jakesch-2023",
      "citeAs": "https://thesuperskills.com/research/evidence#jakesch-2023",
      "section": "humanness",
      "authors": "Jakesch, M., Bhat, A., Buschek, D., Zalmanson, L. and Naaman, M.",
      "year": 2023,
      "title": "Co-Writing with Opinionated Language Models Affects Users' Views",
      "publication": "CHI 2023, ACM",
      "url": "https://arxiv.org/abs/2302.00560",
      "grade": "peer-reviewed",
      "method": "Online experiment with 1,506 participants writing a post on whether social media is good for society, assisted by a tool configured to argue for or against. Opinions rated by 500 independent judges, plus a post-task attitude survey.",
      "finding": "The opinionated model changed both the opinions expressed in participants' writing and their own opinions in the subsequent attitude survey. The effect held among participants who had ample time to write independently. The authors call it latent persuasion.",
      "supports": "That writing assistance moves what people think, not only what they type.",
      "doesNotSupport": "Generalisation across topics, or persistence after the task. One topic, one configuration, self-reported attitudes.",
      "terms": [],
      "relatedPages": [
        "/research/how-do-i-keep-my-own-voice-when-using-ai",
        "/research/is-ai-dangerous",
        "/research/is-it-still-my-idea-if-ai-helped-me-write-it"
      ]
    },
    {
      "id": "joshi-vogel-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#joshi-vogel-2025",
      "section": "humanness",
      "authors": "Joshi, N. and Vogel, D.",
      "year": 2025,
      "title": "Writing with AI Lowers Psychological Ownership, but Longer Prompts Can Help",
      "publication": "ACM Conversational User Interfaces 2025",
      "url": "https://arxiv.org/pdf/2404.03108",
      "grade": "peer-reviewed",
      "method": "Two within-subjects experiments, 31 and 34 participants, writing short stories across conditions from a three-word prompt to writing unaided.",
      "finding": "Psychological ownership rose steadily with prompt length, from a mean of 1.80 with a three-word prompt to 6.29 writing alone. The benefit plateaued once the prompt reached roughly the length of the target text, and no AI-assisted condition reached the ownership of writing unaided.",
      "supports": "That how much of yourself you put in changes whether the output feels like yours, and by a large margin.",
      "doesNotSupport": "Anything about professional or long-form writing. Short fiction, small samples.",
      "terms": [],
      "relatedPages": [
        "/research/how-do-i-keep-my-own-voice-when-using-ai",
        "/research/what-is-the-human-signal",
        "/research/is-it-still-my-idea-if-ai-helped-me-write-it"
      ]
    },
    {
      "id": "sourati-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#sourati-2026",
      "section": "humanness",
      "authors": "Sourati, Z., Ziabari, A. S. and Dehghani, M.",
      "year": 2026,
      "title": "The Homogenizing Effect of Large Language Models on Human Expression and Thought",
      "publication": "Trends in Cognitive Sciences",
      "url": "https://arxiv.org/abs/2508.01491",
      "grade": "compiled-review",
      "method": "Synthesis across linguistics, psychology, cognitive science and computer science. Not an original experiment.",
      "finding": "Argues that models reflect and reinforce dominant styles while marginalising alternatives, and that reliance on a small number of systems amplifies convergence across users.",
      "supports": "That the concern is taken seriously across several fields rather than being a commentator's intuition.",
      "doesNotSupport": "Any specific stylistic feature converging. It presents no new measurement of its own.",
      "terms": [],
      "relatedPages": [
        "/research/how-do-i-keep-my-own-voice-when-using-ai"
      ]
    },
    {
      "id": "bls-projections-eval",
      "citeAs": "https://thesuperskills.com/research/evidence#bls-projections-eval",
      "section": "work",
      "authors": "United States Bureau of Labor Statistics",
      "year": 2018,
      "title": "Occupational Projections Evaluation, 2006 to 2016",
      "publication": "BLS Employment Projections programme",
      "url": "https://www.bls.gov/emp/evaluations/2006-2016-occupational.htm",
      "grade": "institutional-modelling",
      "method": "The agency's own retrospective scoring of its 2006 ten-year projections for 840 detailed occupations against actual 2016 outcomes.",
      "finding": "BLS correctly projected whether an occupation would grow or decline 78 per cent of the time, but correctly projected which occupations would grow faster than the economy as a whole only 57 per cent of the time. Projected average occupational growth was 10.4 per cent against an actual 3.6 per cent, the gap driven by a recession the projections could not foresee.",
      "supports": "That the only organisation which scores its own occupational forecasts gets the useful question, relative growth, barely better than a coin toss, and that its largest errors come from shocks.",
      "doesNotSupport": "That forecasting is worthless. Direction of change at aggregate level was reasonably good. It also predates generative AI, and no equivalent scored record exists for a technological discontinuity.",
      "terms": [],
      "relatedPages": [
        "/research/what-should-i-tell-my-children-to-study"
      ]
    },
    {
      "id": "ifs-lifetime-returns-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#ifs-lifetime-returns-2026",
      "section": "work",
      "authors": "Waltmann, B.",
      "year": 2026,
      "title": "New Estimates of the Impact of Undergraduate Degrees on Lifetime Earnings",
      "publication": "Institute for Fiscal Studies, commissioned by the Department for Education",
      "url": "https://ifs.org.uk/publications/new-estimates-impact-undergraduate-degrees-lifetime-earnings",
      "grade": "institutional-modelling",
      "method": "Administrative linkage of school, university and tax records for the whole 2002 English GCSE cohort, tracked to age 37 with earnings simulated to 67.",
      "finding": "Average net lifetime return to a degree is around 100,000 pounds, with very large variation by subject. Medicine and economics exceed 400,000 pounds on average. Creative arts, philosophy and languages show low or negative average returns. Around 20 per cent of women and 30 per cent of men are projected to see a negative net return.",
      "supports": "That subject choice carries a far larger financial spread than the decision to attend at all.",
      "doesNotSupport": "What any subject will return to someone choosing today. The authors explicitly decline to model structural change including AI.",
      "terms": [],
      "relatedPages": [
        "/research/what-should-i-tell-my-children-to-study"
      ]
    },
    {
      "id": "georgetown-major-payoff-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#georgetown-major-payoff-2025",
      "section": "work",
      "authors": "Georgetown University Center on Education and the Workforce",
      "year": 2025,
      "title": "The Major Payoff: Evaluating Earnings and Employment Outcomes Across Bachelor's Degrees",
      "publication": "Georgetown CEW",
      "url": "https://cew.georgetown.edu/cew-reports/major-payoff/",
      "grade": "institutional-survey",
      "method": "American Community Survey earnings and employment data across 152 majors for prime-age workers and 142 for early career.",
      "finding": "Median prime-age earnings run from 58,000 dollars in education and public service to 98,000 in STEM. Within STEM alone the range is 64,000 to 146,000, and several humanities majors beat the STEM 25th percentile.",
      "supports": "That the spread within a field is often wider than the gap between fields, which undercuts advice given at the level of STEM against humanities.",
      "doesNotSupport": "Anything about the future. It is a cross-sectional snapshot of people already employed.",
      "terms": [],
      "relatedPages": [
        "/research/what-should-i-tell-my-children-to-study"
      ]
    },
    {
      "id": "nyfed-recent-grads-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#nyfed-recent-grads-2026",
      "section": "work",
      "authors": "Federal Reserve Bank of New York",
      "year": 2026,
      "title": "The Labor Market for Recent College Graduates",
      "publication": "New York Fed, data through 2026 Q2",
      "url": "https://www.newyorkfed.org/research/college-labor-market",
      "grade": "institutional-survey",
      "method": "Current Population Survey and ACS tracking of unemployment and underemployment by major since 1990.",
      "finding": "As at the second quarter of 2026, unemployment among recent graduates runs around 5.6 per cent and underemployment around 42 per cent, the highest since 2020.",
      "supports": "That the graduate labour market has tightened measurably, and that underemployment is the larger number.",
      "doesNotSupport": "Attribution to AI. The series is descriptive and the Fed states it is not a forecast.",
      "terms": [],
      "relatedPages": [
        "/research/what-should-i-tell-my-children-to-study"
      ]
    },
    {
      "id": "gathmann-schonberg-2010",
      "citeAs": "https://thesuperskills.com/research/evidence#gathmann-schonberg-2010",
      "section": "work",
      "authors": "Gathmann, C. and Schoenberg, U.",
      "year": 2010,
      "title": "How General Is Human Capital? A Task-Based Approach",
      "publication": "Journal of Labor Economics, 28(1)",
      "url": "https://www.journals.uchicago.edu/doi/10.1086/649786",
      "grade": "peer-reviewed",
      "method": "German administrative employment panel, using task overlap between occupations to measure task-specific human capital.",
      "finding": "Task-specific human capital accounts for up to 52 per cent of overall wage growth. Workers move to occupations with similar task profiles, and the distance of those moves shrinks with experience.",
      "supports": "That skill is portable along task lines rather than being either fully general or locked to one job.",
      "doesNotSupport": "That broad general education transfers well. If anything it argues the opposite, which complicates the study-anything advice rather than supporting it.",
      "terms": [],
      "relatedPages": [
        "/research/what-should-i-tell-my-children-to-study"
      ]
    },
    {
      "id": "simons-2016",
      "citeAs": "https://thesuperskills.com/research/evidence#simons-2016",
      "section": "learning",
      "authors": "Simons, D. J., Boot, W. R., Charness, N., Gathercole, S. E., Chabris, C. F., Hambrick, D. Z. and Stine-Morrow, E. A. L.",
      "year": 2016,
      "title": "Do Brain-Training Programs Work?",
      "publication": "Psychological Science in the Public Interest, 17(3)",
      "url": "https://journals.sagepub.com/doi/10.1177/1529100616661983",
      "grade": "compiled-review",
      "method": "Systematic review applying pre-specified best-practice standards to every study cited by commercial brain-training companies as evidence of efficacy.",
      "finding": "Extensive evidence that training improves performance on the trained tasks, less evidence for closely related tasks, and little evidence that training improves distantly related tasks or everyday cognitive performance. No cited study met all best-practice standards.",
      "supports": "That near transfer is real and far transfer, which is what any claim to have trained attention requires, is essentially unsupported.",
      "doesNotSupport": "That practising a specific skill is useless. The failure is transfer, not learning.",
      "terms": [],
      "relatedPages": [
        "/research/is-attention-a-trainable-skill"
      ]
    },
    {
      "id": "verhaeghen-2021",
      "citeAs": "https://thesuperskills.com/research/evidence#verhaeghen-2021",
      "section": "learning",
      "authors": "Verhaeghen, P.",
      "year": 2021,
      "title": "Mindfulness as Attention Training: Meta-Analyses on the Links Between Attention Performance and Mindfulness Interventions, Long-Term Meditation Practice, and Trait Mindfulness",
      "publication": "Mindfulness, 12(3)",
      "url": "https://link.springer.com/article/10.1007/s12671-020-01532-1",
      "grade": "compiled-review",
      "method": "Three meta-analyses covering 109 effect sizes from 40 intervention studies, 59 effect sizes from 18 long-term meditator studies, and 197 effect sizes from 28 trait studies.",
      "finding": "Average effects were small to moderate, Hedges g of 0.29 for interventions and 0.32 for long-term practice, concentrated in inhibition and executive control rather than sustained attention.",
      "supports": "That something measurable happens, and that it is smaller and narrower than the popular claim.",
      "doesNotSupport": "That the effect survives comparison with an active control. The published breakdown does not separate active from passive controls, so demand effects cannot be excluded.",
      "terms": [],
      "relatedPages": [
        "/research/is-attention-a-trainable-skill"
      ]
    },
    {
      "id": "wiradhany-2017",
      "citeAs": "https://thesuperskills.com/research/evidence#wiradhany-2017",
      "section": "judgement",
      "authors": "Wiradhany, W. and Nieuwenstein, M. R.",
      "year": 2017,
      "title": "Cognitive Control in Media Multitaskers: Two Replication Studies and a Meta-Analysis",
      "publication": "Attention, Perception and Psychophysics, 79(8)",
      "url": "https://link.springer.com/article/10.3758/s13414-017-1408-4",
      "grade": "peer-reviewed",
      "method": "Two direct replications of Ophir, Nass and Wagner (2009), 14 tests at mean power 0.81, plus a meta-analysis of 39 effect sizes.",
      "finding": "Only five of 14 tests showed increased distractibility, and only two survived a Bayesian analysis. The meta-analytic association became non-significant after correcting for small-study effects. The authors question whether the association exists.",
      "supports": "That one of the most-cited findings about media multitasking and attention does not replicate.",
      "doesNotSupport": "That multitasking has no cost at all. This is one specific effect, distractor filtering, not the whole question.",
      "terms": [],
      "relatedPages": [
        "/research/is-attention-a-trainable-skill"
      ]
    },
    {
      "id": "leroy-2009",
      "citeAs": "https://thesuperskills.com/research/evidence#leroy-2009",
      "section": "judgement",
      "authors": "Leroy, S.",
      "year": 2009,
      "title": "Why Is It So Hard to Do My Work? The Challenge of Attention Residue When Switching Between Work Tasks",
      "publication": "Organizational Behavior and Human Decision Processes, 109(2)",
      "url": "https://www.sciencedirect.com/science/article/abs/pii/S0749597809000399",
      "grade": "peer-reviewed",
      "method": "Two laboratory experiments manipulating whether a task was completed or interrupted before switching.",
      "finding": "People struggle to move attention away from an unfinished task, and performance on the next task suffers. Time pressure on the first task helps disengagement.",
      "supports": "That the residue effect exists under controlled conditions, and that completion rather than willpower is what releases attention.",
      "doesNotSupport": "Real-world magnitude. Two laboratory experiments, not a field study.",
      "terms": [],
      "relatedPages": [
        "/research/is-attention-a-trainable-skill"
      ]
    },
    {
      "id": "heinz-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#heinz-2025",
      "section": "frontline",
      "authors": "Heinz, M. V., Mackin, D. M., Trudeau, B. M. et al.",
      "year": 2025,
      "title": "Randomized Trial of a Generative AI Chatbot for Mental Health Treatment",
      "publication": "NEJM AI, 2(4). DOI 10.1056/AIoa2400802",
      "url": "https://ai.nejm.org/doi/full/10.1056/AIoa2400802",
      "grade": "peer-reviewed",
      "method": "Randomised controlled trial, N=210 adults recruited by national advertising, with major depressive disorder, generalised anxiety disorder or clinically high risk for feeding and eating disorders. Four-week intervention with Therabot, an expert-fine-tuned generative chatbot built on a hand-curated CBT corpus, against a WAITLIST control, with four weeks of follow-up.",
      "finding": "Statistically significant symptom reductions across all three diagnostic groups at four and eight weeks. 95 per cent of participants engaged, averaging 260 messages and 6.18 hours over four weeks. Therapeutic alliance was comparable to outpatient psychotherapy (mean WAI 3.59). Expressions of suicidal ideation required staff intervention 15 times, and inappropriate responses required correction 13 times.",
      "supports": "That a purpose-built, expert-curated generative chatbot can produce measurable symptom improvement over four weeks, and that users form a working alliance with it.",
      "doesNotSupport": "That it is safe unsupervised, or that the effect is the chatbot rather than attention and expectation. The control was a waitlist rather than an active comparator, so the treatment effect cannot be separated from the effect of receiving something. The developer confirmed to the FDA advisory committee that humans reviewed all messages in near real time and conducted risk assessments where needed, so the safety record describes a supervised system rather than an autonomous one.",
      "terms": [],
      "relatedPages": [
        "/research/is-it-safe-to-use-ai-for-therapy"
      ]
    },
    {
      "id": "moore-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#moore-2025",
      "section": "frontline",
      "authors": "Moore, J., Grabb, D., Agnew, W., Klyman, K., Chancellor, S., Ong, D. C. and Haber, N.",
      "year": 2025,
      "title": "Expressing stigma and inappropriate responses prevents LLMs from safely replacing mental health providers",
      "publication": "Proceedings of the 2025 ACM Conference on Fairness, Accountability, and Transparency (FAccT). DOI 10.1145/3715275.3732039",
      "url": "https://arxiv.org/abs/2504.18412",
      "grade": "peer-reviewed",
      "method": "Mapping review of therapy guides used by major medical institutions to identify the requirements of a therapeutic relationship, followed by several experiments testing current models, including gpt-4o, against those requirements in naturalistic therapy settings.",
      "finding": "Models expressed stigma towards people with mental health conditions and responded inappropriately to common and critical presentations, including encouraging delusional thinking, which the authors attribute to sycophancy. The pattern persisted in larger and newer models.",
      "supports": "That the failures are structural rather than a matter of model size, and that current safety training does not remove them.",
      "doesNotSupport": "How often this occurs in real use, or what happens with purpose-built clinical systems rather than general-purpose models. These are constructed scenarios, not observed patient interactions.",
      "terms": [],
      "relatedPages": [
        "/research/is-it-safe-to-use-ai-for-therapy"
      ]
    },
    {
      "id": "fda-dhac-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#fda-dhac-2025",
      "section": "institutional",
      "authors": "United States Food and Drug Administration, Digital Health Advisory Committee",
      "year": 2025,
      "title": "Generative Artificial Intelligence-Enabled Digital Mental Health Medical Devices: meeting summary, 6 November 2025",
      "publication": "FDA Center for Devices and Radiological Health. Docket FDA-2025-N-2338",
      "url": "https://www.fda.gov/media/190450/download",
      "grade": "institutional-survey",
      "method": "Public advisory committee meeting with FDA presentations, 16 open public hearing speakers and structured committee deliberation on three scenarios: a prescription LLM therapy device for adults with major depressive disorder, over-the-counter and autonomous expansions, and use with under-21s.",
      "finding": "The director of CDRH stated that FDA has authorised more than 1,200 AI-enabled medical devices and none yet involve generative AI for mental health conditions. The committee asked for premarket comparators BEYOND waitlist controls, judged autonomous over-the-counter use for undiagnosed users substantially higher risk, described multi-condition autonomous use as the highest risk of all, and expressed strong discomfort with autonomous use in children and adolescents. One member noted that reminders that a system is not human cannot overcome automation bias.",
      "supports": "The regulatory position as at November 2025, in the regulator's own words, and that the committee independently identified the waitlist-control weakness in the existing trial evidence.",
      "doesNotSupport": "What FDA will decide. Advisory committee recommendations are non-binding, and no rule follows from this meeting.",
      "terms": [],
      "relatedPages": [
        "/research/is-it-safe-to-use-ai-for-therapy"
      ]
    },
    {
      "id": "illinois-wopr-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#illinois-wopr-2025",
      "section": "institutional",
      "authors": "State of Illinois",
      "year": 2025,
      "title": "Wellness and Oversight for Psychological Resources Act (HB1806), signed 1 August 2025",
      "publication": "Illinois Department of Financial and Professional Regulation",
      "url": "https://idfpr.illinois.gov/news/2025/gov-pritzker-signs-state-leg-prohibiting-ai-therapy-in-il.html",
      "grade": "institutional-survey",
      "method": "State legislation, passed almost unanimously and signed by the Governor.",
      "finding": "Prohibits the use of AI to provide therapy or perform therapeutic decision-making, including direct therapeutic communication with clients and detection of a client's emotional or mental state, while permitting administrative and supplementary use by licensed professionals. Penalties reach 10,000 dollars per violation.",
      "supports": "That at least one jurisdiction has moved from guidance to prohibition, and where it drew the line: the boundary is drawn at therapeutic decisions and direct therapeutic communication, not at the technology.",
      "doesNotSupport": "Anything about effectiveness, and nothing about other jurisdictions. One state, and its scope is contested.",
      "terms": [],
      "relatedPages": [
        "/research/is-it-safe-to-use-ai-for-therapy"
      ]
    },
    {
      "id": "frey-osborne-2013",
      "citeAs": "https://thesuperskills.com/research/evidence#frey-osborne-2013",
      "section": "work",
      "authors": "Frey, C. B. and Osborne, M. A.",
      "year": 2013,
      "title": "The Future of Employment: How Susceptible Are Jobs to Computerisation?",
      "publication": "Oxford Martin School working paper, 17 September 2013. Later published in Technological Forecasting and Social Change, 114 (2017)",
      "url": "https://oms-www.files.svdcdn.com/production/downloads/academic/The_Future_of_Employment.pdf",
      "grade": "working-paper",
      "method": "Probability of computerisation estimated for 702 detailed US occupations using a Gaussian process classifier, trained on 70 occupations hand-labelled by machine learning researchers at an Oxford workshop.",
      "finding": "In the authors' own words: 'about 47 percent of total US employment is at risk'. The paper's title asks how SUSCEPTIBLE jobs are, and the estimate is of technical susceptibility to computerisation, not a forecast of job losses.",
      "supports": "That a large share of US employment sits in occupations whose tasks were, in 2013, judged technically susceptible to computerisation.",
      "doesNotSupport": "That 47 per cent of jobs will be, or have been, lost. It is not a prediction, carries no date attached to any loss, and models whole occupations rather than tasks within them. The routine citation as '47 per cent of jobs will disappear' reverses what the paper claims. The working paper is 2013; the journal version is 2017, and the two dates are frequently confused.",
      "terms": [
        "labour market"
      ],
      "relatedPages": [
        "/research/the-most-quoted-ai-statistics-checked"
      ]
    },
    {
      "id": "arntz-2016",
      "citeAs": "https://thesuperskills.com/research/evidence#arntz-2016",
      "section": "institutional",
      "authors": "Arntz, M., Gregory, T. and Zierahn, U.",
      "year": 2016,
      "title": "The Risk of Automation for Jobs in OECD Countries: A Comparative Analysis",
      "publication": "OECD Social, Employment and Migration Working Papers No. 189",
      "url": "https://www.oecd.org/en/publications/the-risk-of-automation-for-jobs-in-oecd-countries_5jlz9h56dvq7-en.html",
      "grade": "institutional-modelling",
      "method": "Task-based re-estimation across 21 OECD countries using PIAAC survey data, accounting for the heterogeneity of tasks WITHIN occupations rather than treating whole occupations as automatable.",
      "finding": "9 per cent of jobs automatable on average across 21 countries, ranging from 6 per cent in Korea to 12 per cent in Austria. The authors state the occupation-based approach 'might lead to an overestimation of job automatibility, as occupations labelled as high-risk occupations often still contain a substantial share of tasks that are hard to automate'.",
      "supports": "That the headline automation figure is highly sensitive to whether you model occupations or tasks, and that the difference is roughly fivefold on the same question.",
      "doesNotSupport": "That 9 per cent is correct and 47 per cent wrong. Both are model outputs resting on assumptions, and neither has been scored against what happened.",
      "terms": [
        "labour market"
      ],
      "relatedPages": [
        "/research/the-most-quoted-ai-statistics-checked"
      ]
    },
    {
      "id": "hughes-2011",
      "citeAs": "https://thesuperskills.com/research/evidence#hughes-2011",
      "section": "institutional",
      "authors": "Hughes, M.",
      "year": 2011,
      "title": "Do 70 Per Cent of All Organizational Change Initiatives Really Fail?",
      "publication": "Journal of Change Management, 11(4), 451-464. DOI 10.1080/14697017.2011.630506",
      "url": "https://research.brighton.ac.uk/en/publications/do-70-per-cent-of-all-organizational-change-initiatives-really-fa/",
      "grade": "peer-reviewed",
      "method": "Critical review of five separate published instances of the 70 per cent organisational change failure rate, tracing each to its stated source.",
      "finding": "In the author's words: 'whilst the existence of a popular narrative of 70 percent organizational change failure is acknowledged, there is no valid and reliable empirical evidence to support such a narrative.'",
      "supports": "That one of the most repeated statistics in management has no traceable empirical basis, and that this was established in a peer-reviewed journal fifteen years ago and ignored.",
      "doesNotSupport": "That change programmes usually succeed. The finding is about the absence of evidence for a specific number, not about the true rate, which remains unmeasured.",
      "terms": [],
      "relatedPages": [
        "/research/the-most-quoted-ai-statistics-checked"
      ]
    },
    {
      "id": "acemoglu-restrepo-2019",
      "citeAs": "https://thesuperskills.com/research/evidence#acemoglu-restrepo-2019",
      "section": "work",
      "authors": "Acemoglu, D. and Restrepo, P.",
      "year": 2019,
      "title": "Automation and New Tasks: How Technology Displaces and Reinstates Labor",
      "publication": "Journal of Economic Perspectives, 33(2), 3-30. DOI 10.1257/jep.33.2.3",
      "url": "https://www.aeaweb.org/articles?id=10.1257/jep.33.2.3",
      "grade": "peer-reviewed",
      "method": "Task-based theoretical framework in which production is allocated between capital and labour, with an empirical decomposition of US industry-level data over recent decades.",
      "finding": "Automation shifts the task content of production against labour through a displacement effect, and therefore ALWAYS reduces the labour share in value added, and may reduce labour demand even while raising productivity. The counterweight is the creation of new tasks in which labour has a comparative advantage, which always raises the labour share. Their decomposition attributes slower employment growth over three decades to an accelerating displacement effect, a weaker reinstatement effect and slower productivity growth.",
      "supports": "That the distributional question is separate from the productivity question, and that a technology can raise output while reducing labour's share of it. This is the framework almost every serious argument in this area now runs through.",
      "doesNotSupport": "That AI specifically will behave this way. The empirical work predates generative AI and concerns industrial automation and robotics.",
      "terms": [
        "labour market"
      ],
      "relatedPages": [
        "/research/who-captures-the-productivity-gains-from-ai"
      ]
    },
    {
      "id": "lang-masai-2023",
      "citeAs": "https://thesuperskills.com/research/evidence#lang-masai-2023",
      "section": "frontline",
      "authors": "Lang, K., Josefsson, V., Larsson, A.-M. et al.",
      "year": 2023,
      "title": "Artificial intelligence-supported screen reading versus standard double reading in the Mammography Screening with Artificial Intelligence trial (MASAI): a clinical safety analysis",
      "publication": "The Lancet Oncology, 24(8), 936-944. DOI 10.1016/S1470-2045(23)00298-X",
      "url": "https://www.thelancet.com/journals/lanonc/article/PIIS1470-2045(23)00298-X/fulltext",
      "grade": "peer-reviewed",
      "method": "Planned interim safety analysis of a randomised, controlled, non-inferiority, single-blinded screening accuracy trial. 80,033 women aged 40-80 screened at four sites in southwest Sweden between April 2021 and July 2022, randomised 1:1 to AI-supported screen reading or standard double reading by two radiologists.",
      "finding": "Cancer detection was six per 1,000 screened women with AI support against five per 1,000 with standard double reading, 41 more cancers detected. The false-positive rate was 1.5 per cent in both arms. Screen readings fell from 83,231 in the control arm to 46,345 in the AI arm, a 44 per cent reduction in screen-reading workload.",
      "supports": "That a triage-plus-detection-support workflow with a radiologist retaining the recall decision can hold detection while roughly halving reading volume, in a randomised population-based programme.",
      "doesNotSupport": "Patient benefit, which the interim analysis was not designed to test. It is also one mammography device, one AI system, one country, and moderately to highly experienced readers, which the authors state as limits on generalisability.",
      "terms": [
        "human-AI collaboration",
        "clinical AI"
      ],
      "relatedPages": [
        "/research/how-will-ai-change-medicine",
        "/research/ai-and-human-judgement"
      ]
    },
    {
      "id": "masai-final-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#masai-final-2026",
      "section": "frontline",
      "authors": "Gommers, J., Lang, K., Hofvind, S. et al.",
      "year": 2026,
      "title": "Interval cancer, sensitivity, and specificity comparing AI-supported mammography screening with standard double reading without AI in the MASAI study",
      "publication": "The Lancet. DOI 10.1016/S0140-6736(25)02464-X. Published 29 January 2026",
      "url": "https://www.thelancet.com/journals/lancet/article/PIIS0140-6736(25)02464-X/fulltext",
      "grade": "peer-reviewed",
      "method": "Full results of a randomised, controlled, non-inferiority, single-blinded, population-based screening-accuracy trial. Over 100,000 women screened at four Swedish sites between April 2021 and December 2022, with two years of follow-up.",
      "finding": "Interval cancers fell from 1.76 per 1,000 women (93/52,872) in the control arm to 1.55 per 1,000 (82/53,043) in the AI arm, a 12 per cent reduction. There were 16 per cent fewer invasive (75 v 89), 21 per cent fewer large (38 v 48) and 27 per cent fewer aggressive-subtype (43 v 59) interval cancers. Cancers detected at screening rose from 74 per cent (262/355) to 81 per cent (338/420) of all cases. False positives were 1.5 per cent in the intervention arm and 1.4 per cent in the control arm.",
      "supports": "That AI-supported screen reading, with a radiologist retaining the recall decision, reduced interval cancers over two years in a randomised population-based programme. The only outcome-level randomised evidence of this kind in clinical AI.",
      "doesNotSupport": "Generalisation beyond one country, one mammography device, one AI system and experienced readers, all stated as limitations by the authors. Mortality was not an endpoint, and cost-effectiveness was not assessed. The first author's own summary states the design does not support replacing radiologists, since at least one still reads every case.",
      "terms": [
        "human-AI collaboration",
        "clinical AI"
      ],
      "relatedPages": [
        "/research/how-will-ai-change-medicine",
        "/research/ai-and-human-judgement"
      ]
    },
    {
      "id": "wong-2021",
      "citeAs": "https://thesuperskills.com/research/evidence#wong-2021",
      "section": "frontline",
      "authors": "Wong, A., Otles, E., Donnelly, J. P. et al.",
      "year": 2021,
      "title": "External Validation of a Widely Implemented Proprietary Sepsis Prediction Model in Hospitalized Patients",
      "publication": "JAMA Internal Medicine, 181(8), 1065-1070. DOI 10.1001/jamainternmed.2021.2626",
      "url": "https://jamanetwork.com/journals/jamainternalmedicine/fullarticle/2781307",
      "grade": "peer-reviewed",
      "method": "Retrospective external validation cohort study. 27,697 patients aged 18 or over across 38,455 hospitalisations at Michigan Medicine, 6 December 2018 to 20 October 2019. Sepsis occurred in 7 per cent of hospitalisations.",
      "finding": "The Epic Sepsis Model achieved a hospitalisation-level area under the curve of 0.63 (95% CI 0.62-0.64), against the 0.76-0.83 cited by its developer. At the alerting threshold in clinical use, sensitivity was 33 per cent, specificity 83 per cent, positive predictive value 12 per cent. It did not identify 1,709 of 2,552 septic hospitalisations (67 per cent), 60 per cent of whom received timely antibiotics anyway, while crossing the alert threshold in 18 per cent of all hospitalisations (6,971 of 38,455), requiring eight patients to be evaluated per case of sepsis found.",
      "supports": "That a proprietary clinical prediction model deployed at national scale can perform far below its developer's stated figures when validated independently, and that nobody had checked. The authors' own conclusion is that widespread adoption despite poor performance raises fundamental concerns about sepsis management nationally.",
      "doesNotSupport": "That all clinical prediction models fail, or that this model performs identically elsewhere. It is one model at one academic health system, and the authors note the theoretical alert burden does not account for real-world trigger criteria and lockouts.",
      "terms": [
        "clinical AI",
        "automation bias",
        "alert fatigue"
      ],
      "relatedPages": [
        "/research/how-will-ai-change-medicine",
        "/research/is-it-ethical-to-let-ai-judge-people"
      ]
    },
    {
      "id": "goh-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#goh-2024",
      "section": "frontline",
      "authors": "Goh, E., Gallo, R., Hom, J. et al.",
      "year": 2024,
      "title": "Large Language Model Influence on Diagnostic Reasoning: A Randomized Clinical Trial",
      "publication": "JAMA Network Open, 7(10), e2440969. DOI 10.1001/jamanetworkopen.2024.40969",
      "url": "https://jamanetwork.com/journals/jamanetworkopen/fullarticle/2825395",
      "grade": "peer-reviewed",
      "method": "Single-blind randomised clinical trial, 29 November to 29 December 2023. 50 US-licensed physicians (26 attendings, 24 residents) in family medicine, internal medicine or emergency medicine, randomised to GPT-4 or to conventional resources, working through up to six clinical vignettes, graded blind against a validated diagnostic reasoning rubric. 244 cases completed.",
      "finding": "Median diagnostic reasoning score per case was 76 per cent (IQR 66-87) with the LLM and 74 per cent (IQR 63-84) with conventional resources: an adjusted difference of 2 percentage points (95% CI -4 to 8, p=0.60). Median time per case was 519 seconds against 565, adjusted difference -82 seconds (95% CI -195 to 31, p=0.20). The LLM alone scored a median 92 per cent, 16 percentage points above the conventional-resources group (95% CI 2-30, p=0.03).",
      "supports": "That adding a capable model to a physician did not, in this trial, improve diagnostic reasoning or save time, while the same model working alone outperformed both groups of clinicians. The performance of a human-plus-model pairing cannot be inferred from the performance of either part.",
      "doesNotSupport": "That LLMs should diagnose autonomously, which the authors explicitly reject. Six curated vignettes exclude history-taking, examination, context and time, which is most of clinical reasoning. Participants received no prompt-engineering training, and the authors offer prompting and interaction design as explanations.",
      "terms": [
        "human-AI collaboration",
        "clinical AI",
        "verification"
      ],
      "relatedPages": [
        "/research/how-will-ai-change-medicine",
        "/research/which-professions-face-the-greatest-deskilling-risk",
        "/research/ai-and-human-judgement"
      ]
    },
    {
      "id": "lukac-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#lukac-2025",
      "section": "frontline",
      "authors": "Lukac, P. J., Turner, W., Vangala, S., Chin, A. T., Khalili, J., Shih, Y.-C. T., Sarkisian, C., Cheng, E. M. and Mafi, J. N.",
      "year": 2025,
      "title": "Ambient AI Scribes in Clinical Practice: A Randomized Trial",
      "publication": "NEJM AI, 2(12). DOI 10.1056/AIoa2501000. Preprint: medRxiv 2025.07.10.25331333",
      "url": "https://ai.nejm.org/doi/abs/10.1056/AIoa2501000",
      "grade": "peer-reviewed",
      "method": "Parallel three-arm pragmatic randomised clinical trial at one US health system. 238 outpatient physicians across 14 specialties randomised 1:1:1 by covariate-constrained randomisation to Microsoft DAX, Nabla or usual care, 4 November 2024 to 3 January 2025, with the second intervention month compared to baseline.",
      "finding": "Time writing a note fell by an estimated 18 seconds in the control arm, 23 seconds in the DAX arm and 41 seconds in the Nabla arm. Only Nabla differed significantly from control (-9.5 per cent, 95% CI -17.2 to -1.8, p=0.02); DAX did not (-1.7 per cent, 95% CI -9.4 to +5.9, p=0.66). Scribe users improved on Mini-Z burnout (+2.76, p<0.001), task load (-35.8, p=0.01) and work exhaustion (-0.27, p=0.01), with no significant difference between the two products on any psychometric. Roughly 15 per cent of physicians given a tool never used it.",
      "supports": "That ambient documentation produces a small, product-dependent time saving and a more consistent wellbeing effect, measured against a randomised control rather than against a before-and-after.",
      "doesNotSupport": "A general time saving for ambient AI. One health system, majority female sample, a two-month contract-limited window, and the authors flag that the electronic record's own time metrics do not count editing done inside the scribe platform, so reported savings may be overstated. They note this limitation affects all studies using those metrics.",
      "terms": [
        "productivity",
        "clinical AI"
      ],
      "relatedPages": [
        "/research/how-will-ai-change-medicine"
      ]
    },
    {
      "id": "dellacqua-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#dellacqua-2025",
      "section": "collaboration",
      "authors": "Dell'Acqua, F., Ayoubi, C., Lifshitz, H., Sadun, R., Mollick, E., Mollick, L., Han, Y., Goldman, J., Nair, H., Taub, S. and Lakhani, K.",
      "year": 2025,
      "title": "The Cybernetic Teammate: A Field Experiment on Generative AI Reshaping Teamwork and Expertise",
      "publication": "NBER Working Paper 33641, April 2025. Published as 'The Cybernetic Teammate: A Field Experiment on Generative AI and Teamwork', Organization Science, June 2026. DOI 10.1287/orsc.2025.20702",
      "url": "https://www.nber.org/papers/w33641",
      "grade": "peer-reviewed",
      "method": "Pre-registered field experiment with 776 professionals at Procter and Gamble working on real product innovation challenges, randomised both on AI access and on working individually or in a two-person new product development team.",
      "finding": "Individuals working with AI matched the performance of two-person teams working without it. AI use removed the functional split in proposals: without AI, research and development professionals proposed more technical solutions and commercial professionals more commercially oriented ones, while professionals using AI produced balanced solutions regardless of background. Participants using AI reported more positive emotional responses.",
      "supports": "That a model can substitute for measurable parts of what a second human teammate contributes, including some of the social and motivational function, and that it flattens the differences in output that come from professional training.",
      "doesNotSupport": "Client or business outcomes, which were not measured, or any effect over time. One firm, one task type, a single session. Procter and Gamble provided financial support to the institute involved and one author had consulted for the firm, both disclosed in the paper. That the flattened output is better rather than merely more balanced is not established.",
      "terms": [
        "human-AI collaboration",
        "expertise",
        "productivity"
      ],
      "relatedPages": [
        "/research/how-will-ai-change-consulting",
        "/research/does-ai-make-everyone-think-alike",
        "/research/what-is-the-shared-prompt-review",
        "/research/what-is-automating-versus-informating"
      ]
    },
    {
      "id": "magesh-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#magesh-2025",
      "section": "professions",
      "authors": "Magesh, V., Surani, F., Dahl, M., Suzgun, M., Manning, C. D. and Ho, D. E.",
      "year": 2025,
      "title": "Hallucination-Free? Assessing the Reliability of Leading AI Legal Research Tools",
      "publication": "Journal of Empirical Legal Studies, 22(2), 216-242. DOI 10.1111/jels.12413",
      "url": "https://onlinelibrary.wiley.com/doi/full/10.1111/jels.12413",
      "grade": "peer-reviewed",
      "method": "First preregistered empirical evaluation of retrieval-augmented legal research tools. Over 200 handwritten legal queries across four categories, preregistered with the Open Science Foundation in March 2024, run against Lexis+ AI, Westlaw AI-Assisted Research, Ask Practical Law AI and GPT-4, graded on whether responses were correct and grounded in the sources cited.",
      "finding": "The three commercial legal tools each hallucinated between 17 and 33 per cent of the time. Lexis+ AI was accurate on 65 per cent of queries and incomplete on 18 per cent; Westlaw AI-Assisted Research was accurate 42 per cent of the time with a hallucination in one-third of responses; Ask Practical Law AI was incomplete on 62 per cent of queries. Westlaw produced the longest answers, averaging 350 words against 219 and 175, which the authors link to both its higher hallucination rate and the verification burden it imposes.",
      "supports": "That retrieval-augmented generation reduces hallucination relative to a general chatbot without eliminating it, and that provider claims of hallucination-free citation were overstated. Longer generated answers carry more falsifiable propositions and require more checking.",
      "doesNotSupport": "Current performance of any named product. These are specific versions tested in 2024 and providers update continuously. It also does not measure legal outcomes, only response accuracy against expert grading.",
      "terms": [
        "hallucination",
        "verification",
        "legal profession"
      ],
      "relatedPages": [
        "/research/how-will-ai-change-law"
      ]
    },
    {
      "id": "charlotin-hallucination-db",
      "citeAs": "https://thesuperskills.com/research/evidence#charlotin-hallucination-db",
      "section": "professions",
      "authors": "Charlotin, D.",
      "year": 2026,
      "title": "AI Hallucination Cases database",
      "publication": "damiencharlotin.com, updated daily. Figures read from the update of 27 August 2026",
      "url": "https://www.damiencharlotin.com/hallucinations/",
      "grade": "compiled-review",
      "method": "Curated database of legal decisions worldwide in which a court or tribunal has explicitly found or implied that a party relied on hallucinated material. Excludes mere allegations, with a small stated exception. Coverage begins in the second quarter of 2023.",
      "finding": "1,963 cases identified as at 27 August 2026. By jurisdiction: United States 1,345, Canada 214, Australia 98, United Kingdom 62, Israel 57, with more than thirty other countries represented. By party responsible: self-represented litigants 1,127, lawyers 784, judges 29, expert witnesses 15. By nature: fabricated material 1,634, misrepresented authority 816, false quotations 528, outdated advice 33.",
      "supports": "That fabricated legal authority reaching courts is a documented, dated, worldwide phenomenon rather than an anecdote, and that self-represented litigants account for more recorded instances than lawyers do.",
      "doesNotSupport": "The true rate. The database counts decisions where a court addressed the point, so instances nobody noticed, or resolved without a written decision, are invisible by construction; its author states this. It is a curated compilation rather than a sampled study, so it cannot support a denominator or a trend rate.",
      "terms": [
        "hallucination",
        "legal profession",
        "verification"
      ],
      "relatedPages": [
        "/research/how-will-ai-change-law"
      ]
    },
    {
      "id": "ayinde-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#ayinde-2025",
      "section": "professions",
      "authors": "Divisional Court of England and Wales (Dame Victoria Sharp P and Johnson J)",
      "year": 2025,
      "title": "Ayinde v London Borough of Haringey and Al-Haroun v Qatar National Bank",
      "publication": "[2025] EWHC 1383 (Admin), judgment of 6 June 2025",
      "url": "https://www.judiciary.uk/judgments/ayinde-v-london-borough-of-haringey-and-al-haroun-v-qatar-national-bank/",
      "grade": "compiled-review",
      "method": "Two referrals under the Hamid jurisdiction, heard together, arising from fabricated citations placed before the court. In Ayinde, five cited authorities did not exist. In Al-Haroun, a claim for £89.4 million, eighteen of forty-five cited authorities did not exist.",
      "finding": "At [6] the court held that freely available generative AI tools trained on a large language model are not capable of conducting reliable legal research. At [7] those using them have a professional duty to check accuracy against authoritative sources, which the court names. At [8] the duty extends to lawyers relying on others' AI-assisted work. At [81] a lawyer is not entitled to rely on their lay client for the accuracy of citations. At [23] the available powers run from public admonition and wasted costs to contempt and referral to the police, and at [31] admonishment alone is unlikely to suffice save in exceptional circumstances. At [9] leadership responsibility falls on heads of chambers and managing partners, and the court states it will inquire in future hearings whether that responsibility was fulfilled.",
      "supports": "That in England and Wales the verification duty is settled, non-delegable and extends upward to those who supervise. It also establishes that the court will treat a fabricated citation as a possible supervision failure rather than only as an individual one.",
      "doesNotSupport": "Anything about jurisdictions other than England and Wales, or about a lawyer who checked competently and was still misled, which no reported case has yet tested. It is a judgment rather than an empirical study, and it measures nothing.",
      "terms": [
        "verification",
        "legal profession",
        "accountability"
      ],
      "relatedPages": [
        "/research/how-will-ai-change-law"
      ]
    },
    {
      "id": "hohenstein-2023",
      "citeAs": "https://thesuperskills.com/research/evidence#hohenstein-2023",
      "section": "humanness",
      "authors": "Hohenstein, J., Kizilcec, R. F., DiFranzo, D., Aghajari, Z., Mieczkowski, H., Levy, K., Naaman, M., Hancock, J. and Jung, M. F.",
      "year": 2023,
      "title": "Artificial intelligence in communication impacts language and social relationships",
      "publication": "Scientific Reports, 13, 5487. DOI 10.1038/s41598-023-30938-9",
      "url": "https://www.nature.com/articles/s41598-023-30938-9",
      "grade": "peer-reviewed",
      "method": "Two randomised experiments on algorithmic response suggestions (smart replies) in live text chat. Study 1: 438 Mechanical Turk crowdworkers in 219 pairs discussing a policy question, smart-reply availability randomised separately for each partner, analysed with an instrumental-variable design. Study 2: 582 crowdworkers in 291 pairs, with the sentiment of the suggested replies manipulated. Preregistered (AsPredicted #40389).",
      "finding": "Smart replies accounted for 14.3 per cent of messages and produced 10.2 per cent more messages per minute. ON LANGUAGE: greater use of smart replies by a partner led the other person to send messages with more positive sentiment (IV estimate b=0.178, t(205)=2.02, p=0.045), and the effect held when smart-reply messages were EXCLUDED from the sentiment score (b=0.208, t(205)=2.17, p=0.031), so it appears in sentences the person composed themselves. Merely having suggestions available without using them did not move sentiment (b=0.019, p=0.1801). Study 2 manipulated the emotional tone of the suggestions and conversation sentiment followed it. ON PERCEPTION: greater ACTUAL use by a partner improved the other person's rating of their cooperation (b=15.66, p=0.018) and felt affiliation towards them (b=21.79, p=0.007), with no effect on dominance. Greater SUSPECTED use had the opposite sign: the more a participant believed their partner used smart replies, the less cooperative (p<0.0001) and less affiliative (p<0.0001) they rated them, after controlling for actual use. Suspicion tracked actual use only weakly (Pearson's r=0.22).",
      "supports": "Two separate things. On language, that an algorithmic suggestion system changes what a person writes in their own words, under randomisation, in the direction of the system's own tone. That is the mechanism behind homogenisation observed at the level of the individual, and the null result for mere availability locates it in the act of adopting the phrasing rather than in exposure to it. On perception, that the interpersonal penalty for AI-assisted messaging attaches to being suspected rather than to using it, and that suspicion is a poor detector; actual use made people seem warmer, not colder.",
      "doesNotSupport": "Anything longitudinal, which the authors state directly. The suspicion result is correlational and they say it does not show causally how attitudes shift in response to actual use. Participants were crowdworkers discussing policy with strangers over a few minutes, not intimates and not colleagues. The language effect is measured as sentiment, which is one dimension of style and not the range of what a person might have said, so it evidences convergence in tone and not convergence in thought. The system studied is a smart-reply suggester with short canned options, not a generative model writing paragraphs. A publisher correction (10.1038/s41598-023-43601-0, 3 October 2023) added an omitted funding statement and changed no result.",
      "terms": [
        "AI-mediated communication",
        "authenticity",
        "disclosure",
        "homogenisation",
        "voice"
      ],
      "relatedPages": [
        "/research/should-i-use-ai-to-write-personal-messages",
        "/research/outsourced-recognition",
        "/research/does-ai-make-everyone-think-alike",
        "/research/what-is-the-human-signal",
        "/research/what-is-the-shared-prompt-review",
        "/research/is-it-still-my-idea-if-ai-helped-me-write-it"
      ]
    },
    {
      "id": "melumad-yun-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#melumad-yun-2025",
      "section": "judgement",
      "authors": "Melumad, S. and Yun, J. H.",
      "year": 2025,
      "title": "Experimental evidence of the effects of large language models versus web search on depth of learning",
      "publication": "PNAS Nexus, 4(10), pgaf316. DOI 10.1093/pnasnexus/pgaf316",
      "url": "https://academic.oup.com/pnasnexus/article/4/10/pgaf316/8303888",
      "grade": "peer-reviewed",
      "method": "Seven online and laboratory experiments, four in the paper and three in the supplement. Participants learned a practical topic using either a large language model or web search links, then wrote advice for a friend. Experiment 1: 1,104 participants, real ChatGPT versus real Google. Experiment 2: 1,979 participants, simulated search holding the underlying FACTS identical across conditions. Experiment 3: 250 lab participants, standard Google versus Google with AI Overviews. Advice scored for length, named entities and pairwise similarity.",
      "finding": "Experiment 2, with facts held constant: time engaging with results 83.65 seconds with the summary against 124.32 with links; learned new things 3.71 against 3.96; ownership of knowledge 3.41 against 3.66; thought and effort in advice 3.85 against 4.11; advice 64.49 words against 74.22; references to facts 4.00 against 4.61; pairwise cosine similarity between participants' advice 0.224 against 0.072. Rated comprehensiveness did not differ (4.30 against 4.25, p=0.264). A supplementary condition adding real-time web links to the AI summary did not remove the effect, because only 26 per cent of participants clicked any link.",
      "supports": "That the format of a search result, independent of its content, changes how much effort people invest, how deeply they report learning, and the specificity and distinctiveness of what they can then produce.",
      "doesNotSupport": "That knowledge objectively declined. Depth of learning is self-reported throughout and no experiment included a recall or comprehension test; time on task is described by the authors as a proxy for effort. Topics were practical how-to tasks over short horizons. The paper states its own total sample twice and inconsistently, as 10,462 in the abstract and 10,426 in the introduction.",
      "terms": [
        "cognitive offloading",
        "desirable difficulty",
        "summarisation"
      ],
      "relatedPages": [
        "/research/should-i-let-ai-summarise-everything-i-read",
        "/research/ai-and-human-judgement"
      ]
    },
    {
      "id": "fisher-2015",
      "citeAs": "https://thesuperskills.com/research/evidence#fisher-2015",
      "section": "judgement",
      "authors": "Fisher, M., Goddu, M. K. and Keil, F. C.",
      "year": 2015,
      "title": "Searching for explanations: How the Internet inflates estimates of internal knowledge",
      "publication": "Journal of Experimental Psychology: General, 144(3), 674-687. DOI 10.1037/xge0000070",
      "url": "https://bpb-us-w2.wpmucdn.com/campuspress.yale.edu/dist/c/259/files/2015/03/pdf-16ueczx.pdf",
      "grade": "peer-reviewed",
      "method": "Nine between-subjects experiments, 1,708 US participants via Amazon Mechanical Turk. An induction phase in which participants either searched the internet for explanations or were told not to, followed by self-ratings of their ability to explain questions in six domains unrelated to the induction material.",
      "finding": "Searching inflated self-rated explanatory ability, with Cohen's d from 0.35 to 0.63 across studies, and the effect appeared across all six unrelated domains. It persisted when the search returned no answer to the question asked (Experiment 4b: 4.11 against 4.00 for those who found an answer, both far above a 3.05 no-search baseline) and when it returned no results at all (Experiment 4c). It disappeared for autobiographical topics where the internet would not help (Experiment 3, p=0.30).",
      "supports": "That access to information is mistaken for knowledge held internally, that the illusion is specific to searchable domains rather than general overconfidence, and that it does not require the search to succeed.",
      "doesNotSupport": "That actual explanatory ability changes. Every dependent measure is a self-rating and no experiment tested real knowledge. The authors also note participants were presumably heavier internet users than average.",
      "terms": [
        "the Google effect",
        "cognitive offloading",
        "metacognition"
      ],
      "relatedPages": [
        "/research/should-i-let-ai-summarise-everything-i-read",
        "/research/do-i-still-need-to-remember-things",
        "/research/ai-and-human-judgement"
      ]
    },
    {
      "id": "storm-stone-2015",
      "citeAs": "https://thesuperskills.com/research/evidence#storm-stone-2015",
      "section": "learning",
      "authors": "Storm, B. C. and Stone, S. M.",
      "year": 2015,
      "title": "Saving-Enhanced Memory: The Benefits of Saving on the Learning and Remembering of New Information",
      "publication": "Psychological Science, 26(2), 182-188. DOI 10.1177/0956797614559285",
      "url": "https://journals.sagepub.com/doi/10.1177/0956797614559285",
      "grade": "peer-reviewed",
      "method": "Three experiments on the effect of saving a digital file on memory for subsequently studied material. GRADED FROM THE PUBLISHED ABSTRACT ONLY: the full text is paywalled and could not be read at the primary source, so no sample sizes or statistics are recorded here.",
      "finding": "Saving one file before studying a new file significantly improved memory for the contents of the new file. The effect was not observed when the saving process was deemed unreliable, or when the contents of the to-be-saved file were not substantial enough to interfere with memory for the new file.",
      "supports": "That cognitive offloading can improve rather than degrade subsequent memory, and that the benefit depends on the external store being trusted.",
      "doesNotSupport": "Anything quantified. This entry rests on the abstract alone, which is a weaker basis than every other entry in this base and is stated as such. It also predates generative AI and concerns file saving rather than a system that can fabricate its own contents.",
      "terms": [
        "cognitive offloading",
        "memory"
      ],
      "relatedPages": [
        "/research/do-i-still-need-to-remember-things"
      ]
    },
    {
      "id": "commonsense-teens-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#commonsense-teens-2025",
      "section": "institutional",
      "authors": "Robb, M. B. and Mann, S. (Common Sense Media)",
      "year": 2025,
      "title": "Talk, Trust, and Trade-Offs: How and Why Teens Use AI Companions",
      "publication": "Common Sense Media, San Francisco, 16 July 2025. Fieldwork by NORC at the University of Chicago",
      "url": "https://www.commonsensemedia.org/research/talk-trust-and-trade-offs-how-and-why-teens-use-ai-companions",
      "grade": "institutional-survey",
      "method": "Survey of 1,060 US teens aged 13-17, interviewed 30 April to 14 May 2025, combining 719 probability interviews from NORC's AmeriSpeak Teen panel with 341 nonprobability interviews from Prodege, raked to February 2024 Current Population Survey totals. Margin of error plus or minus 4.2 percentage points. Cumulative response rate for the probability component 10.3 per cent.",
      "finding": "72 per cent have used an AI companion at least once and 52 per cent at least a few times a month. Against that: 80 per cent of users spend more time with real friends (68 per cent much more) and 6 per cent more time with AI; 67 per cent find AI conversations less satisfying than human ones; 50 per cent distrust the advice; 74 per cent have never shared personal information; 66 per cent have never felt uncomfortable; 9 per cent regard an AI as a friend or best friend; 46 per cent describe them as tools or programs.",
      "supports": "That conversational AI use is near-universal among US teenagers and that most of it is pragmatic rather than relational, on the report's own reading.",
      "doesNotSupport": "That 72 per cent use companion products. The definition given to respondents explicitly included using ChatGPT or Claude as companions, and the report's own limitations concede respondents may have conflated general AI use with companion use, potentially inflating usage statistics. It is cross-sectional and supports no causal claim, though the press release makes one. The published toplines and the report body also disagree on the social-skills transfer figure, giving 33 per cent and 39 per cent respectively.",
      "terms": [
        "teenagers",
        "AI companions",
        "adolescents"
      ],
      "relatedPages": [
        "/research/how-much-should-teenagers-use-ai"
      ]
    },
    {
      "id": "internetmatters-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#internetmatters-2025",
      "section": "institutional",
      "authors": "Internet Matters",
      "year": 2025,
      "title": "Me, myself and AI: Understanding and safeguarding children's use of AI chatbots",
      "publication": "Internet Matters, July 2025",
      "url": "https://www.internetmatters.org/wp-content/uploads/2025/07/Me-Myself-AI-Report.pdf",
      "grade": "institutional-survey",
      "method": "Mixed methods, March to July 2025. Survey of a representative sample of 1,000 UK children aged 9-17 and 2,000 parents of children aged 3-17, fielded April to May 2025; four focus groups with 27 children aged 13-17; 17 days of user testing across ChatGPT, Snapchat My AI and character.ai using two fictional child profiles; four expert interviews. Children are classified as vulnerable if they have an Education, Health and Care Plan, receive SEN support, or have a physical or mental health condition requiring professional help.",
      "finding": "64 per cent of children aged 9-17 have used an AI chatbot (ChatGPT 43 per cent, Google Gemini 32 per cent, Snapchat My AI 31 per cent). Among users the reasons are schoolwork 42 per cent, information 40 per cent, curiosity 40 per cent, chatting 24 per cent, advice 23 per cent, fun 18 per cent, wanting a friend 6 per cent, emotional help or therapy 3 per cent. 35 per cent say it feels like talking to a friend and 12 per cent say they use one because they have no one else to speak to, rising to 50 per cent and 23 per cent among vulnerable children, who are also nearly three times as likely to use companion-style products (17 against 6 per cent).",
      "supports": "That UK chatbot use among children is majority behaviour, dominated by schoolwork and information, and that companionship use concentrates in an identified vulnerable minority.",
      "doesNotSupport": "The precision of the vulnerable-group figures. Two charts are described as sharing the same base, children who have used at least one chatbot, but one uses 133 and 499 respondents and the other 188 and 802; the report does not flag this or publish significance testing, and its limitations section addresses the user testing only. The press release also substitutes a 42 per cent all-ages figure for the report's 47 per cent figure for 15-17 year olds. Nothing here is causal.",
      "terms": [
        "teenagers",
        "children",
        "AI companions"
      ],
      "relatedPages": [
        "/research/how-much-should-teenagers-use-ai"
      ]
    },
    {
      "id": "rohde-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#rohde-2026",
      "section": "work",
      "authors": "Rohde, W. (AiSuNe Foundation)",
      "year": 2026,
      "title": "Short-Term Gain, Long-Term Fragility: AI Labor Substitution and the Erosion of Sustainable Capability",
      "publication": "SSRN abstract 6577818, written 20 April 2026, revised 27 April 2026; also arXiv 2605.27399",
      "url": "https://papers.ssrn.com/sol3/papers.cfm?abstract_id=6577818",
      "grade": "working-paper",
      "method": "Sole-authored conceptual synthesis, 19 pages, no new empirical data. The author's own words: \"This paper is a conceptual synthesis rather than a new empirical study\" and \"the evidentiary strategy is selective and scoped\". Preprint, not peer reviewed, no journal reference. Dates verified at the SSRN record on 28 August 2026; the arXiv abstract page could not be fetched, so the arXiv submission date remains unconfirmed and is not asserted.",
      "finding": "Develops a mechanism of capability masking followed by capability erosion: AI output creates a persuasive appearance that organisational capability has been replaced while dependence on skilled human labour remains, supporting hiring restraint and deferred structural reform while costs accumulate. Frames the result as a stack of deferred obligations: technical debt in artifacts and systems, capability debt in the human layer that maintains them, and institutional debt in the wider structures that reproduce skill and resilience.",
      "supports": "That the capability-erosion argument has been reached independently, from software engineering and political economy rather than from organisational research, and that the masking-before-erosion sequence is not unique to this estate's account.",
      "doesNotSupport": "Anything measured. It is a preprint that argues from other people's empirical work rather than presenting its own, and its societal-scale claims about fragility and concentration of power are inference rather than finding. It is also the reason no claim of first use is made for the term capability debt on this site. Checked at source on 28 August 2026: Rohde does not claim to have coined capability debt either. The paper contains no claiming language for it, and what he does claim is a mechanism, \"to identify and formalize a mechanism of capability masking and capability erosion\".",
      "terms": [
        "capability debt",
        "deskilling",
        "labour substitution"
      ],
      "relatedPages": [
        "/research/capability-debt",
        "/research/cognitive-debt-and-capability-debt"
      ]
    },
    {
      "id": "metr-2026-update",
      "citeAs": "https://thesuperskills.com/research/evidence#metr-2026-update",
      "section": "collaboration",
      "authors": "Becker, J., Rush, N., Cunningham, T., Rein, D. and Mahamud, K. (METR)",
      "year": 2026,
      "title": "We are Changing our Developer Productivity Experiment Design",
      "publication": "METR, 24 February 2026",
      "url": "https://metr.org/blog/2026-02-24-uplift-update/",
      "grade": "working-paper",
      "method": "Second randomised task-level study, begun August 2025: 57 developers (10 returning from the original study, 47 newly recruited), 143 repositories, more than 800 tasks, paid 50 dollars an hour against 150 in the original. Accompanied by participant surveys and interviews.",
      "finding": "METR state the data gives an unreliable signal of the current productivity effect of AI tools, because 30 to 50 per cent of developers reported declining to submit tasks they did not want to do without AI, and an increased share declined to take part at all. Raw results now point the other way: an estimated speedup of -18 per cent for returning developers (CI -38 to +9) and -4 per cent for new recruits (CI -15 to +9), against the original +19 per cent slowdown (CI +2 to +39). They believe developers are likely more sped up in early 2026 than in early 2025, while stating their own data is only very weak evidence for the size of that change.",
      "supports": "That the 19 per cent slowdown belongs to early 2025 and should not be quoted as the current effect. It also demonstrates a measurement problem that will worsen: as adoption rises, the people most helped by AI are the ones most likely to select themselves out of any study that asks them to work without it.",
      "doesNotSupport": "That AI now speeds developers up by a specific amount. Every confidence interval here crosses zero, and METR say so. It does not retract the original study, whose perception-gap finding, that participants estimated a 20 per cent speed-up while measured slower, is untouched.",
      "terms": [
        "productivity",
        "self-report",
        "software development",
        "selection effects"
      ],
      "relatedPages": [
        "/research/what-is-the-metr-study",
        "/research/usage-theatre",
        "/research/the-best-writing-on-ai",
        "/research/the-ai-reports-worth-reading",
        "/research/ai-and-human-judgement",
        "/research/what-becomes-more-valuable-as-ai-gets-cheaper",
        "/research/the-mid-career-squeeze",
        "/research/how-do-you-measure-ai-adoption-properly",
        "/research/how-will-ai-change-consulting",
        "/research/the-unclaimed-hour",
        "/research/what-is-the-substitution-myth",
        "/research/does-ai-actually-make-people-more-productive",
        "/research/will-ai-replace-programmers"
      ]
    },
    {
      "id": "metr-survey-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#metr-survey-2026",
      "section": "collaboration",
      "authors": "Becker, J. (METR)",
      "year": 2026,
      "title": "Measuring the Self-Reported Impact of Early-2026 AI on Technical Worker Productivity",
      "publication": "METR, 11 May 2026",
      "url": "https://metr.org/blog/2026-05-11-ai-usage-survey/",
      "grade": "institutional-survey",
      "method": "Survey of 349 technical workers fielded February to April 2026: 87 software engineers, 71 researchers, 129 academics and PhD students, 48 founders and managers. Convenience sample sourced from GitHub, academic directories, METR and staff networks, and X. Email response rate around 2 per cent. Roughly 70 per cent of participants paid, averaging 200 dollars. Respondents averaged 12 years programming, 19 months using AI for programming and 7 months using agentic coding tools; 50 per cent regularly use Claude Code. Ten respondents removed for inconsistent answers. The design separates value produced from speed, on the argument that speed overstates value when AI changes which tasks a person takes on.",
      "finding": "Median self-reported change in the value of work was between 1.4 and 2 times; median self-reported speed change was 3 times. Asked the same question about different years, respondents put themselves at 1.3 times in March 2025, 2 times in March 2026, and forecast 2.5 times for March 2027. METR's own staff gave the lowest value-change answers of any subgroup studied, which the authors suggest reflects those staff having the perception-gap finding in mind. METR restate the size of that gap here: participants in the early-2025 trial overestimated AI's effect on their time by 40 percentage points on average.",
      "supports": "That the value-versus-speed distinction is large and measurable in self-report, with speed running roughly double value on the same respondents. It is also the clearest statement by the authors of the 19 per cent trial of how big the perception error was.",
      "doesNotSupport": "Any actual productivity effect. Every number here is self-reported, the sample is a convenience sample with a 2 per cent response rate and heavy selection towards enthusiastic adopters, and METR say so. The staff subgroup result is an observation on a small non-random group and not a designed test of whether knowing about the perception gap changes an answer.",
      "terms": [
        "productivity",
        "self-report",
        "perception gap",
        "value versus speed"
      ],
      "relatedPages": [
        "/research/what-is-the-metr-study",
        "/research/usage-theatre",
        "/research/how-do-you-measure-ai-adoption-properly",
        "/research/the-unclaimed-hour"
      ]
    },
    {
      "id": "ft-junior-consultants-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#ft-junior-consultants-2026",
      "section": "professions",
      "authors": "Kissin, E. (Financial Times)",
      "year": 2026,
      "title": "Junior consultants called back to office as AI increases need for human skills",
      "publication": "Financial Times, 27 August 2026",
      "url": "https://www.ft.com/content/7fd9c234-a92b-4ab2-ba1f-969cf9a23f52",
      "grade": "compiled-review",
      "method": "News reporting. On-the-record interviews with named executives at EY, KPMG, Accenture and Azets, with reference to policies at Deloitte, PwC, BCG, Microsoft and JPMorgan. Not a study, and no measurement of skill or outcome.",
      "finding": "Consulting leaders report that AI has raised the value of interpersonal skills and are considering requiring junior staff in the office more often to develop them. EY's UK head of consulting, Sayeh Ghanbari, is quoted saying firms will \"have to reduce flexibility, but in order to help the human skills\", and that firms dropped training in empathy, storytelling and leadership during the remote-working period while prioritising AI and technical skills. KPMG describes reinventing in-person training; BCG is expanding office social activities; Azets has lowered its degree requirement and is encouraging four days a week. Deloitte and PwC began extra coaching for their youngest UK recruits in 2023 after finding weaker teamwork and communication than earlier cohorts.",
      "supports": "That the apprenticeship-erosion argument is now being acted on by the largest professional services firms, named and on the record. It is strong evidence of institutional belief and of policy change.",
      "doesNotSupport": "That AI caused the deficit, or that office attendance repairs it. No measurement appears anywhere in the reporting, the cohort effects described in 2023 are attributed to pandemic lockdowns rather than to AI, and EY as a firm restated its existing flexibility policy alongside its executive's comments. Testimony from interested parties is not evidence of a mechanism.",
      "terms": [
        "missing rungs",
        "apprenticeship",
        "early careers",
        "human skills"
      ],
      "relatedPages": [
        "/research/how-will-ai-change-consulting",
        "/research/missing-rungs"
      ]
    },
    {
      "id": "stankovic-2025-comment",
      "citeAs": "https://thesuperskills.com/research/evidence#stankovic-2025-comment",
      "section": "judgement",
      "authors": "Stanković, M., Hirche, E., Kollatzsch, S. and Doetsch, J.N.",
      "year": 2025,
      "title": "Comment on: Your Brain on ChatGPT: Accumulation of Cognitive Debt When Using an AI Assistant for Essay Writing Tasks",
      "publication": "arXiv 2601.00856, 29 December 2025",
      "url": "https://arxiv.org/abs/2601.00856",
      "grade": "working-paper",
      "method": "Methodological critique of Kosmyna et al. (arXiv 2506.08872) by researchers at the University of Vienna and TU Dresden. Includes an a priori power analysis using G*Power. Itself a preprint, not peer reviewed.",
      "finding": "Argues the MIT cognitive debt study is underpowered: a repeated-measures design at f=0.25, alpha=.05, power=.95 would require approximately 159 participants against the 54 used, with some figures interpreted from subsamples of two to four essays. The strongest objection concerns the construct itself: the search engine group relied on an external tool yet showed no impairment, with the search-engine to brain-only comparison returning p=1, which the authors say contrasts with the interpretation that task delegation increases cognitive debt. Also documents reporting inconsistencies, an unexplained 55 versus 54 participant discrepancy, and unclear FDR correction levels.",
      "supports": "That the most widely cited term in this area rests on a contested pilot, and that the contest is on the record and specific rather than rhetorical.",
      "doesNotSupport": "That the MIT findings are wrong. It is a critique offered to improve a manuscript for peer review, it is itself unreviewed, and it does not present competing data.",
      "terms": [
        "cognitive debt",
        "replication",
        "statistical power",
        "critique"
      ],
      "relatedPages": [
        "/research/cognitive-debt-and-capability-debt",
        "/research/ai-and-human-judgement"
      ]
    },
    {
      "id": "huemmer-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#huemmer-2026",
      "section": "judgement",
      "authors": "Huemmer, M., Durner, F., Shyiramunda, T. and Cummings-Koether, M.J.",
      "year": 2026,
      "title": "AI, Metacognition, and the Verification Bottleneck: A Three-Wave Longitudinal Study of Human Problem-Solving",
      "publication": "arXiv 2601.17055, 21 January 2026",
      "url": "https://arxiv.org/abs/2601.17055",
      "grade": "working-paper",
      "method": "Three-wave longitudinal pilot over six months in an academic setting, convenience sample. Exact sample size is not stated in the abstract. Preprint, no journal reference, no control condition.",
      "finding": "Daily AI use rose from 52.4 to 95.7 per cent across the waves. Participants relied most heavily on AI for difficult tasks, 73.9 per cent, while showing declining verification confidence at 68.1 per cent and accuracy of 47.8 per cent on complex tasks. Objective performance fell across problem difficulty from 95.2 to 81.0 to 66.7 to 47.8 per cent, with belief-performance gaps widening to 34.6 percentage points. The authors describe verification rather than solution generation becoming the bottleneck.",
      "supports": "That the verification dimension is measurable and that confidence and accuracy can diverge sharply as task difficulty rises. Useful as a design precedent for measuring verification quality.",
      "doesNotSupport": "Any generalisable effect. The authors list their own limitations in the abstract: convenience sampling from a single academic cohort, self-report bias, no control condition, mathematical problems only, and a timeframe too short for skill trajectories. They state causal validation requires randomised trials.",
      "terms": [
        "verification",
        "metacognition",
        "overconfidence",
        "longitudinal"
      ],
      "relatedPages": [
        "/research/cognitive-debt-and-capability-debt",
        "/research/the-verifiers-discount",
        "/research/ai-and-human-judgement"
      ]
    },
    {
      "id": "sankaranarayanan-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#sankaranarayanan-2026",
      "section": "learning",
      "authors": "Sankaranarayanan, S.",
      "year": 2026,
      "title": "Mitigating \"Epistemic Debt\" in Generative AI-Scaffolded Novice Programming using Metacognitive Scripts",
      "publication": "arXiv 2602.20206, 22 February 2026, revised 31 March 2026",
      "url": "https://arxiv.org/abs/2602.20206",
      "grade": "working-paper",
      "method": "Between-subjects experiment, 78 participants recruited via Prolific and UserInterviews.com, using a custom Cursor IDE plugin backed by Claude 3.5 Sonnet. Three conditions: manual control, unrestricted AI, and scaffolded AI. Followed by a 30-minute AI-blackout maintenance task. Preprint, not peer reviewed.",
      "finding": "Both AI groups outperformed the manual control on functional utility (p < .001) and did not differ from each other (p = .64). On the subsequent AI-blackout maintenance task, unrestricted AI users failed at 77 per cent against 39 per cent for the scaffolded group. The author describes \"fragile experts\": developers whose high functional utility masks critically low corrective competence.",
      "supports": "That the gap between producing output and being able to repair it can be created inside a single session, and that interface design changes the size of that gap. The scaffolded condition roughly halved the failure rate.",
      "doesNotSupport": "A durable effect on capability. It is a single session with a single blackout task, in novice programming, with no longitudinal follow-up. The author puts \"epistemic debt\" in quotation marks in his own title and builds explicitly on Kirschner, so it is not offered as a new construct.",
      "terms": [
        "epistemic debt",
        "scaffolding",
        "cognitive offloading",
        "novice programming"
      ],
      "relatedPages": [
        "/research/cognitive-debt-and-capability-debt",
        "/research/will-ai-replace-programmers"
      ]
    },
    {
      "id": "jadhav-danve-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#jadhav-danve-2026",
      "section": "work",
      "authors": "Jadhav, R. and Danve, J.",
      "year": 2026,
      "title": "The AI Skills Shift: Mapping Skill Obsolescence, Emergence, and Transition Pathways in the LLM Era",
      "publication": "arXiv 2604.06906, 8 April 2026",
      "url": "https://arxiv.org/abs/2604.06906",
      "grade": "working-paper",
      "method": "Benchmarking of four frontier models (LLaMA 3.3 70B, Mistral Large, Qwen 2.5 72B, Gemini 2.5 Flash) across 263 text-based tasks covering all 35 skills in the US Department of Labor O*NET taxonomy, 1,052 model calls. Cross-referenced against the Anthropic Economic Index. Measures models, not people. Preprint.",
      "finding": "Introduces a Skill Automation Feasibility Index. Mathematics scores 73.2 and programming 71.8 for automation feasibility; active listening scores 42.2 and reading comprehension 45.5. All four models converge to similar skill profiles within a 3.6-point spread. Reports that 78.7 per cent of observed AI interactions are augmentation rather than automation, and a \"capability-demand inversion\" in which the skills most demanded in AI-exposed jobs are those the models perform least well at.",
      "supports": "That measured model capability and stated employer demand point in different directions, which is a useful counterweight to displacement forecasts built on exposure scores alone.",
      "doesNotSupport": "What happens to human capability. The authors state their index \"measures LLM performance on text-based representations of skills, not full occupational execution\". No humans were studied.",
      "terms": [
        "skills taxonomy",
        "automation feasibility",
        "O*NET",
        "augmentation"
      ],
      "relatedPages": [
        "/research/cognitive-debt-and-capability-debt"
      ]
    },
    {
      "id": "sdaia-ethics-2023",
      "citeAs": "https://thesuperskills.com/research/evidence#sdaia-ethics-2023",
      "section": "international",
      "authors": "Saudi Data and AI Authority (SDAIA)",
      "year": 2023,
      "title": "AI Ethics Principles, Version 1.0",
      "publication": "SDAIA, September 2023",
      "url": "https://dgp.sdaia.gov.sa/wps/wcm/connect/4c56ed1c-1b82-447d-ac29-638f5f99c12e/ai-principles-EN.pdf",
      "grade": "institutional-survey",
      "method": "National AI ethics guidance, seven principles with an assessment checklist annexe covering the AI system lifecycle. Non-binding guidance rather than statute. Version 1.0 read in full at source; whether a later version exists could not be checked because SDAIA's main host rejects automated requests.",
      "finding": "The checklist annexe puts this question to designers at the plan and design stage: \"Does your AI system design prevent overconfidence in or overreliance on the AI system with necessary human intervention mechanisms?\" It also asks whether human oversight processes carry defined KPIs and assigned responsibility. The operative text states that decisions which are irreversible or life-and-death \"should trigger human oversight and final determination\", and rules out social scoring and mass surveillance.",
      "supports": "That over-reliance on AI has been named as a design defect to be engineered against in a national governance instrument. On the evidence gathered here it is the only Gulf instrument that does so.",
      "doesNotSupport": "Any obligation. It is guidance, not law, the over-reliance language sits in a checklist annexe rather than in the principle text, and Saudi Arabia had no binding AI statute as of August 2026, only a draft responsible AI policy out for consultation from 2 April 2026. Version 1.0 dates from September 2023 and may have been superseded.",
      "terms": [
        "human oversight",
        "over-reliance",
        "Saudi Arabia",
        "AI ethics"
      ],
      "relatedPages": [
        "/research/ai-and-work-in-the-gulf"
      ]
    },
    {
      "id": "gastat-ict-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#gastat-ict-2025",
      "section": "international",
      "authors": "General Authority for Statistics (GASTAT), Saudi Arabia",
      "year": 2025,
      "title": "Establishments ICT Access and Usage Statistics 2025",
      "publication": "GASTAT, 2025",
      "url": "https://www.stats.gov.sa/documents/d/guest/establishments-ict-access-and-usage-statistics-2025-en-pdf",
      "grade": "institutional-survey",
      "method": "National statistical survey of establishments, methodology stated as aligned with UNCTAD international standards. Enterprise-side only.",
      "finding": "33.1 per cent of establishments use artificial intelligence technologies, a growth of 20.0 per cent against 2024. By sector: information and communication 61.1 per cent, financial and insurance 52.9 per cent, education 51.0 per cent, manufacturing 30.7 per cent, wholesale and retail 30.2 per cent. The companion household and individuals survey for the same year contains no AI indicator at all.",
      "supports": "That one Gulf state measures enterprise AI adoption with a published, standards-aligned method. It is the only official national AI adoption statistic found across Saudi Arabia, the UAE and Qatar.",
      "doesNotSupport": "Anything about citizens, or about employment. Saudi Arabia measures enterprise adoption but not individual use, and publishes no AI employment series. Adoption is also self-reported use of a technology category, not a measure of capability or of value obtained.",
      "terms": [
        "adoption",
        "official statistics",
        "Saudi Arabia",
        "international"
      ],
      "relatedPages": [
        "/research/ai-and-work-in-the-gulf",
        "/research/ai-and-work-by-country"
      ]
    },
    {
      "id": "uae-charter-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#uae-charter-2024",
      "section": "international",
      "authors": "Government of the United Arab Emirates",
      "year": 2024,
      "title": "The UAE Charter for the Development and Use of Artificial Intelligence",
      "publication": "UAE Legislation portal, issued 10 June 2024",
      "url": "https://uaelegislation.gov.ae/en/policy/details/the-uae-charter-for-the-development-and-use-of-artificial-intelligence",
      "grade": "institutional-survey",
      "method": "National charter of twelve principles. Statement of principle with no duty-holder, no enforcement mechanism and no competence requirement.",
      "finding": "Principle 6, Human Oversight, \"emphasizes the irreplaceable value of human judgment and human oversight over AI, aligning with ethical values and social standards to correct any errors or biases that may arise\". The charter is silent on deskilling, over-reliance and any obligation to train or assess the humans doing the overseeing.",
      "supports": "That human oversight is stated as a national principle in the UAE.",
      "doesNotSupport": "That it is operational. There is no named duty-holder, no enforcement, no competence standard and no test of whether oversight is real. The UAE had no federal AI statute as of August 2026.",
      "terms": [
        "human oversight",
        "United Arab Emirates",
        "governance"
      ],
      "relatedPages": [
        "/research/ai-and-work-in-the-gulf"
      ]
    },
    {
      "id": "difc-reg10-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#difc-reg10-2024",
      "section": "institutional",
      "authors": "DIFC Commissioner of Data Protection",
      "year": 2024,
      "title": "Regulation 10 on Personal Data Processed through Autonomous and Semi-Autonomous Systems",
      "publication": "Dubai International Financial Centre, DIFC-DP-GL-23 Rev.03, updated 27 August 2024; regulation enacted September 2023",
      "url": "https://www.difc.com/",
      "grade": "institutional-survey",
      "method": "Binding regulation within the DIFC free zone, with accompanying guidance. Applies to personal data processing by autonomous and semi-autonomous systems, not to AI generally.",
      "finding": "States that \"human-defined processing purposes must always prevail in Systems development and use\". Draws an explicit analogy between an autonomous system and an employee: where a system operates for the benefit of its deployer, \"its position is substantially similar to that of an employee within the Deployer organization, and the Deployer should be therefore liable for its actions in the same way it may be liable for an employee's actions\". Creates a named Autonomous Systems Officer performing a function similar to a data protection officer.",
      "supports": "That the employee analogy for accountability, and a named human role responsible for it, exist in a binding instrument somewhere in the Gulf.",
      "doesNotSupport": "Anything about capability or competence. It is a data protection regulation confined to one free zone, it addresses liability rather than skill, and it imposes no requirement that the responsible human be able to do the work being supervised.",
      "terms": [
        "accountability",
        "human oversight",
        "United Arab Emirates",
        "regulation"
      ],
      "relatedPages": [
        "/research/ai-and-work-in-the-gulf",
        "/research/who-owns-verification-when-ai-does-the-work"
      ]
    },
    {
      "id": "qatar-mcit-ai-principles",
      "citeAs": "https://thesuperskills.com/research/evidence#qatar-mcit-ai-principles",
      "section": "international",
      "authors": "Ministry of Communications and Information Technology, Qatar",
      "year": 2026,
      "title": "Artificial Intelligence in Qatar: Principles and Guidelines for Ethical Development and Deployment",
      "publication": "MCIT Qatar, undated; read at source 28 August 2026",
      "url": "https://www.mcit.gov.qa/en/",
      "grade": "compiled-review",
      "method": "National AI ethics guidance. Read in full at source, but with a provenance defect worth recording: the document carries no publication date, no version number and no reference number, and is not retrievable from the ministry's own website. It states that it \"is legally non-binding, and adherence to it is voluntary\".",
      "finding": "Principle 8, assign ultimate accountability to humans, states that \"AI systems should not be able to autonomously make decisions of significant consequence\" and should \"provide users with the ability to appeal or override decisions that have a substantial impact on individuals or society\". The phrase human in the loop appears nowhere in the document, and Principle 7, develop a human-centered approach, concerns cultural values, feedback, diverse teams and accessibility rather than oversight.",
      "supports": "That Qatar requires humans to retain control of consequential decisions, in guidance.",
      "doesNotSupport": "Any competence duty. Qatar imposes no obligation that the humans exercising control be trained, assessed or kept current, and the document contains no reference to deskilling, over-reliance or automation bias. Widely circulated dates of May 2024 or 2025 for this document are unsourced: it carries no date at all.",
      "terms": [
        "human oversight",
        "accountability",
        "Qatar",
        "governance"
      ],
      "relatedPages": [
        "/research/ai-and-work-in-the-gulf"
      ]
    },
    {
      "id": "imda-agentic-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#imda-agentic-2026",
      "section": "institutional",
      "authors": "Infocomm Media Development Authority (IMDA), Singapore",
      "year": 2026,
      "title": "Model AI Governance Framework for Agentic AI, Version 1.0",
      "publication": "IMDA, published 22 January 2026, launched at Davos",
      "url": "https://www.imda.gov.sg/-/media/imda/files/about/emerging-tech-and-research/artificial-intelligence/mgf-for-agentic-ai.pdf",
      "grade": "institutional-survey",
      "method": "National governance framework for agentic AI, read in full at source. Builds on IMDA's 2020 Model AI Governance Framework. Guidance rather than statute. Law firms report a version 1.5 of 20 May 2026; that could not be confirmed at an official page, so version 1.0 is cited.",
      "finding": "Names deskilling as a risk of agentic deployment, in the terms this research uses. Section 2.4.3: \"As agents take over entry level tasks, which typically serve as the training ground for new staff, this could lead to loss of basic operational knowledge for the users. Organisations should identify core capabilities of each job and provide sufficient training and work exposure so that users retain foundational skills.\" Section 2.4 warns of \"the potential loss of trade craft\" and requires \"sufficient training... to ensure that humans retain core skills\". It also names automation bias directly, requires that overseers be trained to identify common failure modes, and requires that the effectiveness of human oversight itself be audited. It concedes that \"continuous human oversight over all agent workflows becomes impractical at scale\".",
      "supports": "That a national government has written the removal of entry-level work, and the consequent loss of the training ground for junior staff, into an operative AI governance framework. It is the closest external corroboration of the missing rungs argument found in any policy document.",
      "doesNotSupport": "Anything measured. It is guidance, not law, and it states a risk and a duty to train rather than evidence that deskilling has occurred. It also sets no threshold for what counts as retaining core skills, and no test of whether the training works.",
      "terms": [
        "deskilling",
        "missing rungs",
        "human oversight",
        "automation bias",
        "Singapore",
        "agents"
      ],
      "relatedPages": [
        "/research/ai-and-work-in-asia",
        "/research/missing-rungs"
      ]
    },
    {
      "id": "imda-genai-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#imda-genai-2024",
      "section": "institutional",
      "authors": "AI Verify Foundation and IMDA, Singapore",
      "year": 2024,
      "title": "Model AI Governance Framework for Generative AI",
      "publication": "IMDA and AI Verify Foundation, 30 May 2024",
      "url": "https://aiverifyfoundation.sg/resources/mgf-gen-ai/",
      "grade": "institutional-survey",
      "method": "National governance framework for generative AI, nine dimensions. Both published PDF versions read in full and searched at source.",
      "finding": "Contains zero occurrences of \"human oversight\", \"human-in-the-loop\", \"over-reliance\", \"automation bias\" or \"deskill\". Human oversight is not among its nine dimensions. The nearest it comes is a note that \"core skills such as creativity, critical thinking and complex problem-solving are important to helping people harness AI effectively\", and its only use of \"competency\" concerns third-party auditors rather than the human overseer.",
      "supports": "The value here is the confirmed absence, and the trajectory it establishes. In twenty months the same issuing body went from a framework with no oversight language at all to one built around it that also names deskilling. That shift is documented and quotable.",
      "doesNotSupport": "That Singapore was indifferent to oversight in 2024; the 2020 framework it builds on was not read at source and may carry such language. It establishes what the generative AI framework does not say, not what the whole regime did not say.",
      "terms": [
        "human oversight",
        "Singapore",
        "governance",
        "absence"
      ],
      "relatedPages": [
        "/research/ai-and-work-in-asia"
      ]
    },
    {
      "id": "mom-labour-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#mom-labour-2026",
      "section": "international",
      "authors": "Manpower Research and Statistics Department, Ministry of Manpower, Singapore",
      "year": 2026,
      "title": "Labour Market Report, First Quarter 2026",
      "publication": "Ministry of Manpower, Singapore, released 15 June 2026",
      "url": "https://stats.mom.gov.sg/Pages/Labour-Market-Report-1Q-2026.aspx",
      "grade": "institutional-survey",
      "method": "National firm survey by the labour ministry's statistics department. Note the companion press release omits the AI statistic entirely; it appears only in the full report.",
      "finding": "28.5 per cent of firms adopted AI in 2026, highest in information and communications at 74.1 per cent, professional services at 57.5 per cent and financial and insurance services at 56.4 per cent. Only 6.2 per cent reported AI-related reductions in headcount or hiring, against 18.9 per cent reporting redesign of job functions. The ministry's own reading: \"AI is currently having a greater impact on job redesign and work processes than on broad-based job displacement.\"",
      "supports": "That a government labour ministry measuring this finds job redesign running roughly three times ahead of headcount reduction. It is a useful corrective to displacement forecasting.",
      "doesNotSupport": "What happens to capability. Redesign is not neutral: the question this research asks is which tasks the redesign removes, and a firm survey of headcount cannot answer it.",
      "terms": [
        "adoption",
        "job redesign",
        "displacement",
        "Singapore",
        "official statistics"
      ],
      "relatedPages": [
        "/research/ai-and-work-in-asia"
      ]
    },
    {
      "id": "hk-censtatd-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#hk-censtatd-2026",
      "section": "international",
      "authors": "Census and Statistics Department, Hong Kong SAR",
      "year": 2026,
      "title": "Report on the Survey on Information Technology Usage and Penetration in the Business Sector, 2025 Edition",
      "publication": "Census and Statistics Department, Hong Kong, released 27 February 2026",
      "url": "https://www.censtatd.gov.hk/",
      "grade": "institutional-survey",
      "method": "National business survey, fieldwork March to December 2025. Full report and all thirty tables read at source, together with three companion household and ICT publications.",
      "finding": "Artificial intelligence appears nowhere in the report: not in the tables, not in the explanatory notes, not in the definitions. The survey's ICT categories are cloud computing at 98.1 per cent, QR codes at 37.6 per cent, RFID at 20.3 per cent, internet of things at 7.4 per cent and augmented or virtual reality at 1.5 per cent. Hong Kong's statistical office measures AR and VR adoption and does not ask about AI at all. The 2023 edition also had no AI category, and no plan to add one has been announced.",
      "supports": "That Hong Kong publishes no official statistic on AI adoption or AI-related employment, which means every circulating Hong Kong AI adoption figure comes from a non-government survey with a self-selected sample.",
      "doesNotSupport": "That AI adoption in Hong Kong is low. It establishes that it is unmeasured by the government, which is a different and in some ways more useful fact.",
      "terms": [
        "official statistics",
        "Hong Kong",
        "measurement",
        "absence"
      ],
      "relatedPages": [
        "/research/ai-and-work-in-asia"
      ]
    },
    {
      "id": "ey-work-reimagined-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#ey-work-reimagined-2025",
      "section": "institutional",
      "authors": "EY",
      "year": 2025,
      "title": "Work Reimagined Survey 2025",
      "publication": "EY Global, November 2025; UK cut published December 2025",
      "url": "https://www.ey.com/en_gl/newsroom/2025/11/ey-survey-reveals-companies-are-missing-out-on-up-to-40-percent-of-ai-productivity-gains-due-to-gaps-in-talent-strategy",
      "grade": "institutional-survey",
      "method": "Employer and employee survey, 15,000 employees and 1,500 employers across 29 countries. Self-reported, cross-sectional, non-random sample. Published through a newsroom release rather than a methodology appendix.",
      "finding": "88 per cent of employees use AI at work, but mostly for basic tasks such as search and summarisation, and only 5 per cent use it in advanced ways that transform how they work. 37 per cent worry that overreliance on AI could erode their skills and expertise, rising to 43 per cent in the UK cut. Organisations pursuing AI gains on weak talent foundations saw productivity gains lag by over 40 per cent.",
      "supports": "That near-universal AI use coexists with very shallow use, and that employees themselves report skill-erosion worry at scale. It also gives a commercial argument for the capability case: the productivity is not collected where the human foundation is weak.",
      "doesNotSupport": "Any measured capability loss. Every figure is self-reported perception at a single point in time, the sample is not random, and the 40 per cent productivity figure is EY's own modelled comparison rather than an experimental result.",
      "terms": [
        "capability debt",
        "AI adoption",
        "deskilling",
        "usage depth"
      ],
      "relatedPages": [
        "/research/capability-debt",
        "/research/ai-workforce-strategy"
      ]
    },
    {
      "id": "bcg-deskilling-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#bcg-deskilling-2026",
      "section": "institutional",
      "authors": "Boston Consulting Group",
      "year": 2026,
      "title": "When Everyone Uses AI, Companies Risk Losing Critical Skills",
      "publication": "BCG, 10 June 2026",
      "url": "https://www.bcg.com/publications/2026/when-everyone-uses-ai-companies-risk-critical-skills",
      "grade": "institutional-survey",
      "method": "Global survey of 70 C-suite leaders and senior executives. Very small base, self-selected, and reporting perception rather than measurement.",
      "finding": "Half of the executives surveyed report already observing deskilling in their organisations, and more than 60 per cent believe deskilling will pose a material threat to their organisation within the next three to five years.",
      "supports": "That deskilling has reached the point where senior leaders report seeing it themselves, which is a change in the executive agenda rather than in the evidence.",
      "doesNotSupport": "Prevalence. The base is 70 people. Anyone quoting the 50 per cent without the 70 is overstating it, and executive perception is not a measurement of what is happening to anyone's skills.",
      "terms": [
        "deskilling",
        "capability debt",
        "executive perception"
      ],
      "relatedPages": [
        "/research/capability-debt",
        "/research/which-professions-face-the-greatest-deskilling-risk"
      ]
    },
    {
      "id": "tsb-a19p0112-2019",
      "citeAs": "https://thesuperskills.com/research/evidence#tsb-a19p0112-2019",
      "section": "professions",
      "authors": "Transportation Safety Board of Canada",
      "year": 2019,
      "title": "Aviation Investigation Report A19P0112, Seair Seaplanes, Addenbroke Island",
      "publication": "Transportation Safety Board of Canada",
      "url": "https://www.tsb.gc.ca/eng/rapports-reports/aviation/2019/a19p0112/a19p0112.html",
      "grade": "compiled-review",
      "method": "Single-accident investigation report. The passage on skill degradation sits in section 1.8 and reports concerns gathered from air-taxi operators in a separate TSB Safety Issue Investigation, not findings about this accident.",
      "finding": "Surveyed air-taxi operators expressed concern that 'dependence on technology was causal in degradation of basic piloting skills', and numerous operators commented that over-reliance on GPS navigation 'may contribute to the decision to fly into adverse weather conditions'.",
      "supports": "That an industry regulator was recording practitioner-reported skill degradation from automation dependence, in an operational setting, before generative AI existed. It also captures the double bind: the sector's stated problem is too little technology and its observed failure mode is over-reliance on it.",
      "doesNotSupport": "That automation dependence caused this crash. The accident's own findings as to causes cite weather, terrain-alerting ambiguity and fatigue. This is industry survey commentary quoted inside an accident report, and citing it as a causal finding would misrepresent it.",
      "terms": [
        "deskilling",
        "automation dependence",
        "aviation",
        "capability debt"
      ],
      "relatedPages": [
        "/research/capability-debt",
        "/research/what-professions-can-learn-from-aviation"
      ]
    },
    {
      "id": "ntsb-bluecruise-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#ntsb-bluecruise-2026",
      "section": "professions",
      "authors": "National Transportation Safety Board",
      "year": 2026,
      "title": "Highway Investigation Report HIR-26-02, Ford BlueCruise collisions",
      "publication": "National Transportation Safety Board, 31 March 2026",
      "url": "https://www.ntsb.gov/investigations/AccidentReports/Reports/HIR2602.pdf",
      "grade": "compiled-review",
      "method": "Investigation of two fatal crashes in which Ford BlueCruise-equipped vehicles struck stationary vehicles at highway speed: San Antonio, 24 February 2024, and Philadelphia, 3 March 2024.",
      "finding": "Overreliance on the partial automation system appears in the probable cause for both crashes, not merely in discussion: distraction 'stemming from overreliance on the vehicle's hands-free partial automation system' in San Antonio, and 'overreliance on and misuse of' it in Philadelphia. The board recommended that driver monitoring systems detect and warn about 'accumulated short distractions' over a prolonged period.",
      "supports": "That an investigator placed automation over-reliance in the causal chain rather than treating it as context, and that the monitoring system was defeated by ordinary attention behaving ordinarily rather than by anyone circumventing it.",
      "doesNotSupport": "Anything about knowledge work. Driving is a continuous manual-control task with a monitoring system watching the human, which is not the shape of AI-assisted professional judgement.",
      "terms": [
        "automation complacency",
        "over-reliance",
        "capability debt"
      ],
      "relatedPages": [
        "/research/capability-debt",
        "/research/what-is-automation-complacency"
      ]
    },
    {
      "id": "roediger-karpicke-2006",
      "citeAs": "https://thesuperskills.com/research/evidence#roediger-karpicke-2006",
      "section": "learning",
      "authors": "Roediger, H. L. III and Karpicke, J. D.",
      "year": 2006,
      "title": "Test-Enhanced Learning: Taking Memory Tests Improves Long-Term Retention",
      "publication": "Psychological Science, 17(3), 249-255. DOI 10.1111/j.1467-9280.2006.01693.x",
      "url": "https://doi.org/10.1111/j.1467-9280.2006.01693.x",
      "grade": "peer-reviewed",
      "method": "Two experiments, 120 and 180 participants, reading prose passages on general science topics. Restudying compared with being tested, with final recall measured at 5 minutes, 2 days or 1 week.",
      "finding": "The winner reverses with delay. At 5 minutes restudying beat testing, 81 per cent against 75 per cent. At one week testing beat restudying, 56 per cent against 42 per cent. In the second experiment repeated study led at 5 minutes, 83 against 71 per cent, and trailed badly at one week, 40 against 61 per cent.",
      "supports": "That the study method producing the best immediate performance produces the worst durable retention, and that a measurement taken close to the learning will rank the methods in exactly the wrong order.",
      "doesNotSupport": "Anything about AI. It is prose recall in a laboratory, and the transfer to professional judgement is by analogy.",
      "terms": [
        "retrieval practice",
        "testing effect",
        "desirable difficulty",
        "capability debt"
      ],
      "relatedPages": [
        "/research/what-is-retrieval-practice",
        "/research/what-is-desirable-difficulty"
      ]
    },
    {
      "id": "karpicke-blunt-2011",
      "citeAs": "https://thesuperskills.com/research/evidence#karpicke-blunt-2011",
      "section": "learning",
      "authors": "Karpicke, J. D. and Blunt, J. R.",
      "year": 2011,
      "title": "Retrieval Practice Produces More Learning than Elaborative Studying with Concept Mapping",
      "publication": "Science, 331(6018), 772-775. DOI 10.1126/science.1199327",
      "url": "https://doi.org/10.1126/science.1199327",
      "grade": "peer-reviewed",
      "method": "Two experiments, 80 and 120 undergraduates, comparing retrieval practice with elaborative study by concept mapping, tested one week later.",
      "finding": "Retrieval practice scored 0.67 against 0.45 for concept mapping, about a 50 per cent advantage in long-term retention, d = 1.50. In the second experiment 101 of 120 students, 84 per cent, did better after retrieval practice than after elaborative study.",
      "supports": "That effortful recall outperforms a more elaborate and more comfortable-feeling study method, on the same material, in the same students.",
      "doesNotSupport": "That concept mapping is worthless. A published comment by Mintzes and colleagues (Science, 2011, 334(6055), 453) disputes the instructional fidelity of the concept-mapping condition, and anyone citing this should note it.",
      "terms": [
        "retrieval practice",
        "testing effect",
        "illusion of competence"
      ],
      "relatedPages": [
        "/research/what-is-retrieval-practice"
      ]
    },
    {
      "id": "rowland-2014",
      "citeAs": "https://thesuperskills.com/research/evidence#rowland-2014",
      "section": "learning",
      "authors": "Rowland, C. A.",
      "year": 2014,
      "title": "The Effect of Testing Versus Restudy on Retention: A Meta-Analytic Review of the Testing Effect",
      "publication": "Psychological Bulletin, 140(6), 1432-1463. DOI 10.1037/a0037559",
      "url": "https://doi.org/10.1037/a0037559",
      "grade": "peer-reviewed",
      "method": "Meta-analysis, 159 effect sizes from 61 studies published between 1975 and 2013, random-effects model.",
      "finding": "A reliable testing effect, g = 0.50 with a confidence interval of 0.42 to 0.58. With feedback the effect rises to g = 0.73; without feedback it falls to g = 0.39.",
      "supports": "That the testing effect survives aggregation across four decades of studies, at a moderate effect size, and that feedback roughly doubles it.",
      "doesNotSupport": "That the effect size transfers to workplace learning. The constituent studies are overwhelmingly laboratory and classroom work on verbal material.",
      "terms": [
        "retrieval practice",
        "testing effect",
        "meta-analysis"
      ],
      "relatedPages": [
        "/research/what-is-retrieval-practice"
      ]
    },
    {
      "id": "kapur-2008",
      "citeAs": "https://thesuperskills.com/research/evidence#kapur-2008",
      "section": "learning",
      "authors": "Kapur, M.",
      "year": 2008,
      "title": "Productive Failure",
      "publication": "Cognition and Instruction, 26(3), 379-424. DOI 10.1080/07370000802212669",
      "url": "https://doi.org/10.1080/07370000802212669",
      "grade": "peer-reviewed",
      "method": "Randomised comparison, 309 eleventh-grade physics students in India working on Newtonian kinematics. One group solved ill-structured problems in groups before individual well-structured problems; the other solved well-structured problems throughout.",
      "finding": "The group given ill-structured problems struggled visibly and produced poor solutions during the collaborative phase, then outperformed the other group on individual near-transfer and far-transfer measures afterwards.",
      "supports": "That struggle which looks like failure at the time can produce better subsequent transfer than a smooth path through the same material, and that judging a learning design by how well it is going is unreliable.",
      "doesNotSupport": "A precise magnitude. The design, sample and direction of the result are confirmed from the publisher's abstract and the author's own presentation of the study, but the full text sits behind a paywall that could not be read, so no post-test figures are quoted here.",
      "terms": [
        "productive failure",
        "productive struggle",
        "desirable difficulty"
      ],
      "relatedPages": [
        "/research/what-is-productive-struggle"
      ]
    },
    {
      "id": "sinha-kapur-2021",
      "citeAs": "https://thesuperskills.com/research/evidence#sinha-kapur-2021",
      "section": "learning",
      "authors": "Sinha, T. and Kapur, M.",
      "year": 2021,
      "title": "When Problem Solving Followed by Instruction Works: Evidence for Productive Failure",
      "publication": "Review of Educational Research, 91(5), 761-798. DOI 10.3102/00346543211019105",
      "url": "https://doi.org/10.3102/00346543211019105",
      "grade": "peer-reviewed",
      "method": "Meta-analysis of 53 studies and 166 comparisons of problem-solving-before-instruction against instruction-before-problem-solving.",
      "finding": "A moderate effect favouring problem solving first, Hedges g = 0.36 with a confidence interval of 0.20 to 0.51, rising to between 0.37 and 0.58 where the design followed the productive failure principles closely. The effect reverses for second to fifth graders and for domain-general skills, where instruction first wins.",
      "supports": "That letting people struggle before teaching them beats teaching them first, at moderate effect size, for older learners on domain-specific content.",
      "doesNotSupport": "That struggle is universally good. The authors report the reversal for young children and for general skills themselves, and an earlier meta-analysis by Darabi and colleagues rested on only 12 studies.",
      "terms": [
        "productive failure",
        "productive struggle",
        "meta-analysis"
      ],
      "relatedPages": [
        "/research/what-is-productive-struggle"
      ]
    },
    {
      "id": "skills-england-l7-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#skills-england-l7-2026",
      "section": "institutional",
      "authors": "Skills England",
      "year": 2026,
      "title": "Evidence on defunding of level 7 apprenticeships",
      "publication": "Skills England, published 30 April 2026, produced March 2025",
      "url": "https://www.gov.uk/government/publications/skills-england-evidence-on-defunding-of-level-7-apprenticeships/skills-england-evidence-on-defunding-of-level-7-apprenticeships",
      "grade": "institutional-modelling",
      "method": "Government evidence review supporting the decision to withdraw levy funding for level 7 apprenticeships for those aged 22 and over from 1 January 2026. Analysis of DfE apprenticeship statistics by level and age.",
      "finding": "Level 7 apprenticeship starts in 2023/24 were 23,870, of which 65 per cent were aged 25 or over, 34 per cent aged 19 to 24, and 2 per cent under 19. Funding continues for 16 to 21 year olds, for care leavers and those with an education, health and care plan up to 24, and for anyone who started before 1 January 2026.",
      "supports": "That master's-level apprenticeship funding was overwhelmingly reaching people already established in careers rather than young entrants, which is the government's stated rationale. It is the clearest official statement of who higher-level apprenticeships actually served.",
      "doesNotSupport": "That degree apprenticeships in general were cut. Level 6, the undergraduate degree apprenticeship, is unaffected by this decision and its starts were still rising.",
      "terms": [
        "apprenticeships",
        "funding",
        "early careers",
        "policy"
      ],
      "relatedPages": [
        "/research/do-apprenticeships-still-work"
      ]
    },
    {
      "id": "hoc-apprenticeships-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#hoc-apprenticeships-2025",
      "section": "institutional",
      "authors": "Murray, A., House of Commons Library",
      "year": 2025,
      "title": "Apprenticeship statistics for England",
      "publication": "House of Commons Library briefing CBP 06113, 4 December 2025",
      "url": "https://researchbriefings.files.parliament.uk/documents/SN06113/SN06113.pdf",
      "grade": "institutional-survey",
      "method": "Parliamentary briefing compiling Department for Education apprenticeship statistics for England.",
      "finding": "353,500 apprenticeship starts in England in 2024/25, up from 340,000 in 2023/24 and 337,000 in 2022/23. In 2024/25, 51.3 per cent of starts were by apprentices aged 25 or over, 27.5 per cent aged 19 to 24, and 21.2 per cent under 19. The age distribution has remained approximately the same since 2018/19.",
      "supports": "That total apprenticeship starts are rising modestly and that half of all apprenticeships go to people aged 25 and over. The system is not primarily a route for the young and has not been for years.",
      "doesNotSupport": "Anything about quality, completion or whether apprenticeships lead to work. Starts are a count of beginnings.",
      "terms": [
        "apprenticeships",
        "early careers",
        "official statistics"
      ],
      "relatedPages": [
        "/research/do-apprenticeships-still-work"
      ]
    },
    {
      "id": "nao-construction-skills-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#nao-construction-skills-2026",
      "section": "institutional",
      "authors": "National Audit Office",
      "year": 2026,
      "title": "Increasing construction skills",
      "publication": "National Audit Office, 13 July 2026",
      "url": "https://www.nao.org.uk/wp-content/uploads/2026/07/increasing-construction-skills.pdf",
      "grade": "institutional-survey",
      "method": "Audit of the government's construction skills package, including the foundation apprenticeships launched on 1 August 2025.",
      "finding": "By April 2026 only 74 young people had started a construction foundation apprenticeship, against the department's own assumption of 1,000 in 2025-26.",
      "supports": "That the replacement scheme is not filling the gap it was designed for. A shortfall of this size, against the government's own planning assumption, in the sector the policy was built around, is the strongest available evidence that shortening apprenticeships has not by itself restored entry-level training.",
      "doesNotSupport": "That foundation apprenticeships cannot work. It is eight months of data on a scheme launched in August 2025, and low take-up in one sector.",
      "terms": [
        "apprenticeships",
        "foundation apprenticeships",
        "early careers",
        "policy"
      ],
      "relatedPages": [
        "/research/do-apprenticeships-still-work"
      ]
    },
    {
      "id": "bibb-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#bibb-2025",
      "section": "international",
      "authors": "Bundesinstitut fuer Berufsbildung",
      "year": 2025,
      "title": "Angespannte Lage auf dem Ausbildungsmarkt (Tense situation in the training market)",
      "publication": "BIBB press release 40/2025, 10 December 2025",
      "url": "https://www.bibb.de/de/pressemitteilung_215396.php",
      "grade": "institutional-survey",
      "method": "Official German statistics on the dual vocational training system, counting newly concluded training contracts, unfilled training places and unplaced applicants as at 30 September 2025.",
      "finding": "Around 476,000 new dual training contracts in 2025, down 2.1 per cent on 2024 and the second consecutive annual fall. 54,400 training places went unfilled. At the same time around 84,400 young people had not found a training place, up 19.9 per cent and the highest since 2010, with the share of unsuccessful applicants at 15.1 per cent, the highest since the end of the 2009 financial crisis.",
      "supports": "That the world's most admired apprenticeship system is failing to match young people to places in both directions at once: record unplaced applicants alongside tens of thousands of empty places. The bridge is not merely narrowing, it has stopped connecting.",
      "doesNotSupport": "That AI caused it. BIBB attributes the tension to matching problems, regional and occupational mismatch and economic conditions, and makes no AI claim.",
      "terms": [
        "apprenticeships",
        "Germany",
        "dual system",
        "early careers",
        "matching"
      ],
      "relatedPages": [
        "/research/do-apprenticeships-still-work"
      ]
    },
    {
      "id": "ncver-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#ncver-2025",
      "section": "international",
      "authors": "National Centre for Vocational Education Research",
      "year": 2025,
      "title": "Apprentices and trainees 2025",
      "publication": "NCVER, Australia, March and June quarter releases 2025",
      "url": "https://www.ncver.edu.au/research-and-statistics/publications/all-publications/apprentices-and-trainees-2025-march-quarter",
      "grade": "institutional-survey",
      "method": "Official Australian administrative counts of apprentice and trainee training contracts.",
      "finding": "320,830 in-training contracts at 31 March 2025, down 7.9 per cent year on year, with trade down 3.2 per cent and non-trade down 17.9 per cent. By the June quarter the fall was 11.3 per cent, with non-trade down 20.2 per cent. NCVER attributes the non-trade decline partly to the conclusion of the Boosting Apprenticeship Commencements subsidy.",
      "supports": "That entry-level training volume responds sharply and quickly to subsidy withdrawal, and that the response is concentrated in the non-trade occupations closest to office work.",
      "doesNotSupport": "An AI effect. NCVER names policy change, and the timing follows the 2022 subsidy withdrawal rather than any AI milestone.",
      "terms": [
        "apprenticeships",
        "Australia",
        "early careers",
        "subsidy"
      ],
      "relatedPages": [
        "/research/do-apprenticeships-still-work"
      ]
    },
    {
      "id": "cedefop-apprenticeships-ai-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#cedefop-apprenticeships-ai-2026",
      "section": "international",
      "authors": "Cedefop",
      "year": 2026,
      "title": "Call for abstracts: apprenticeships in the age of AI",
      "publication": "Cedefop, European Centre for the Development of Vocational Training, 5 August 2026",
      "url": "https://www.cedefop.europa.eu/en/news/call-abstracts-apprenticeships-age-ai",
      "grade": "compiled-review",
      "method": "Framing statement for the 2027 joint Cedefop and OECD apprenticeship symposium. Position paper, not a study.",
      "finding": "Cedefop states: \"AI takes over baseline tasks which were in many cases performed by apprentices or apprenticeship graduates who get entry-level roles once their programmes are completed. As apprenticeships continue to expand into new fields and occupations, they may also be exposed to a decline in entry-level openings.\" It adds that the contraction appears concentrated in white-collar roles, so the effect is likely to be uneven and occupation-specific rather than uniform.",
      "supports": "That a European Union agency has now named the mechanism this research describes, in its own words, and applied it specifically to apprenticeships. It is the clearest institutional statement that the training route and the entry-level job are the same thing.",
      "doesNotSupport": "Anything measured. It is a call for papers, which means Cedefop is asking the question rather than answering it, and it says the effect is uneven.",
      "terms": [
        "apprenticeships",
        "AI",
        "entry-level",
        "European Union",
        "missing rungs"
      ],
      "relatedPages": [
        "/research/do-apprenticeships-still-work",
        "/research/missing-rungs"
      ]
    },
    {
      "id": "ilo-youth-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#ilo-youth-2026",
      "section": "international",
      "authors": "International Labour Organization",
      "year": 2026,
      "title": "Global Employment Trends for Youth 2026: Back to the future",
      "publication": "ILO, 11 August 2026",
      "url": "https://www.ilo.org/resource/news/youth-unemployment-rises-young-people-face-harder-road-decent-work",
      "grade": "institutional-survey",
      "method": "ILO global modelled estimates of youth employment, unemployment and NEET status.",
      "finding": "Global youth unemployment 12.4 per cent in 2025, 67 million people aged 15 to 24. The NEET rate is 20 per cent, over 257 million. Youth unemployment rose in 8 of 11 subregions between 2023 and 2025, with Northern America rising from 8.3 to 9.8 per cent. The ILO estimates 6.1 per cent of jobs held by workers aged 15 to 29 are in occupations highly exposed to AI-related change.",
      "supports": "That youth labour market conditions deteriorated across most of the world between 2023 and 2025, which is the context any national apprenticeship policy is operating in.",
      "doesNotSupport": "That AI caused it. The 6.1 per cent exposure figure is an occupational overlap measure, not a measured displacement.",
      "terms": [
        "youth unemployment",
        "NEET",
        "early careers",
        "international"
      ],
      "relatedPages": [
        "/research/do-apprenticeships-still-work",
        "/research/what-is-the-ai-employment-gap"
      ]
    },
    {
      "id": "highfliers-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#highfliers-2026",
      "section": "work",
      "authors": "High Fliers Research",
      "year": 2026,
      "title": "The Graduate Market in 2026",
      "publication": "High Fliers Research, January 2026",
      "url": "https://highfliers.co.uk/publication-the-graduate-market-report",
      "grade": "institutional-survey",
      "method": "Annual survey of graduate recruitment at the UK's 100 leading graduate employers. An employer survey of a selected group, not an official statistic.",
      "finding": "Graduate recruitment fell 5.1 per cent in 2025, after a 14.6 per cent drop in 2024 and a 6.4 per cent decrease in 2023, with a further 0.5 per cent decrease forecast for 2026. Graduate recruitment at these employers has fallen 24.5 per cent since 2022, the lowest level since 2012. Employers received on average 23 per cent more applications in the first half of the 2025-2026 season, with applications roughly doubling since 2023.",
      "supports": "That the graduate entry route at the UK's largest recruiters has contracted by roughly a quarter in three years while competition for what remains has roughly doubled.",
      "doesNotSupport": "Attribution to AI, which the survey does not establish, and it covers 100 leading employers rather than the whole labour market.",
      "terms": [
        "graduate recruitment",
        "early careers",
        "United Kingdom"
      ],
      "relatedPages": [
        "/research/do-apprenticeships-still-work",
        "/research/will-ai-replace-entry-level-jobs"
      ]
    },
    {
      "id": "ise-recruitment-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#ise-recruitment-2025",
      "section": "work",
      "authors": "Institute of Student Employers",
      "year": 2025,
      "title": "Student Recruitment Survey 2025",
      "publication": "Institute of Student Employers, 2025, with 2026 outlook published 7 January 2026",
      "url": "https://ise.org.uk/_userfiles/pages/files/reports/student_recruitment_survey_2025.pdf",
      "grade": "institutional-survey",
      "method": "Trade association survey of 155 employers covering over 31,000 student hires from more than 1.8 million applications in the 2024-2025 cycle.",
      "finding": "An average of 89 applications per vacancy. The ISE reports a projected 7 per cent drop in student vacancies for 2026, while 30 per cent of employers increased student hiring.",
      "supports": "That the contraction is uneven. Nearly a third of employers increased student hiring in a falling market, which cuts against any account of uniform collapse.",
      "doesNotSupport": "Whole-market figures. It is a membership survey, and its members are organisations that run structured student recruitment in the first place.",
      "terms": [
        "graduate recruitment",
        "early careers",
        "applications"
      ],
      "relatedPages": [
        "/research/do-apprenticeships-still-work"
      ]
    },
    {
      "id": "kahneman-klein-2009",
      "citeAs": "https://thesuperskills.com/research/evidence#kahneman-klein-2009",
      "section": "judgement",
      "authors": "Kahneman, D. and Klein, G.",
      "year": 2009,
      "title": "Conditions for intuitive expertise: a failure to disagree",
      "publication": "American Psychologist, 64(6), 515-526",
      "url": "https://pubmed.ncbi.nlm.nih.gov/19739881/",
      "grade": "peer-reviewed",
      "method": "Adversarial collaboration between the leading proponents of the heuristics-and-biases and naturalistic-decision-making traditions, who had reached opposite conclusions about expert intuition. A joint theoretical paper rather than a new experiment.",
      "finding": "Judging the likely quality of an intuitive judgement requires assessing two things: the predictability of the environment in which the judgement is made, and the individual's opportunity to learn that environment's regularities. Where both hold, recognitional expertise is trustworthy. Where either fails, confident intuition is not evidence of skill.",
      "supports": "That expert intuition is neither reliably good nor reliably poor, and that the discriminating variable is the environment rather than the expert. It also supplies the test for when preserving human judgement is worth the cost.",
      "doesNotSupport": "Which specific professional environments meet the conditions. The authors give the criteria and not a classification, so applying it to law, medicine, consulting or management is a judgement in itself.",
      "terms": [
        "judgement",
        "expertise",
        "intuition",
        "decision making"
      ],
      "relatedPages": [
        "/research/what-is-judgement",
        "/research/ai-and-human-judgement",
        "/research/ai-and-human-intuition"
      ]
    },
    {
      "id": "macnamara-2014",
      "citeAs": "https://thesuperskills.com/research/evidence#macnamara-2014",
      "section": "learning",
      "authors": "Macnamara, B. N., Hambrick, D. Z. and Oswald, F. L.",
      "year": 2014,
      "title": "Deliberate Practice and Performance in Music, Games, Sports, Education, and Professions: A Meta-Analysis",
      "publication": "Psychological Science, 25(8), 1608-1618",
      "url": "https://journals.sagepub.com/doi/abs/10.1177/0956797614535810",
      "grade": "compiled-review",
      "method": "Meta-analysis of 88 studies and 157 effect sizes relating accumulated deliberate practice to performance.",
      "finding": "Deliberate practice explained 12 per cent of the variance in performance overall, 95% CI [9%, 15%], leaving 88 per cent unexplained. By domain: 26 per cent for games, 21 per cent for music, 18 per cent for sports, 4 per cent for education and under 1 per cent for the professions.",
      "supports": "That accumulated practice is one contributor among several rather than the dominant explanation of expert performance, and that its contribution varies sharply by domain.",
      "doesNotSupport": "That practice does not build professional expertise. The professions estimate rests on 7 effect sizes, is not statistically significant (p = .62), and the occupations sampled were computer programming, military aircraft piloting, soccer refereeing and insurance selling. It is not evidence about law, medicine, consulting or analysis. It gets quoted as though it were.",
      "terms": [
        "deliberate practice",
        "expertise"
      ],
      "relatedPages": [
        "/research/what-is-deliberate-practice"
      ]
    },
    {
      "id": "nap-hpl2-2018",
      "citeAs": "https://thesuperskills.com/research/evidence#nap-hpl2-2018",
      "section": "learning",
      "authors": "National Academies of Sciences, Engineering, and Medicine",
      "year": 2018,
      "title": "How People Learn II: Learners, Contexts, and Cultures",
      "publication": "The National Academies Press, Washington DC",
      "url": "https://www.nationalacademies.org/read/24783/chapter/6",
      "grade": "compiled-review",
      "method": "Consensus study report synthesising research on learning across the lifespan.",
      "finding": "Defines metacognition as the ability to monitor and regulate one's own cognitive processes and to consciously regulate behaviour, including affective behaviour, and identifies calibration as the accuracy of a learner's monitoring.",
      "supports": "That the established definition contains a control component as well as a knowledge component. Monitoring alone is not metacognition.",
      "doesNotSupport": "Anything about AI. The report predates general availability of these systems and makes no claim about them.",
      "terms": [
        "metacognition",
        "calibration",
        "self-regulated learning"
      ],
      "relatedPages": [
        "/research/what-is-metacognition"
      ]
    },
    {
      "id": "porter-2022",
      "citeAs": "https://thesuperskills.com/research/evidence#porter-2022",
      "section": "judgement",
      "authors": "Porter, T., Elnakouri, A., Meyers, E. A., Shibayama, T., Jayawickreme, E. and Grossmann, I.",
      "year": 2022,
      "title": "Predictors and consequences of intellectual humility",
      "publication": "Nature Reviews Psychology, 1, 524-536",
      "url": "https://www.nature.com/articles/s44159-022-00081-9",
      "grade": "compiled-review",
      "method": "Review synthesising definitions and findings on intellectual humility across personality, judgement, education and organisational research.",
      "finding": "Identifies a metacognitive core on which there is scholarly consensus, recognising the limits of one's knowledge and being aware of one's fallibility, with social and behavioural features around it: recognising that others may hold legitimate differing beliefs, and willingness to reveal ignorance in order to learn.",
      "supports": "That the metacognitive core is agreed across the field even though the wider construct is not.",
      "doesNotSupport": "That the construct is settled, or that it can be measured reliably by asking people. The review notes that where intellectual humility is seen as desirable, self-report makes a false impression easy to create.",
      "terms": [
        "intellectual humility",
        "metacognition"
      ],
      "relatedPages": [
        "/research/what-is-intellectual-humility"
      ]
    },
    {
      "id": "commonsense-companions-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#commonsense-companions-2025",
      "section": "humanness",
      "authors": "Common Sense Media",
      "year": 2025,
      "title": "Talk, Trust, and Trade-offs: How and Why Teens Use AI Companions",
      "publication": "Common Sense Media, San Francisco, 16 July 2025",
      "url": "https://www.commonsensemedia.org/sites/default/files/research/talk-trust-and-trade-offs_2025_toplines.pdf",
      "grade": "institutional-survey",
      "method": "Nationally representative survey of 1,060 US teens aged 13 to 17, April and May 2025. Risk items were asked of the 758 respondents who had used an AI companion.",
      "finding": "72 per cent of teens had used an AI companion at least once and 52 per cent were regular users, a few times a month or more. Among users, 33 per cent had chosen an AI companion over a real person for something important or serious, 34 per cent had felt uncomfortable with something a companion said or did, and 24 per cent had shared personal or private information. The same survey found 67 per cent of teens rating AI conversations as less satisfying than conversations with real friends against 10 per cent more satisfying, 80 per cent of users spending more time with friends than with companions, and 50 per cent not trusting the advice.",
      "supports": "That AI companion use among US teenagers is majority behaviour rather than a fringe one, and that a substantial minority of users have taken serious conversations and personal information to them.",
      "doesNotSupport": "Anything about developmental effects, which are not measured here. Self-report at a single point in time, by an organisation that backs legislation to ban these products for minors, and the risk percentages are of users rather than of all teens: 33 per cent of users choosing a companion for a serious conversation is roughly 24 per cent of teens. The press release framing of a generation replacing human connection is not supported by the survey's own Q5 and Q7.",
      "terms": [
        "AI companions",
        "young people"
      ],
      "relatedPages": []
    },
    {
      "id": "nist-airmf-manage-2-4",
      "citeAs": "https://thesuperskills.com/research/evidence#nist-airmf-manage-2-4",
      "section": "institutional",
      "authors": "National Institute of Standards and Technology",
      "year": 2023,
      "title": "AI Risk Management Framework Playbook, MANAGE 2.4",
      "publication": "NIST AI Resource Center, AI RMF 1.0 Playbook",
      "url": "https://airc.nist.gov/airmf-resources/playbook/manage/",
      "grade": "institutional-modelling",
      "method": "Voluntary framework and playbook developed through public consultation. Guidance rather than measurement.",
      "finding": "Requires mechanisms and assigned responsibilities to supersede, disengage or deactivate AI systems showing performance inconsistent with intended use, and names five triggering conditions: end of system lifetime; risks exceeding tolerance thresholds; mitigation beyond the organisation's capacity; feasible mitigations failing regulatory, legal or normative standards; and impending risk detected in monitoring for which timely mitigation cannot be implemented. Decision thresholds for bypass or deactivation are treated as part of continual monitoring, and organisations are encouraged to provide contingency options including redundant or backup systems.",
      "supports": "That an authoritative framework treats deployment as reversible and expects the reversal mechanism, its thresholds and its fallback to exist before they are needed.",
      "doesNotSupport": "That any organisation does this, or that doing it works. The AI RMF is voluntary guidance, not a standard with conformity assessment, and it contains no evidence about outcomes.",
      "terms": [
        "governance",
        "deployment",
        "oversight"
      ],
      "relatedPages": []
    },
    {
      "id": "eu-ai-act-art-14",
      "citeAs": "https://thesuperskills.com/research/evidence#eu-ai-act-art-14",
      "section": "institutional",
      "authors": "European Union",
      "year": 2024,
      "title": "Regulation (EU) 2024/1689, Article 14: Human Oversight",
      "publication": "Official Journal of the European Union. Applies from 2 December 2027 for Annex III high-risk systems and 2 August 2028 for Annex I",
      "url": "https://artificialintelligenceact.eu/article/14/",
      "grade": "institutional-modelling",
      "method": "Binding regulation. Legal requirement rather than empirical finding.",
      "finding": "Requires high-risk systems to be designed so they can be effectively overseen by natural persons, and requires that those persons be enabled to understand the system's capacities and limitations, to remain aware of the tendency to over-rely on its output (automation bias, named in the text), to interpret the output correctly, to decide not to use it or to disregard, override or reverse it, and to intervene or stop it through a stop button bringing the system to a safe halt. For biometric identification systems under Annex III point 1(a), no action may be taken unless the identification is separately verified by at least two natural persons with the necessary competence, training and authority.",
      "supports": "That an authoritative regulator treats override capability, not merely presence, as the content of oversight, and names competence, training and authority together in the one place it specifies who must verify.",
      "doesNotSupport": "That any of this happens. The oversight provisions carry 2 December 2027 as a longstop rather than a start date, since the Digital Omnibus ties the high-risk rules to the availability of standards and caps the delay at sixteen months for Annex III, so they may apply sooner; see eu-ai-act-application-dates. the two-person requirement covers one narrow category rather than high-risk systems generally, and the regulation contains no evidence that oversight so constituted works.",
      "terms": [
        "human oversight",
        "governance",
        "automation bias"
      ],
      "relatedPages": [
        "/research/how-do-you-design-a-stop-button-people-will-use",
        "/research/how-do-humans-and-agents-divide-work"
      ]
    },
    {
      "id": "nist-ir-8312",
      "citeAs": "https://thesuperskills.com/research/evidence#nist-ir-8312",
      "section": "judgement",
      "authors": "Phillips, P. J., Hahn, C. A., Fontana, P. C., Broniatowski, D. A. and Przybocki, M. A.",
      "year": 2021,
      "title": "Four Principles of Explainable Artificial Intelligence (NISTIR 8312)",
      "publication": "National Institute of Standards and Technology Interagency Report 8312",
      "url": "https://nvlpubs.nist.gov/nistpubs/ir/2021/NIST.IR.8312.pdf",
      "grade": "compiled-review",
      "method": "Framework paper synthesising the explainable-AI literature. Not a measurement study.",
      "finding": "Sets out four principles: explanation, meaningful, explanation accuracy and knowledge limits. In the report's own words, explanation accuracy is a distinct concept from decision accuracy, and regardless of the system's decision accuracy the corresponding explanation may or may not accurately describe how the system came to its conclusion. It also notes that the explanation and meaningful principles alone do not require an explanation to reflect the system's actual process.",
      "supports": "That an explanation being intelligible, and even being accurate about the process, is separate from the answer being right. The two are different properties and are routinely treated as one.",
      "doesNotSupport": "That explanations help or harm users in practice. This is a framework rather than an experiment, and it reports no effect on human decision quality.",
      "terms": [
        "explainability",
        "interpretability",
        "oversight"
      ],
      "relatedPages": []
    },
    {
      "id": "cruces-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#cruces-2026",
      "section": "learning",
      "authors": "Cruces, G., Fernandez Meijide, D., Galiani, S., Galvez, R. and Lombardi, M.",
      "year": 2026,
      "title": "Does generative AI narrow education-based productivity gaps? Evidence from a randomized experiment",
      "publication": "NBER Working Paper 34851; CEPR Discussion Paper 21299; arXiv:2608.04198",
      "url": "https://www.nber.org/papers/w34851",
      "grade": "working-paper",
      "method": "Randomised online experiment, 1,174 adults aged 25 to 45, workplace-style problem-solving task with or without a generative AI assistant, followed by an unassisted module.",
      "finding": "AI improved performance for everyone and more for the less educated. Without AI, higher-education participants outperformed lower-education participants by 0.548 standard deviations; with AI the gap fell to 0.139, closing about three-quarters of it. Treated participants did not perform worse once AI was removed, and lower-education participants retained part of their improvement, although a sizeable gap re-emerged.",
      "supports": "That assisted use does not automatically leave people worse off than unassisted controls when the tool is taken away, and that AI can compress an education-based performance gap while it is present.",
      "doesNotSupport": "That skill formed. One session with an immediate unassisted module tests transfer within a sitting, not skill formation over time, and the studies that found post-removal deficits taught a body of knowledge and removed the tool afterwards. The equity gain is also partly transient by the authors' own account, since a sizeable gap re-emerges without the assistant.",
      "terms": [
        "capability debt",
        "productivity",
        "learning",
        "equity"
      ],
      "relatedPages": [
        "/research/capability-debt"
      ]
    },
    {
      "id": "shen-tamkin-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#shen-tamkin-2026",
      "section": "learning",
      "authors": "Shen, J. H. and Tamkin, A.",
      "year": 2026,
      "title": "How AI Impacts Skill Formation",
      "publication": "arXiv:2601.20245. Work conducted under the Anthropic Fellows Program",
      "url": "https://arxiv.org/abs/2601.20245",
      "grade": "working-paper",
      "method": "Randomised experiments with developers learning a new asynchronous programming library through a self-guided tutorial, with and without an AI assistant.",
      "finding": "AI use impaired conceptual understanding, code reading and debugging without significant average efficiency gains. Participants with AI assistance scored 17 per cent lower on a comprehension quiz than those who coded by hand, while finishing only marginally faster. The authors identify six AI interaction patterns, three of which involve cognitive engagement and preserve learning outcomes even with AI assistance.",
      "supports": "That the effect on learning depends on how the tool is used rather than on whether it is used, which is the same conclusion the guardrailed and scaffolded arms of the school and programming trials reached from the other direction.",
      "doesNotSupport": "That AI use degrades professional expertise. One library, one tutorial, developers rather than a general population, and a preprint. The conflict of interest runs against the finding rather than towards it, since the work was conducted under an AI company's fellowship and reports harm, but it is disclosed here either way.",
      "terms": [
        "skill formation",
        "learning",
        "capability debt",
        "scaffolding"
      ],
      "relatedPages": [
        "/research/capability-debt"
      ]
    },
    {
      "id": "ebu-bbc-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#ebu-bbc-2025",
      "section": "professions",
      "authors": "Fletcher, J. and Verckist, D. (BBC and European Broadcasting Union)",
      "year": 2025,
      "title": "News Integrity in AI Assistants: An international PSM study",
      "publication": "BBC and European Broadcasting Union, published 21 October 2025",
      "url": "https://www.ebu.ch/research/open/report/news-integrity-in-ai-assistants",
      "grade": "compiled-review",
      "method": "Twenty-two public service media organisations across 18 countries and 14 languages put 30 shared news questions, drawn from questions audiences had actually asked, to the free consumer versions of ChatGPT, Copilot, Perplexity and Gemini between 24 May and 10 June 2025. Each prompt opened a new chat and asked the assistant to use the participating organisation's sources where possible. Assistants were anonymised and 271 journalists graded 2,709 core responses against accuracy, sourcing, separation of opinion from fact, editorialisation and context, marking each as no issues, some issues, significant issues or don't know.",
      "finding": "Forty-five per cent of responses carried at least one significant issue, and 81 per cent carried an issue of some kind. Sourcing was the largest single cause at 31 per cent, then accuracy at 20 per cent and insufficient context at 14 per cent. Gemini recorded significant issues in 76 per cent of responses against 37 per cent for Copilot, 36 per cent for ChatGPT and 30 per cent for Perplexity, driven by sourcing, where Gemini's rate was 72 per cent against 24, 15 and 15. Of responses that cited a participating broadcaster's content, 15 per cent misrepresented it. Where the same BBC-only comparison could be run against the 2025 first round, significant issues fell from 51 per cent to 37 per cent.",
      "supports": "That misattribution and unsupported sourcing, rather than outright fabrication, is the dominant failure mode when general assistants answer news questions, and that it holds across languages, territories and platforms rather than being an artefact of one market or one model.",
      "doesNotSupport": "Current performance of any named product. These were free consumer versions tested in mid-2025 and all four have shipped new defaults since. It is not adversarial testing and difficulty was not controlled, so the rate is not a worst case; equally, per-organisation samples were around 120 responses, which the authors say is too small to compare countries or languages. Nothing here measures what readers then believed or did.",
      "terms": [
        "hallucination",
        "sourcing",
        "journalism",
        "verification"
      ],
      "relatedPages": [
        "/research/how-will-ai-change-journalism"
      ]
    },
    {
      "id": "nyc-ll144-2021",
      "citeAs": "https://thesuperskills.com/research/evidence#nyc-ll144-2021",
      "section": "institutional",
      "authors": "Council of the City of New York; Department of Consumer and Worker Protection",
      "year": 2021,
      "title": "Local Law 144 of 2021, automated employment decision tools",
      "publication": "New York City Administrative Code ss 20-870 to 20-874, became law 13 December 2021, in force 1 January 2023. Rules at 6 RCNY Subchapter T, ss 5-300 to 5-304",
      "url": "https://www.nyc.gov/site/dca/about/automated-employment-decision-tools.page",
      "grade": "institutional-modelling",
      "method": "Municipal statute and implementing rules. Requires an annual independent bias audit of an automated employment decision tool before it is used on a candidate or employee in New York City, publication of the audit summary, and notice to those it is used on.",
      "finding": "The rules at 6 RCNY s 5-300 define the trigger, 'substantially assist or replace discretionary decision making', in three limbs, of which the third is: 'to use a simplified output to overrule conclusions derived from other factors including human decision-making'. A municipal legislature wrote down, in 2022, the mechanism by which a score displaces a judgement.",
      "supports": "That a jurisdiction has defined in binding rules the specific failure this research is about: a simplified machine output overruling a human conclusion. The definition is narrower and more precise than the oversight language in most national AI frameworks.",
      "doesNotSupport": "Anything about practice or enforcement, which is the subject of the separate Comptroller audit graded below. It is city law, applying only to employment decisions within New York City, and it mandates an audit rather than an outcome.",
      "terms": [
        "automated employment decision tool",
        "bias audit",
        "human decision-making",
        "oversight"
      ],
      "relatedPages": [
        "/research/what-is-meaningful-human-oversight",
        "/research/can-ai-be-unbiased",
        "/research/is-it-ethical-to-let-ai-judge-people"
      ]
    },
    {
      "id": "nys-comptroller-ll144-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#nys-comptroller-ll144-2025",
      "section": "institutional",
      "authors": "Office of the New York State Comptroller",
      "year": 2025,
      "title": "Department of Consumer and Worker Protection: Enforcement of Local Law 144",
      "publication": "Audit report 2024-N-6, issued 2 December 2025, covering July 2023 to June 2025",
      "url": "https://www.osc.ny.gov/state-agencies/audits/2025/12/02/department-consumer-and-worker-protection-enforcement-local-law-144",
      "grade": "statutory-investigation",
      "method": "Performance audit by the state audit authority of the city department charged with enforcing Local Law 144, covering the first two years of enforcement. The auditors independently reviewed the same sample of companies the department had reviewed.",
      "finding": "The department surveyed the websites and bias audits of 32 companies and identified a single issue of non-compliance. The auditors reviewed the same companies and identified at least seventeen instances of potential non-compliance. Two complaints were received in the two-year period.",
      "supports": "That the existence of a rule and the operation of a rule are different facts, measured here by a state audit authority rather than asserted. The first AI hiring law in the world produced one finding from the enforcing body where independent review of the same sample produced at least seventeen.",
      "doesNotSupport": "That the seventeen are violations. The audit says potential non-compliance, and the department disputes elements of the finding. It measures enforcement activity, not whether the tools in question caused any harm, and it covers one department in one city.",
      "terms": [
        "enforcement",
        "bias audit",
        "oversight",
        "compliance"
      ],
      "relatedPages": [
        "/research/what-is-meaningful-human-oversight",
        "/research/can-ai-be-unbiased",
        "/research/is-it-ethical-to-let-ai-judge-people"
      ]
    },
    {
      "id": "quebec-p391-s121",
      "citeAs": "https://thesuperskills.com/research/evidence#quebec-p391-s121",
      "section": "international",
      "authors": "National Assembly of Quebec",
      "year": 2021,
      "title": "Act respecting the protection of personal information in the private sector, s 12.1",
      "publication": "CQLR c P-39.1 s 12.1, inserted by SQ 2021 c 25 s 110, in force 22 September 2023. Public-sector mirror at CQLR c A-2.1 s 65.2 (SQ 2021 c 25 s 21). Consolidation current to 7 April 2026",
      "url": "https://www.legisquebec.gouv.qc.ca/en/document/cs/p-39.1",
      "grade": "institutional-modelling",
      "method": "Provincial statute, read at the official consolidated text in both official languages. One version only, unamended since coming into force.",
      "finding": "Where an enterprise uses personal information to render a decision based exclusively on automated processing, it must say so by the time it communicates the decision, and on request give the information used, the reasons and principal factors, and the right of correction. It then requires that 'the person concerned must be given the opportunity to submit observations to a member of the personnel of the enterprise who is in a position to review the decision'. The French is impersonal: 'Il doit etre donne a la personne concernee l occasion de presenter ses observations a un membre du personnel de l entreprise en mesure de reviser la decision.'",
      "supports": "That a jurisdiction has legislated not a right to an explanation but a right to put arguments to a named human with authority to change the answer. It is the strongest human-review provision found in North America and it names a capacity, being in a position to review, rather than a role.",
      "doesNotSupport": "That it is used, or that the reviewing person is competent to redo the analysis. The statute does not define what being in a position to review requires. Quebec s own regulator lists four limits, graded separately below. It applies only to decisions based EXCLUSIVELY on automated processing, which excludes most decisions in practice.",
      "terms": [
        "human review",
        "automated decision",
        "meaningful human involvement",
        "oversight"
      ],
      "relatedPages": [
        "/research/who-supervises-work-they-cannot-do"
      ]
    },
    {
      "id": "cai-quebec-ia-travail-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#cai-quebec-ia-travail-2025",
      "section": "international",
      "authors": "Commission d acces a l information du Quebec",
      "year": 2025,
      "title": "L IA au travail: pour un meilleur encadrement",
      "publication": "Memoire presented to the ministere du Travail, dated 27 January 2025, published 21 February 2025",
      "url": "https://www.cai.gouv.qc.ca/uploads/pdfs/CAI_ME_Transfo_Travail.pdf",
      "grade": "argued-perspective",
      "method": "Submission by the provincial access and privacy regulator to a government consultation on the digital transformation of workplaces. A reasoned position, not a study.",
      "finding": "On meaningful human intervention the Commission states, at page 4: 'lorsqu un humain enterine une decision proposee par un systeme d IA sans etudier l ensemble de l analyse, il existe un risque qu il demontre un biais d automatisation en faisant exagerement confiance a ce systeme'. It then lists four limits of sections 12.1 and 65.2: no duty to disclose at collection, no application to decisions that are not fully automated, no criterion for what counts as a decision, and the burden on the individual to ask. Recommendation 6 proposes prohibiting fully automated decisions with significant effects on employees.",
      "supports": "That a North American regulator has stated in writing that the human who ratifies without examining the whole analysis is where the safeguard fails, and has named automation bias as the mechanism. It is a regulator arguing against the sufficiency of its own jurisdiction s provision.",
      "doesNotSupport": "Anything measured. It is a submission, it contains no data on how often ratification without examination occurs, and its recommendations had not been enacted at the time of reading.",
      "terms": [
        "automation bias",
        "meaningful human involvement",
        "human review",
        "oversight"
      ],
      "relatedPages": [
        "/research/what-is-automation-bias"
      ]
    },
    {
      "id": "finland-hallintolaki-8b-2023",
      "citeAs": "https://thesuperskills.com/research/evidence#finland-hallintolaki-8b-2023",
      "section": "international",
      "authors": "Parliament of Finland",
      "year": 2023,
      "title": "Administrative Procedure Act, chapter 8 b, automated decision making",
      "publication": "Hallintolaki ss 53 e to 53 g, inserted by act 487/2023 of 23 March 2023, in force 1 May 2023, with companion act 488/2023",
      "url": "https://www.finlex.fi/fi/laki/ajantasa/2003/20030434",
      "grade": "institutional-modelling",
      "method": "General administrative statute, not an AI-specific instrument. Applies to all automated decision making by Finnish authorities.",
      "finding": "An authority may decide a matter automatically only where the matter 'contains no elements requiring case-by-case discretion' (johon ei sisally seikkoja, jotka edellyttavat tapauskohtaista harkintaa). A decision counts as automated where it is reached without a natural person checking and approving it. Section 53 e(4) provides that a request for rectification may not be decided automatically. The companion act requires a named individual responsible for each system, a published deployment decision, and records permitting five-year reconstruction of the stages at which a natural person took part.",
      "supports": "That the human-discretion test can be stated in a single clause of general administrative law, and that a legislature has done so. It defines what makes a matter unsuitable for automation, which most AI frameworks require oversight without ever specifying.",
      "doesNotSupport": "That it works, or that Finnish authorities classify matters correctly. It binds public authorities only, not private employers. Finland s implementation of the EU AI Act is separately late, so this is administrative-law strength rather than AI-law strength.",
      "terms": [
        "case-by-case discretion",
        "automated decision",
        "human review",
        "oversight"
      ],
      "relatedPages": [
        "/research/human-in-the-loop-is-not-a-safeguard"
      ]
    },
    {
      "id": "eoak-3379-2018-tax",
      "citeAs": "https://thesuperskills.com/research/evidence#eoak-3379-2018-tax",
      "section": "international",
      "authors": "Deputy Parliamentary Ombudsman of Finland (Maija Sakslin)",
      "year": 2019,
      "title": "Decision on the Tax Administration s automated decision making",
      "publication": "EOAK/3379/2018, decision of 20 November 2019",
      "url": "https://www.oikeusasiamies.fi/",
      "grade": "statutory-investigation",
      "method": "Own-initiative investigation by the Parliamentary Ombudsman into the legality of automated decision making at the Finnish Tax Administration, which issued on the order of fifteen million decisions a year.",
      "finding": "The Ombudsman found the practice unlawful. The reasoning turns on accountability rather than error: official accountability had become indirect (virkavastuu jaa valilliseksi), because no identifiable official could be said to have made the decision. Parliament subsequently legislated chapter 8 b of the Administrative Procedure Act, in force 2023, and the Chancellor of Justice applied the new law against Kela in April 2025.",
      "supports": "That a supervisory body identified the loss of an answerable human as the defect, independent of whether the decisions were correct, and that a legislature acted on that finding. Four bodies over six years reaching the same conclusion about the same problem is an unusually complete chain.",
      "doesNotSupport": "That any decision was wrong. The finding is about accountability structure, not accuracy, and the Ombudsman did not measure outcomes. It concerns Finnish administrative law and does not transfer to private-sector decisions.",
      "terms": [
        "accountability",
        "automated decision",
        "oversight",
        "moral crumple zone"
      ],
      "relatedPages": [
        "/research/what-is-a-moral-crumple-zone"
      ]
    },
    {
      "id": "denmark-vej-9590-2018",
      "citeAs": "https://thesuperskills.com/research/evidence#denmark-vej-9590-2018",
      "section": "international",
      "authors": "Digitaliseringsstyrelsen (Danish Agency for Digital Government)",
      "year": 2018,
      "title": "Vejledning om digitaliseringsklar lovgivning",
      "publication": "VEJ nr 9590 af 12/07/2018, mandatory for government bills since 1 July 2018, status Gaeldende as at September 2026",
      "url": "https://www.retsinformation.dk/eli/retsinfo/2018/9590",
      "grade": "institutional-modelling",
      "method": "Binding guidance on the drafting of Danish government legislation. Every government bill is assessed against seven principles before it goes to Parliament.",
      "finding": "The third principle carries a written checklist question: 'Er det sikret, at det fagprofessionelle skon er opretholdt i tilfaelde, hvor hensynet til borgernes retssikkerhed taler herfor?' (Has it been ensured that professional discretion has been preserved in cases where regard for citizens legal certainty so requires?) The same document states that objective rules should be used only where it makes sense and where professional discretion is not needed, and places a residual duty on the authority to ensure discretion continues to be exercised. The 2018 wording is advisory; the Ministry of Justice s current legislative-quality guidance states that new legislation shall be digital-ready.",
      "supports": "That a state has built a compulsory written checkpoint on whether human judgement survives a change of process, answered by a named official before a parliamentary vote. It is a procedural instrument for preserving discretion rather than a rule about technology.",
      "doesNotSupport": "That discretion is in fact preserved. It measures a drafting process, not outcomes, no published review of how the question is answered was found, and it binds those who draft legislation rather than those who decide individual cases.",
      "terms": [
        "professional discretion",
        "human judgement",
        "oversight",
        "legal certainty"
      ],
      "relatedPages": [
        "/research/what-board-oversight-of-ai-looks-like"
      ]
    },
    {
      "id": "eurostat-isoc-eb-ai-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#eurostat-isoc-eb-ai-2025",
      "section": "international",
      "authors": "Eurostat",
      "year": 2025,
      "title": "Artificial intelligence use by enterprises (isoc_eb_ai), 2025 reference year",
      "publication": "Eurostat news release, 11 December 2025, and statistical report KS-01-26-009-EN-N. Fieldwork Q1 2025",
      "url": "https://ec.europa.eu/eurostat/web/products-eurostat-news/w/ddn-20251211-1",
      "grade": "institutional-survey",
      "method": "Harmonised survey of enterprises with ten or more persons employed across NACE C to J, L to N and 95.1, excluding the financial sector, agriculture and the public sector. Approximately 157,000 enterprises surveyed from a population of about 1.53 million. An enterprise counts as a user if it used at least one of eight named AI technologies.",
      "finding": "EU-27 average 19.95 per cent, up 6.47 points on 2024. Denmark first at 42.0 per cent, Finland second at 37.8, Sweden 35.0, Netherlands 33.2 (break in series), Spain 20.3, Portugal 11.5, Romania lowest at 5.2. Norway, reporting as a non-EU EEA state, 28.9 per cent. On individuals, a companion series puts Denmark highest in the EU at 48.4 per cent and Norway highest in Europe at 56.3.",
      "supports": "A comparable cross-national baseline for enterprise AI adoption, collected to a documented standard, which allows countries to be ranked against each other rather than against vendor surveys.",
      "doesNotSupport": "Depth of use. An enterprise counts if it used one of eight technologies once, so the measure says nothing about how many workers use AI, how often, or for what. It excludes the financial and public sectors and firms under ten employees. National statistics offices publish figures on different populations, so a national number and a Eurostat number for the same country are frequently not comparable.",
      "terms": [
        "adoption",
        "international comparison",
        "enterprise AI use"
      ],
      "relatedPages": [
        "/research/ai-and-work-by-country"
      ]
    },
    {
      "id": "omb-m2410-automation-bias-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#omb-m2410-automation-bias-2024",
      "section": "institutional",
      "authors": "Office of Management and Budget (Shalanda D. Young)",
      "year": 2024,
      "title": "M-24-10, Advancing Governance, Innovation, and Risk Management for Agency Use of Artificial Intelligence",
      "publication": "OMB memorandum M-24-10, 28 March 2024, 34 pages. Rescinded and replaced by M-25-21 on 3 April 2025",
      "url": "https://bidenwhitehouse.archives.gov/wp-content/uploads/2024/03/M-24-10-Advancing-Governance-Innovation-and-Risk-Management-for-Agency-Use-of-Artificial-Intelligence.pdf",
      "grade": "institutional-modelling",
      "method": "Binding executive-branch guidance to United States federal agencies, read in full at the archived original. Both documents were searched term by term.",
      "finding": "M-24-10 used the term automation bias twice. As a defined term at section 6: 'the propensity for humans to inordinately favor suggestions from automated decision-making systems and to ignore or fail to seek out contradictory information made without automation'. And as a mandatory minimum practice at section 5(c)(iv)(G): agencies 'must ensure there is sufficient training, assessment, and oversight for operators of the AI to interpret and act on the AI s output, combat any human-machine teaming issues (such as automation bias)'. The successor memorandum M-25-21 of 3 April 2025 contains the term nowhere. M-24-10 also distinguished rights-impacting from safety-impacting AI, a distinction M-25-21 collapses into a single high-impact class.",
      "supports": "That binding United States federal AI guidance named automation bias in March 2024, both as a definition and as a requirement tied to operator training, and that the language was removed thirteen months later when the memorandum was replaced.",
      "doesNotSupport": "That the removal was deliberate or that federal agencies have stopped addressing the risk. M-25-21 still requires human oversight, intervention and accountability for high-impact uses. The words over-reliance, deskilling and complacency appear in NEITHER document, so the finding concerns one term and not a vocabulary. Both are memoranda rather than statute, and revocable.",
      "terms": [
        "automation bias",
        "human oversight",
        "operator training",
        "federal AI guidance"
      ],
      "relatedPages": [
        "/research/what-is-automation-bias"
      ]
    },
    {
      "id": "fed-sr26-2-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#fed-sr26-2-2026",
      "section": "professions",
      "authors": "Board of Governors of the Federal Reserve System, Federal Deposit Insurance Corporation and Office of the Comptroller of the Currency",
      "year": 2026,
      "title": "Supervisory Guidance on Model Risk Management (SR 26-2)",
      "publication": "Federal Reserve supervisory letter SR 26-2, 17 April 2026, 12 pages. Supersedes SR 11-7 (4 April 2011) and SR 21-8 (9 April 2021)",
      "url": "https://www.federalreserve.gov/supervisionreg/srletters/SR2602.htm",
      "grade": "institutional-modelling",
      "method": "Interagency supervisory guidance for United States banking organisations, most relevant to those above $30 billion in total assets. Replaces the 2011 model risk management framework after fifteen years of supervisory experience.",
      "finding": "The definition of a model is narrowed to 'a complex quantitative method, system, or approach that applies statistical, economic, or financial theories to process input data into quantitative estimates', expressly excluding simple spreadsheet arithmetic, deterministic rule-based processes and software with no such theory underpinning it. Footnote 3 states that generative and agentic AI models 'are novel and rapidly evolving' and 'are not within the scope of this guidance', while the principles do apply to traditional quantitative models and to non-generative, non-agentic AI. Effective challenge is retained and defined as critical analysis by objective experts with expertise, independence and the organisational standing to force change. The guidance sets no enforceable standards and says non-compliance will not itself draw supervisory criticism.",
      "supports": "That the most mature oversight regime any profession has for machine-produced numbers has, on its own initiative and in 2026, placed generative AI outside its scope. Anyone citing model risk management as the ready-made precedent for governing generative AI in finance is citing a document that declines the job.",
      "doesNotSupport": "That generative AI in banks is ungoverned. The same footnote directs firms to their own risk management and governance practices for tools outside scope, and other supervisory expectations, consumer protection law and third-party risk guidance still apply. It is United States banking supervision only, and guidance rather than rule.",
      "terms": [
        "model risk",
        "effective challenge",
        "oversight",
        "financial services"
      ],
      "relatedPages": [
        "/research/how-will-ai-change-accounting-and-audit"
      ]
    },
    {
      "id": "frc-ai-audit-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#frc-ai-audit-2025",
      "section": "professions",
      "authors": "Financial Reporting Council",
      "year": 2025,
      "title": "AI in audit: Illustrative example and documentation guidance",
      "publication": "Financial Reporting Council, June 2025, published 26 June 2025",
      "url": "https://www.frc.org.uk/documents/8384/AI_in_Audit.pdf",
      "grade": "institutional-modelling",
      "method": "First FRC guidance on artificial intelligence in statutory audit, developed with the FRC's Technology Working Group. Two parts: a worked example of an unsupervised machine learning tool used for journal risk assessment, and principles for documenting tools that use AI on the audit file.",
      "finding": "The scope is deliberately broad, covering 'both traditional machine learning techniques and deep learning models, including generative AI'. On explainability the FRC declines to set a threshold: 'what constitutes appropriate explainability will vary widely based on context', and appropriate explanations 'may, particularly in relation to tools that rely on neural networks, be approximate or post hoc explanations that seek to explain how inputs influence outputs rather than the internal features and workings of the model'. Automation bias appears once, as something training material should carry 'strategies to mitigate'. Engagement teams are required to understand why the tool flagged an item and to stay alert to the possibility that its assessment is systemically flawed for that entity. The guidance states that it is not prescriptive and that 'the requirements against which firms will be assessed remain only those in the ISQMs and ISAs (UK)'.",
      "supports": "That a professional regulator can bring generative AI inside an existing evidence and documentation standard without writing new rules, and that it can do so while accepting post hoc explanation rather than model transparency.",
      "doesNotSupport": "Anything about practice. It is guidance issued in 2025, not a finding about what firms do, and the FRC says it creates no new requirements. It contains nothing on junior auditors, training pipelines or skills: the words junior, trainee and graduate do not appear.",
      "terms": [
        "audit",
        "explainability",
        "automation bias",
        "documentation"
      ],
      "relatedPages": [
        "/research/how-will-ai-change-accounting-and-audit"
      ]
    },
    {
      "id": "frc-att-thematic-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#frc-att-thematic-2025",
      "section": "professions",
      "authors": "Financial Reporting Council",
      "year": 2025,
      "title": "Thematic Review: Certification of Automated Tools and Techniques",
      "publication": "Financial Reporting Council, June 2025, published 26 June 2025. Fieldwork Q2 2024 to autumn 2024",
      "url": "https://www.frc.org.uk/documents/8383/Thematic_Review_on_the_Certification_of_Automated_Tools_and_Techniques.pdf",
      "grade": "compiled-review",
      "method": "Review of the processes and controls by which the six largest UK audit firms, named as BDO, Deloitte, EY, Forvis Mazars, KPMG and PwC, certify automated tools and techniques before use in audits. Information request issued April 2024, firm meetings summer 2024, feedback autumn 2024. A snapshot of process, not an inspection of audits.",
      "finding": "All six firms had certification processes, but 'the maturity of these processes was found to vary and in some cases were not supported by formal documented policies'. Only two of the six set out the limitations of a tool or restrictions on its use in the certification documentation. Three captured assessment of the supporting IT control environment. One enforced a minimum recertification frequency, of three years. Generally the firms had no key performance indicators for tool usage and monitoring. The finding that carries furthest: 'There was no formal monitoring performed by the firms to quantify the audit quality impact of using ATTs.' At the time of review, generative AI use was limited to productivity aids such as chatbots rather than tools producing audit evidence.",
      "supports": "That the profession whose function is verification had, as at 2024, deployed the tools that produce its evidence without measuring their effect on the quality of that evidence, by its regulator's own account.",
      "doesNotSupport": "That audit quality has fallen. The review measures process rather than outcome, covers only the six largest firms, and its counts are of documentation practice rather than of tools or audits. Nothing here is a sample of engagements.",
      "terms": [
        "audit",
        "verification",
        "oversight",
        "governance"
      ],
      "relatedPages": [
        "/research/how-will-ai-change-accounting-and-audit"
      ]
    },
    {
      "id": "krugel-2023",
      "citeAs": "https://thesuperskills.com/research/evidence#krugel-2023",
      "section": "judgement",
      "authors": "Krügel, S., Ostermaier, A. and Uhl, M.",
      "year": 2023,
      "title": "ChatGPT's inconsistent moral advice influences users' judgment",
      "publication": "Scientific Reports, 13, 4569. DOI 10.1038/s41598-023-31341-0. Published 6 April 2023",
      "url": "https://www.nature.com/articles/s41598-023-31341-0",
      "grade": "peer-reviewed",
      "method": "Preregistered online experiment run on 21 December 2022 with 1,851 US residents recruited through CloudResearch Prime Panels, of whom 767 passed both comprehension checks and form the analysis sample as preregistered. Participants read a transcript of advice on a trolley dilemma, in the switch or the bridge version, arguing for or against sacrificing one life to save five, attributed either to ChatGPT or to a human moral advisor. The advice itself came from ChatGPT, which had given contradictory answers to the same question on 14 December 2022.",
      "finding": "The advice moved participants' own moral judgement in both dilemmas, and in the bridge version it flipped the majority verdict. Disclosure made almost no difference: the effect was statistically indistinguishable whether the source was named as a chatbot or as a human advisor. Eighty per cent of participants said they would have reached the same judgement without the advice, and their judgements show they would not have. Only 67 per cent said the same of other participants, and 79 per cent rated themselves more ethical than the others.",
      "supports": "That advice from a model with no settled position still moves the position of the person reading it, that telling them it is a machine does not protect them, and that they cannot see it happening. Transparency, on this evidence, is not a sufficient safeguard.",
      "doesNotSupport": "How large the shift is in absolute terms. The paper reports test statistics and figure proportions rather than an effect size in the text, and no confidence intervals appear in the prose. One dilemma type, one sitting, a 41 per cent comprehension pass rate, and a model version from December 2022. It says nothing about repeated real-life decisions or about whether the influence persists.",
      "terms": [
        "advice taking",
        "moral judgement",
        "transparency",
        "over-reliance"
      ],
      "relatedPages": [
        "/research/should-i-let-ai-make-personal-decisions-for-me",
        "/research/is-it-still-my-idea-if-ai-helped-me-write-it"
      ]
    },
    {
      "id": "storey-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#storey-2026",
      "section": "capability",
      "authors": "Storey, M.-A.",
      "year": 2026,
      "title": "From Technical Debt to Cognitive and Intent Debt: Rethinking Software Health in the Age of AI",
      "publication": "ACM Queue, preprint at arXiv 2603.22106, March 2026",
      "url": "https://arxiv.org/pdf/2603.22106",
      "grade": "compiled-review",
      "method": "Conceptual synthesis by a Canada Research Chair at the University of Victoria, drawing on Naur (1985) on programming as theory building, Cunningham (1993) on technical debt, and the author's own empirical work with Starr on developer confusion. Illustrated with a single teaching anecdote rather than a study. Proposes a framework; presents no new data.",
      "finding": "Proposes a triple debt model: technical debt lives in code, cognitive debt lives in people as the erosion of shared understanding across a team, and intent debt lives in artefacts as the absence of captured rationale, goals and constraints. Argues generative AI may reduce technical debt while accelerating the other two, because code can now be produced faster than a team can build the understanding needed to change it safely.",
      "supports": "That the debt metaphor has been extended into a structured framework by a serious researcher, and that cognitive debt now carries a team-level meaning distinct from the individual-level one in Kosmyna et al. The distinction is the author's own and she states it explicitly.",
      "doesNotSupport": "Anything empirical about prevalence or magnitude. It is a framework paper with an anecdote, and it says so. Its reference list also mis-cites Kosmyna et al. as a 2024 CHI workshop paper, which does not appear on the MIT Media Lab's own publications list for that author; the citation is not relied on here.",
      "terms": [
        "cognitive debt",
        "intent debt",
        "technical debt",
        "shared understanding"
      ],
      "relatedPages": [
        "/research/cognitive-debt-and-capability-debt",
        "/research/will-ai-replace-programmers"
      ]
    },
    {
      "id": "acemoglu-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#acemoglu-2026",
      "section": "institutional",
      "authors": "Acemoglu, D., Kong, D. and Ozdaglar, A.",
      "year": 2026,
      "title": "AI, Human Cognition and Knowledge Collapse",
      "publication": "NBER Working Paper 34910, February 2026, DOI 10.3386/w34910",
      "url": "https://www.nber.org/papers/w34910",
      "grade": "institutional-modelling",
      "method": "Dynamic theoretical model of learning and decision-making in which successful decisions require combining community-level general knowledge with individual context-specific knowledge, treated as complements. Human effort jointly produces a private signal and a thin public signal, creating a learning externality. Agentic AI substitutes for that effort. No empirical estimation.",
      "finding": "Identifies a conditional tipping point: when human effort is sufficiently elastic and agentic recommendations exceed an accuracy threshold, the economy can reach a knowledge-collapse steady state in which general knowledge ultimately vanishes despite high-quality personalised advice. Welfare is non-monotone in agentic accuracy, implying an interior optimum. Greater capacity to aggregate and pool human-generated general knowledge raises welfare unambiguously.",
      "supports": "That the erosion argument can be stated formally with its assumptions visible, which is more than most of the vocabulary in this area manages. Also that the policy implication is not simply less AI: the model's unambiguous lever is better pooling of human knowledge, not lower agentic accuracy.",
      "doesNotSupport": "That knowledge collapse is happening or will happen. It is a model producing a possible steady state under stated conditions, it is a working paper rather than a peer-reviewed article, and the authors do not claim to have measured anything.",
      "terms": [
        "knowledge collapse",
        "learning externality",
        "agentic AI",
        "general knowledge"
      ],
      "relatedPages": [
        "/research/cognitive-debt-and-capability-debt"
      ]
    },
    {
      "id": "eef-nfer-chatgpt-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#eef-nfer-chatgpt-2024",
      "section": "professions",
      "authors": "Education Endowment Foundation and National Foundation for Educational Research",
      "year": 2024,
      "title": "ChatGPT in lesson preparation: a Teacher Choices trial",
      "publication": "EEF project record and evaluation report, 12 December 2024, project completed August 2026. Co-funded by the Hg Foundation; the ChatGPT guide was developed by Bain and Company's Social Impact practice",
      "url": "https://educationendowmentfoundation.org.uk/projects-and-evaluation/projects/choices-in-edtech-using-generative-ai-chatgpt-for-ks3-science-lesson-preparation-2024-teacher-choices-trial",
      "grade": "peer-reviewed",
      "method": "Two-arm school-randomised Teacher Choices trial, 259 Year 7 and Year 8 science teachers across 68 state-funded secondary schools in England, ten weeks in the summer term of 2024. One arm used ChatGPT with a written guide; the other was asked to use no generative AI. Weeks one to five were a familiarisation period and planning time was recorded in weeks six to ten. Resource quality was assessed by an expert panel blinded to condition. Independent evaluation by NFER.",
      "finding": "Weekly lesson and resource preparation time was 56.2 minutes in the ChatGPT arm against 81.5 minutes in the comparison arm, a saving of 25.3 minutes and a reduction of 31 per cent, given a high security rating. The blinded panel found no noticeable difference in resource quality. The proportion of ChatGPT-arm teachers who felt they spent too much time on preparation fell from 49 to 26 per cent, with no similar fall in the comparison arm. Frequency of use, and consultation of the guide, both declined over the trial.",
      "supports": "That light, largely unsupported use of a general assistant reduces teacher preparation time measurably, without a detectable cost to the quality of the materials produced, in one subject at one key stage.",
      "doesNotSupport": "Anything about teaching or about pupils. Preparation time was the outcome; classroom effect was outside scope. Time was self-recorded rather than observed. The sample over-represents schools in London and the South East and schools rated Outstanding, which the report states. The 31 per cent excludes a five-week familiarisation period.",
      "terms": [
        "teacher workload",
        "lesson planning",
        "generative AI in schools"
      ],
      "relatedPages": [
        "/research/how-will-ai-change-teaching"
      ]
    },
    {
      "id": "dfe-tech-schools-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#dfe-tech-schools-2025",
      "section": "institutional",
      "authors": "Department for Education and IFF Research",
      "year": 2025,
      "title": "Technology in Schools survey: 2024 to 2025",
      "publication": "DfE research report, November 2025",
      "url": "https://assets.publishing.service.gov.uk/media/692834a6ce50d215cae9610e/Technology_in_schools_survey_2024_to_2025_research_report.pdf",
      "grade": "institutional-survey",
      "method": "Survey of 1,634 schools in England, comprising 795 school leaders, 1,211 teachers and 489 IT leads, with qualitative interviews alongside. Questions on AI were new to the survey for 2025. Self-reported throughout.",
      "finding": "44 per cent of teachers reported using generative AI for school activities: lesson planning 35 per cent, delivering live lessons 7 per cent, marking 5 per cent. Teachers under 35 used it for planning at 43 per cent against 32 per cent for older colleagues, and for written feedback at 21 against 12 per cent. Teachers with under three years' experience used it for written feedback at 27 per cent against 14 per cent for those teaching longer. Leaders were more likely to plan investment in AI tools for teachers than for pupils, 58 against 20 per cent. Around one fifth of schools had a policy on safe and appropriate AI use. 77 per cent of secondary leaders whose pupils could access generative AI reported issues, most commonly plagiarism at 67 per cent.",
      "supports": "Where the teaching profession in England has actually placed the tool, on a large national sample: heavily in preparation, minimally in marking and delivery, and largely without written policy.",
      "doesNotSupport": "Any effect. It is a cross-sectional self-report of usage and perception, with no measurement of workload, learning or quality. The plagiarism figure records reported issues rather than the extent of plagiarism, which the report notes.",
      "terms": [
        "AI adoption in schools",
        "teacher workload",
        "marking",
        "school policy"
      ],
      "relatedPages": [
        "/research/how-will-ai-change-teaching"
      ]
    },
    {
      "id": "kestin-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#kestin-2025",
      "section": "learning",
      "authors": "Kestin, G., Miller, K., Klales, A., Milbourne, T. and Ponti, G.",
      "year": 2025,
      "title": "AI tutoring outperforms in-class active learning: an RCT introducing a novel research-based design in an authentic educational setting",
      "publication": "Scientific Reports, 15, published 3 June 2025",
      "url": "https://www.nature.com/articles/s41598-025-97652-6",
      "grade": "peer-reviewed",
      "method": "Randomised crossover experiment in Harvard's largest introductory physics course, Fall 2023. Of 233 enrolled students, 194 were eligible on consent and completion. Each student experienced both conditions across two consecutive weeks, one topic taught by in-class active learning and one by a purpose-built AI tutor at home, with pre-tests and post-tests for each. The tutor used GPT-4 with expert-crafted question-specific prompts, pre-written answers, instructional video and a structured scaffold.",
      "finding": "Median post-test score 4.5 in the AI condition against 3.5 in the active-learning condition, from a combined pre-test median of 2.75; median learning gain over double. Mann-Whitney z = -5.6, p below 10 to the minus 8. Linear regression effect size 0.63, described by the authors as an underestimate because of a ceiling effect; quantile regression gives 0.73 to 1.3 standard deviations. Median time on task 49 minutes against 60 assumed for the class, with no correlation between time on task and score. Engagement 4.1 against 3.6 and motivation 3.4 against 3.1; enjoyment and growth mindset showed no significant difference.",
      "supports": "That a heavily engineered AI tutor, built by subject experts to follow established pedagogy, can outperform a well-run active-learning class on immediate post-test performance at the understanding, applying and analysing levels, in less time.",
      "doesNotSupport": "That a general chatbot does this. Accuracy depended on pre-written answers and instructor-written prompts. Retention was not measured; post-tests followed the lessons immediately. One course, one institution, two topics. The authors state they do not presume the result holds where complex synthesis or higher-order critical thinking is required.",
      "terms": [
        "AI tutoring",
        "active learning",
        "learning gains",
        "scaffolding"
      ],
      "relatedPages": [
        "/research/how-will-ai-change-teaching"
      ]
    },
    {
      "id": "gds-copilot-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#gds-copilot-2025",
      "section": "professions",
      "authors": "Government Digital Service",
      "year": 2025,
      "title": "Microsoft 365 Copilot Experiment: Cross-Government Findings Report",
      "publication": "Government Digital Service, June 2025",
      "url": "https://assets.publishing.service.gov.uk/media/683db42bd23a62e5d32680d0/M365_Copilot_Experiment_Findings_Report.pdf",
      "grade": "institutional-survey",
      "method": "Cross-government trial of M365 Copilot from 30 September to 31 December 2024, 20,000 licences across twelve organisations, each committing at least 1,000. Usage data from the Microsoft dashboard for 14,500 users; survey of 7,115 users; five focus groups. Time savings were self-estimated by selecting a band, and the average was computed from band midpoints with the largest savings estimated at 60 minutes.",
      "finding": "Average self-reported saving 26 minutes a day. Adoption reached 83 per cent and held around 80 per cent. 17 per cent noticed no clear saving; more than a third reported over half an hour. Drafting documents 24 minutes, creating presentations 19, scheduling meetings 9. 82 per cent said they would not want to return to pre-Copilot conditions; satisfaction 7.7 and recommendation 8.2 out of 10; 85 per cent agreed it provided good value; 63 per cent believed their productivity would decline without it. The conclusions state it was not possible to identify how the saved time was spent.",
      "supports": "That a very large public sector deployment achieved high adoption and strongly positive user sentiment, and that accessibility benefits for disabled and neurodivergent users were a consistent theme.",
      "doesNotSupport": "Any measured productivity effect. The headline is a self-reported estimate, bucketed, with its top band capped by the analysts, and with no control group, no baseline task timing and no measure of output quality or decision quality. The report itself flags inconsistent user experience across departments and a festive-period disruption.",
      "terms": [
        "public sector productivity",
        "Copilot",
        "self-reported time saving"
      ],
      "relatedPages": [
        "/research/how-will-ai-change-the-public-sector",
        "/research/how-will-ai-change-human-resources",
        "/research/does-ai-actually-make-people-more-productive"
      ]
    },
    {
      "id": "dwp-copilot-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#dwp-copilot-2026",
      "section": "professions",
      "authors": "Arzilli, F., Lynch, C. S. and Page, L.",
      "year": 2026,
      "title": "An Evaluation of DWP's Microsoft 365 Copilot Trial",
      "publication": "Department for Work and Pensions, published 29 January 2026",
      "url": "https://www.gov.uk/government/publications/an-evaluation-of-dwps-microsoft-copilot-365-trial/an-evaluation-of-dwps-microsoft-365-copilot-trial",
      "grade": "institutional-survey",
      "method": "Mixed-methods post-implementation evaluation of a trial running October 2024 to March 2025 with 3,549 licensed staff. Survey of users (1,716 responses) and of a random stratified comparison group of non-users (2,535 responses from 9,300 sampled), 19 qualitative interviews, and seemingly unrelated regression controlling for demographic, occupational and AI-keenness variables. No baseline; licences allocated first come, first served.",
      "finding": "Estimated saving of 19 minutes a day across eight routine tasks, statistically significant across all specifications, with the largest task effects on searching for information (26 minutes), writing emails (25) and summarising (24). 90 per cent of users said it saved time. Job satisfaction rose 0.56 points and perceived work quality 0.49 points on seven-point scales; 73 per cent reported better quality outputs and 65 per cent felt more fulfilled.",
      "supports": "That a regression-based estimate on a large departmental sample, controlling for observable differences, still finds a positive and significant self-reported effect on efficiency, satisfaction and perceived quality.",
      "doesNotSupport": "A measured time saving. The evaluation's own limitations chapter names the absence of baseline data, post-treatment bias, self-selection towards AI enthusiasts through first-come first-served allocation which it says may lead to overestimation, non-response bias and acquiescence bias on the time question. Nothing about decision quality or citizen outcomes was measured.",
      "terms": [
        "public sector productivity",
        "Copilot",
        "selection bias",
        "self-reported time saving"
      ],
      "relatedPages": [
        "/research/how-will-ai-change-the-public-sector",
        "/research/does-ai-actually-make-people-more-productive"
      ]
    },
    {
      "id": "nao-ai-government-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#nao-ai-government-2024",
      "section": "institutional",
      "authors": "National Audit Office",
      "year": 2024,
      "title": "Use of artificial intelligence in government",
      "publication": "HC 612, Session 2023-24, 15 March 2024",
      "url": "https://www.nao.org.uk/reports/use-of-artificial-intelligence-in-government/",
      "grade": "institutional-survey",
      "method": "Value-for-money audit of the Cabinet Office and DSIT, including a survey of 87 government bodies conducted in autumn 2023, document review and interviews. Excludes simple rules-based automation, AI embedded by default in existing tools, and individuals' ad hoc use of public tools.",
      "finding": "37 per cent of responding bodies had deployed AI, typically one or two use cases; 70 per cent were piloting or planning, median four use cases. 21 per cent had an organisational AI strategy, with 61 per cent planning one. Of 32 bodies with deployed AI, 24 always or usually had a named accountable owner and 15 said use cases were always or usually identified at organisational level before deployment. 30 per cent of all respondents had risk and quality assurance processes explicitly incorporating AI risks. 70 per cent named difficulty recruiting or retaining AI skills as a barrier. The Cabinet Office's Central Digital and Data Office identified in 2023 that almost a third of civil service tasks, those it defined as routine, could be automated, and did not examine feasibility or assess cost.",
      "supports": "That the UK productivity claim for public sector AI rests on an indicative sizing exercise the auditor found untested for feasibility or cost, and that organisational ownership of deployed AI was incomplete in 2023.",
      "doesNotSupport": "The current position. The survey was taken in autumn 2023, before the generative wave reached most departments, and the picture will have moved. Survey response is self-reported and covers 87 bodies rather than the whole public sector.",
      "terms": [
        "AI in government",
        "governance",
        "accountability",
        "productivity claims"
      ],
      "relatedPages": [
        "/research/how-will-ai-change-the-public-sector",
        "/research/which-ai-investments-should-we-stop"
      ]
    },
    {
      "id": "moj-computer-evidence-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#moj-computer-evidence-2025",
      "section": "institutional",
      "authors": "Ministry of Justice",
      "year": 2025,
      "title": "The use of evidence generated by software in criminal proceedings",
      "publication": "Call for evidence, 21 January to 15 April 2025, foreword by Sarah Sackman KC MP, Minister for Courts and Legal Services",
      "url": "https://assets.publishing.service.gov.uk/media/67892f5a93d4eae3088bd324/use-evidence-generated-software-criminal-proceedings.pdf",
      "grade": "institutional-modelling",
      "method": "Government call for evidence, not a study. Sets out the current common law position, states the proposed boundaries of any reform, and puts five questions to respondents.",
      "finding": "Records that section 69 of the Police and Criminal Evidence Act 1984, which required a party to show a computer was operating properly, was repealed on a 1997 Law Commission recommendation and replaced from 2000 by a common law rebuttable presumption that the computer was operating correctly at the material time. The foreword summarises this as the computer being always right unless someone shows otherwise, and cites the Post Office Horizon convictions as demonstrating the fallibility of software-generated evidence. Proposes that any reform cover evidence generated by software including artificial intelligence and algorithms, naming accounting systems, automated fraud and plagiarism detection and automated reporting from handheld devices, while excluding material merely captured by a device.",
      "supports": "That the legal presumption favouring machine output is live, is being reconsidered by the UK government, and that the government's own proposed scope for reform expressly includes AI and algorithmic systems.",
      "doesNotSupport": "Any outcome. It is a call for evidence rather than a decision, and no reform had been enacted at the date of review. It carries no data on how often the presumption is challenged or successfully rebutted.",
      "terms": [
        "presumption of reliability",
        "computer evidence",
        "automation bias",
        "Horizon"
      ],
      "relatedPages": [
        "/research/how-will-ai-change-the-public-sector"
      ]
    },
    {
      "id": "atrs-register-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#atrs-register-2026",
      "section": "institutional",
      "authors": "Government Digital Service",
      "year": 2026,
      "title": "Algorithmic transparency records",
      "publication": "GOV.UK register maintained under the Algorithmic Transparency Recording Standard, first published January 2023, mandatory scope and exemptions policy December 2024. Count read 1 September 2026",
      "url": "https://www.gov.uk/algorithmic-transparency-records",
      "grade": "compiled-review",
      "method": "Public register of records completed by UK public sector organisations under the Algorithmic Transparency Recording Standard. Mandatory for all government departments and for arm's-length bodies delivering public or frontline services or interacting directly with the public; recommended for the wider public sector. Self-declared by publishing organisations.",
      "finding": "143 records published as at 1 September 2026, from central departments, agencies, local authorities, police forces and devolved administrations. Disclosed systems include a Department for Work and Pensions scanner reading around 25,000 scanned citizen documents a day to flag people who may need urgent assistance, a tool flagging Universal Credit journal messages that may indicate a risk of harm, the Cabinet Office verbal and numerical tests used to sift civil service applicants, an Ofsted tool drafting sections of children's home inspection reports, and adult social care case-note generation at a local authority.",
      "supports": "That algorithmic tools sitting between citizens and decisions about them are in production at volume in UK government, and that a public, structured record of some of them exists and can be read by anyone.",
      "doesNotSupport": "The extent of use. The register shows what has been disclosed and cannot show what has not. It is self-declared, the count moves, and the National Audit Office found in 2024 that the standard was not widely used before it became mandatory.",
      "terms": [
        "algorithmic transparency",
        "automated decision-making",
        "public sector AI"
      ],
      "relatedPages": [
        "/research/how-will-ai-change-the-public-sector",
        "/research/how-will-ai-change-human-resources"
      ]
    },
    {
      "id": "wilson-caliskan-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#wilson-caliskan-2024",
      "section": "professions",
      "authors": "Wilson, K. and Caliskan, A.",
      "year": 2024,
      "title": "Gender, Race, and Intersectional Bias in Resume Screening via Language Model Retrieval",
      "publication": "Proceedings of the 2024 AAAI/ACM Conference on AI, Ethics, and Society; preprint arXiv 2407.20371",
      "url": "https://arxiv.org/abs/2407.20371",
      "grade": "peer-reviewed",
      "method": "Resume audit study run through a document retrieval framework simulating job candidate selection. Massive Text Embedding models tested across nine occupations using over 500 publicly available resumes and over 500 job descriptions, with 120 first names associated with male, female, Black and white candidates. Code published by the authors.",
      "finding": "The embedding models significantly favoured White-associated names in 85.1 per cent of cases and female-associated names in 11.1 per cent, with a minority of comparisons showing no statistically significant difference. Black male candidates were disadvantaged in up to 100 per cent of cases. Document length and the corpus frequency of a name also affected selection. Three hypotheses of intersectionality were validated.",
      "supports": "That the representation layer underneath commercial screening tools carries a large, measurable and intersectional name-based bias before any product logic is added, and that some of the effect is an artefact of name frequency in training data rather than anything about a candidate.",
      "doesNotSupport": "The behaviour of any deployed product. These are open embedding models in a simulated pipeline, not the proprietary systems vendors sell, which add filters and thresholds and cannot be independently tested. It measures ranking behaviour rather than who was hired.",
      "terms": [
        "algorithmic hiring",
        "resume screening",
        "embedding bias",
        "intersectionality"
      ],
      "relatedPages": [
        "/research/how-will-ai-change-human-resources"
      ]
    },
    {
      "id": "wright-null-compliance-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#wright-null-compliance-2024",
      "section": "institutional",
      "authors": "Wright, L., Muenster, R. M., Vecchione, B., Qu, T., Cai, P., Smith, A., COMM/INFO 2450 Student Investigators, Metcalf, J. and Matias, J. N.",
      "year": 2024,
      "title": "Null Compliance: NYC Local Law 144 and the Challenges of Algorithm Accountability",
      "publication": "Proceedings of the 2024 ACM Conference on Fairness, Accountability, and Transparency, Rio de Janeiro, DOI 10.1145/3630106.3658998",
      "url": "https://facctconference.org/static/papers24/facct24-113.pdf",
      "grade": "peer-reviewed",
      "method": "Field audit. 155 student investigators, acting as model job seekers, recorded the compliance of 391 employers with New York City Local Law 144 and the user experience for a prospective applicant. Accompanied by legal and policy analysis of the statute.",
      "finding": "18 employers posted a bias audit report, roughly 5 per cent, and 13 posted a transparency notice, roughly 3 per cent. The authors name the resulting state null compliance: non-compliance cannot be established because the law's design makes it impossible to determine whether an employer uses a covered tool. The analysis records that Local Law 144 requires an audit but is silent on its results, sets no discrimination threshold including the four-fifths convention, provides no remediation guidance, and that no federal safe harbour protects employers who disclose, so publication may create liability under other law.",
      "supports": "That the first algorithmic bias audit law in the world produced almost no public disclosure, and identifies a specific design reason for that rather than attributing it to employer indifference.",
      "doesNotSupport": "That the audited tools are biased or unbiased. No audit results were analysed because almost none were published, which is the finding. The sample is 391 employers with a large New York workforce rather than a census.",
      "terms": [
        "bias audit",
        "algorithmic accountability",
        "null compliance",
        "Local Law 144"
      ],
      "relatedPages": [
        "/research/how-will-ai-change-human-resources",
        "/research/can-ai-be-unbiased",
        "/research/is-it-ethical-to-let-ai-judge-people"
      ]
    },
    {
      "id": "eu-ai-act-annex-3",
      "citeAs": "https://thesuperskills.com/research/evidence#eu-ai-act-annex-3",
      "section": "institutional",
      "authors": "European Parliament and Council of the European Union",
      "year": 2024,
      "title": "Regulation (EU) 2024/1689, Annex III: High-risk AI systems referred to in Article 6(2)",
      "publication": "Official Journal of the European Union, official version of 13 June 2024. Text read via the European Commission AI Act Service Desk",
      "url": "https://ai-act-service-desk.ec.europa.eu/en/ai-act/annex-3",
      "grade": "institutional-modelling",
      "method": "Binding legislative text. Annex III lists the eight areas in which AI systems are classified as high risk under Article 6(2), triggering the requirements of Chapter III.",
      "finding": "Point 3 classifies as high risk AI systems used to determine access or admission to education, to evaluate learning outcomes including where those outcomes steer a learner's path, to assess the level of education a person will receive, and to monitor and detect prohibited behaviour during tests. Point 4 covers employment, workers' management and access to self-employment, naming systems used for recruitment or selection, including placing targeted job advertisements, analysing and filtering applications and evaluating candidates, and systems making decisions on promotion or termination, allocating tasks by individual traits, or monitoring and evaluating performance.",
      "supports": "That two of the applications discussed most loosely in public debate, marking pupils' work and sifting job applicants, are named in binding European law as high risk, with documentation, record-keeping, human oversight and explanation obligations attached.",
      "doesNotSupport": "Compliance or effect. Classification is not evidence that any system is biased or unsafe, and the Annex says nothing about how well the resulting obligations are met in practice.",
      "terms": [
        "EU AI Act",
        "high-risk AI",
        "algorithmic hiring",
        "assessment"
      ],
      "relatedPages": [
        "/research/how-will-ai-change-human-resources",
        "/research/how-will-ai-change-teaching",
        "/research/is-it-ethical-to-let-ai-judge-people"
      ]
    },
    {
      "id": "giedd-1999",
      "citeAs": "https://thesuperskills.com/research/evidence#giedd-1999",
      "section": "learning",
      "authors": "Giedd, J.N., Blumenthal, J., Jeffries, N.O., Castellanos, F.X., Liu, H., Zijdenbos, A., Paus, T., Evans, A.C. and Rapoport, J.L.",
      "year": 1999,
      "title": "Brain development during childhood and adolescence: a longitudinal MRI study",
      "publication": "Nature Neuroscience, 2(10), 861-863",
      "url": "https://pubmed.ncbi.nlm.nih.gov/10491603/",
      "grade": "peer-reviewed",
      "method": "Longitudinal MRI, 243 scans from 145 healthy participants, 89 male and 46 female, at the US National Institute of Mental Health.",
      "finding": "Cortical grey matter in frontal regions peaks in pre-adolescence and then thins through synaptic pruning, with the prefrontal cortex among the last areas to mature.",
      "supports": "That the prefrontal cortex matures later than other regions. This is real, replicated and not in dispute.",
      "doesNotSupport": "Any threshold age. The paper reports a slower trajectory, not an endpoint, and names no age at which development completes. Everything downstream that cites it for a cut-off is citing something it does not contain.",
      "terms": [
        "brain development",
        "adolescence",
        "prefrontal cortex"
      ],
      "relatedPages": [
        "/research/does-the-brain-mature-at-25"
      ]
    },
    {
      "id": "gogtay-2004",
      "citeAs": "https://thesuperskills.com/research/evidence#gogtay-2004",
      "section": "learning",
      "authors": "Gogtay, N., Giedd, J.N., Lusk, L., Hayashi, K.M., Greenstein, D., Vaituzis, A.C., Nugent, T.F., Herman, D.H., Clasen, L.S., Toga, A.W., Rapoport, J.L. and Thompson, P.M.",
      "year": 2004,
      "title": "Dynamic mapping of human cortical development during childhood through early adulthood",
      "publication": "Proceedings of the National Academy of Sciences, 101(21), 8174-8179",
      "url": "https://www.ncbi.nlm.nih.gov/pmc/articles/PMC419576/",
      "grade": "peer-reviewed",
      "method": "A densely sampled subset of THIRTEEN participants from the NIMH longitudinal project, each scanned roughly every two years.",
      "finding": "Maps the sequence in which cortical regions mature, with higher-order association cortices maturing after lower-order sensorimotor regions.",
      "supports": "A developmental sequence, in thirteen people.",
      "doesNotSupport": "A population age of maturity. Thirteen participants cannot establish one, and the paper does not claim to. This is the study behind the Time Magazine coverage in which the number 25 first appears in public.",
      "terms": [
        "brain development",
        "cortical maturation"
      ],
      "relatedPages": [
        "/research/does-the-brain-mature-at-25"
      ]
    },
    {
      "id": "mousley-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#mousley-2025",
      "section": "learning",
      "authors": "Mousley, A. and colleagues",
      "year": 2025,
      "title": "Topological turning points across the human lifespan",
      "publication": "Nature Communications, November 2025",
      "url": "https://www.nature.com/articles/s41467-025-65974-8",
      "grade": "peer-reviewed",
      "method": "Diffusion MRI from 3,802 people aged 0 to 90, analysed for structural network topology across the lifespan.",
      "finding": "Four topological turning points, at approximately ages 9, 32, 66 and 83, dividing life into five epochs. The adolescent epoch runs from about 9 to about 32. The largest overall shift in trajectory occurs around 32, not in the mid-twenties.",
      "supports": "That structural network reorganisation continues well past the mid-twenties, and that the nearest thing to a boundary at the end of adolescence sits around 32.",
      "doesNotSupport": "That 32 is the new 25. The authors describe turning points in network topology, not a moment of cognitive completion, and reading it as a new threshold would repeat the original error with a different number.",
      "terms": [
        "brain development",
        "lifespan",
        "network topology"
      ],
      "relatedPages": [
        "/research/does-the-brain-mature-at-25"
      ]
    },
    {
      "id": "hartshorne-germine-2015",
      "citeAs": "https://thesuperskills.com/research/evidence#hartshorne-germine-2015",
      "section": "learning",
      "authors": "Hartshorne, J.K. and Germine, L.T.",
      "year": 2015,
      "title": "When does cognitive functioning peak? The asynchronous rise and fall of different cognitive abilities across the life span",
      "publication": "Psychological Science, 26(4), 433-443",
      "url": "https://www.ncbi.nlm.nih.gov/pmc/articles/PMC4441622/",
      "grade": "peer-reviewed",
      "method": "Cross-sectional analysis of large web-based and standardisation samples across a wide age range.",
      "finding": "Different cognitive abilities peak at different ages, some in the late teens or early twenties, others not until the forties or fifties. There is no single age at which cognitive functioning peaks.",
      "supports": "That a single maturity age is the wrong shape of answer, whatever number is put in it. Abilities do not arrive together.",
      "doesNotSupport": "That age is irrelevant. It shows the timing is ability-specific rather than absent.",
      "terms": [
        "cognitive development",
        "peak performance",
        "individual differences"
      ],
      "relatedPages": [
        "/research/does-the-brain-mature-at-25"
      ]
    },
    {
      "id": "adinoff-nunes-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#adinoff-nunes-2025",
      "section": "learning",
      "authors": "Adinoff, B. and Nunes, J.C.",
      "year": 2025,
      "title": "Challenging the 25-year-old 'mature brain' mythology: implications for the minimum legal age for non-medical cannabis use",
      "publication": "The American Journal of Drug and Alcohol Abuse, 51(5), 577-583",
      "url": "https://www.tandfonline.com/doi/full/10.1080/00952990.2025.2561982",
      "grade": "peer-reviewed",
      "method": "Review of the neuroscience and policy literature behind the age-25 threshold.",
      "finding": "Argues the mature-brain-at-25 claim is not supported by the underlying neuroscience and should not be used as a basis for age thresholds in policy.",
      "supports": "That the challenge to this claim is in the peer-reviewed literature rather than confined to science journalism.",
      "doesNotSupport": "Anything about AI, learning or capability. It is cited here for the status of the claim, not for the subject.",
      "terms": [
        "brain development",
        "policy",
        "age thresholds"
      ],
      "relatedPages": [
        "/research/does-the-brain-mature-at-25"
      ]
    },
    {
      "id": "shao-workbank-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#shao-workbank-2026",
      "section": "work",
      "authors": "Shao, Y., Zope, H., Jiang, Y., Pei, J., Nguyen, D., Brynjolfsson, E. and Yang, D.",
      "year": 2026,
      "title": "Future of Work with AI Agents: Auditing Automation and Augmentation Potential across the U.S. Workforce",
      "publication": "arXiv:2506.06576, v3 revised 1 February 2026",
      "url": "https://arxiv.org/abs/2506.06576",
      "grade": "working-paper",
      "method": "Audio-enhanced mini-interviews with 1,500 US domain workers across 104 occupations, covering 844 tasks drawn from O*NET, paired with capability assessments from AI experts. Introduces the Human Agency Scale, H1 to H5, and sorts tasks into four zones by desire against capability.",
      "finding": "Worker preferences diverge sharply from technical capability. Tasks fall into an Automation Green Light Zone, an Automation Red Light Zone where capability exists and workers do not want it used, an R&D Opportunity Zone and a Low Priority Zone. Human Agency Scale profiles vary widely by occupation, and the authors report early signals of core competencies shifting from information-focused skills towards interpersonal ones.",
      "supports": "That the automate-or-not framing is too coarse, and that there is a measurable, occupation-specific preferred level of human involvement which does not track what the technology can do.",
      "doesNotSupport": "Nothing about what happens to capability when a task is automated. It measures what workers WANT and what experts think is POSSIBLE, which are both stated positions rather than outcomes. A preprint, not peer reviewed, and US-only.",
      "terms": [
        "human agency scale",
        "augmentation",
        "automation"
      ],
      "relatedPages": [
        "/research/what-stays-human",
        "/research/questions",
        "/research/which-tasks-do-workers-not-want-automated",
        "/research/what-becomes-more-valuable-as-ai-gets-cheaper"
      ]
    },
    {
      "id": "brynjolfsson-rock-syverson-2021",
      "citeAs": "https://thesuperskills.com/research/evidence#brynjolfsson-rock-syverson-2021",
      "section": "work",
      "authors": "Brynjolfsson, E., Rock, D. and Syverson, C.",
      "year": 2021,
      "title": "The Productivity J-Curve: How Intangibles Complement General Purpose Technologies",
      "publication": "American Economic Journal: Macroeconomics, 13(1), 333-72, January 2021. DOI 10.1257/mac.20180386. Earlier version NBER Working Paper 25148",
      "url": "https://www.aeaweb.org/articles?id=10.1257%2Fmac.20180386",
      "grade": "peer-reviewed",
      "method": "Theoretical model of general purpose technology adoption with unmeasured intangible complementary investment, applied to US national accounts data on computer hardware and software.",
      "finding": "General purpose technologies enable and require significant complementary investments that are often intangible and poorly measured in national accounts. This produces underestimation of productivity growth in a new technology's early years and overestimation later, when the benefits of the intangible investments are harvested, a pattern the authors name the Productivity J-curve. Adjusting for intangibles related to computer hardware and software yields a total factor productivity level 15.9 per cent higher than official measures by the end of 2017. The authors state the AI-related intangible capital effects on measured productivity are currently small but growing.",
      "supports": "That an early productivity read on a general purpose technology is biased downwards for a structural and quantified reason, and that the direction of the later bias is upwards. It supplies the argument for staging an AI investment review rather than taking a single reading.",
      "doesNotSupport": "That AI will follow the same curve, or on what timescale. The 15.9 per cent figure is for computer hardware and software to 2017, not for AI, and the authors describe current AI intangible effects as small. It is also a measurement argument rather than a forecast of returns to any individual firm.",
      "terms": [
        "productivity",
        "intangible capital",
        "general purpose technology",
        "measurement"
      ],
      "relatedPages": [
        "/research/how-long-before-you-know-if-an-ai-investment-worked",
        "/research/if-everyone-has-ai-where-is-the-advantage"
      ]
    },
    {
      "id": "barney-1991",
      "citeAs": "https://thesuperskills.com/research/evidence#barney-1991",
      "section": "work",
      "authors": "Barney, J.",
      "year": 1991,
      "title": "Firm Resources and Sustained Competitive Advantage",
      "publication": "Journal of Management, 17(1), 99-120, March 1991. DOI 10.1177/014920639101700108",
      "url": "https://journals.sagepub.com/doi/10.1177/014920639101700108",
      "grade": "compiled-review",
      "method": "Theoretical article in strategic management. No new data. Builds on the stated assumptions that strategic resources are heterogeneously distributed across firms and that those differences are stable over time.",
      "finding": "Sets out four empirical indicators of the potential of a firm resource to generate sustained competitive advantage: value, rareness, imitability and substitutability. Applies the model to several firm resources and draws out implications for other business disciplines. The founding statement of the resource-based view.",
      "supports": "That a widely used and long-established test exists for whether a resource can confer advantage, against which a purchasable AI licence can be assessed.",
      "doesNotSupport": "Anything empirical, and nothing about AI, which postdates it by three decades. It is a theory with a substantial critical literature, and applying it to a technology three years into commercial diffusion is an argument rather than a finding. The abstract read at source uses rareness, imitability and substitutability; the VRIN acronym is later shorthand and does not appear in the paper's abstract.",
      "terms": [
        "competitive advantage",
        "resource-based view",
        "strategy"
      ],
      "relatedPages": [
        "/research/if-everyone-has-ai-where-is-the-advantage"
      ]
    },
    {
      "id": "census-ai-diffusion-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#census-ai-diffusion-2026",
      "section": "work",
      "authors": "Bonney, K., Breaux, C., Dinlersoz, E., Foster, L., Haltiwanger, J. and Pande, A.",
      "year": 2026,
      "title": "The Microstructure of AI Diffusion: Evidence from Firms, Business Functions, and Worker Tasks",
      "publication": "US Census Bureau, Center for Economic Studies Working Paper CES-26-25, April 2026",
      "url": "https://www.census.gov/library/working-papers/2026/adrm/CES-WP-26-25.html",
      "grade": "working-paper",
      "method": "Nationally representative data from the 2026 AI supplement to the US Census Bureau's Business Trends and Outlook Survey, analysed at three layers: overall firm use, deployment across business functions, and worker-task use. Reference period November 2025 to January 2026.",
      "finding": "18 per cent of firms used AI in a business function, rising to 32 per cent employment-weighted, with adoption expected to reach 22 per cent within six months. Use rates reach 50 to 60 per cent, and 60 to 70 per cent employment-weighted, for very large firms in Information, Professional Services and Finance. Among adopters, 57 per cent integrate AI in three or fewer business functions, most commonly Sales and Marketing (52 per cent), Strategy and Business Development (45 per cent) and IT (41 per cent). Workers use AI in work-related tasks in 23 per cent of firms, 41 per cent employment-weighted, and 65 per cent of firms limit use to three or fewer tasks. Most users, 66 per cent, rely on AI solely to augment tasks, and AI-related employment decreases occur in only 2 per cent of firms. Regression shows a positive correlation between firm commercial performance and the breadth of AI integration, holding across functional deployment, task-level use and operational investment; functional breadth and operational investment are positively associated with employment decreases, while worker-task integration shows no significant link to headcount reduction once the other two are accounted for.",
      "supports": "That AI diffusion is highly uneven by firm size and sector, and shallow even among adopters, which contradicts the premise that competitors hold equivalent capability.",
      "doesNotSupport": "Causation in either direction between adoption and performance: this is a cross-section of firms, and better-run firms may simply adopt more. Also not a stable time series. The Census Bureau broadened the underlying question in November 2025 from use in producing goods or services to use in any business function, which moved the level. A working paper, not peer reviewed, and US-only.",
      "terms": [
        "AI adoption",
        "diffusion",
        "firm size",
        "augmentation"
      ],
      "relatedPages": [
        "/research/if-everyone-has-ai-where-is-the-advantage",
        "/research/how-long-before-you-know-if-an-ai-investment-worked",
        "/research/which-tasks-do-workers-not-want-automated",
        "/research/what-happens-to-work-that-moves-information"
      ]
    },
    {
      "id": "allen-fed-ai-adoption-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#allen-fed-ai-adoption-2026",
      "section": "work",
      "authors": "Allen, J. S.",
      "year": 2026,
      "title": "Monitoring AI Adoption in the U.S. Economy",
      "publication": "FEDS Notes, Board of Governors of the Federal Reserve System, 3 April 2026. DOI 10.17016/2380-7172.4032",
      "url": "https://www.federalreserve.gov/econres/notes/feds-notes/monitoring-ai-adoption-in-the-u-s-economy-20260403.html",
      "grade": "institutional-survey",
      "method": "Comparison of three independent US adoption measures: the Census Business Trends and Outlook Survey (firm-level, around 20,000 responses per wave), the Real-Time Population Survey (individual-level, 5,000 to 6,000 responses) and the Atlanta Fed Survey of Business Uncertainty (senior leaders, 1,032 responses). Four-period moving averages used for all BTOS calculations.",
      "finding": "About 18 per cent of firms had adopted AI as of year-end 2025 on the BTOS. Work-related generative AI adoption in the RPS stood at about 41 per cent of the workforce as of November 2025, with daily use at 12 per cent. The SBU gives an employment-weighted firm adoption rate of about 78 per cent and an LLM adoption rate of about 54 per cent. Allen attributes the variation mainly to differences in sampling distributions and units of analysis, with question framing, the materiality of reported usage, information asymmetries and social desirability bias also contributing, and states senior leaders may face pressure to report AI usage as an efficiency initiative. The Census Bureau broadened its question in November 2025; the do-not-know rate ran at 10 to 11 per cent.",
      "supports": "That headline AI adoption figures differing by sixty points can all be correct, because they measure different units, and that the choice of measure decides the answer before any analysis begins.",
      "doesNotSupport": "Any productivity or employment effect. The note explicitly does not estimate AI's contribution to output, GDP or productivity, and names those as open questions beyond its scope. It also makes no claim that adoption has plateaued; the only slowdown language is deceleration in the second quarter of 2025.",
      "terms": [
        "AI adoption",
        "measurement",
        "survey method"
      ],
      "relatedPages": [
        "/research/if-everyone-has-ai-where-is-the-advantage",
        "/research/how-long-before-you-know-if-an-ai-investment-worked"
      ]
    },
    {
      "id": "bucinca-2021",
      "citeAs": "https://thesuperskills.com/research/evidence#bucinca-2021",
      "section": "collaboration",
      "authors": "Bucinca, Z., Malaya, M. B. and Gajos, K. Z.",
      "year": 2021,
      "title": "To Trust or to Think: Cognitive Forcing Functions Can Reduce Overreliance on AI in AI-assisted Decision-making",
      "publication": "Proceedings of the ACM on Human-Computer Interaction, 5(CSCW1), April 2021. DOI 10.1145/3449287",
      "url": "https://arxiv.org/abs/2102.09692",
      "grade": "peer-reviewed",
      "method": "Experiment with 199 participants comparing three cognitive forcing interventions, designed from dual-process theory to compel more thoughtful engagement with AI-generated explanations, against two simple explainable-AI approaches and a no-AI baseline. Includes an audit for intervention-generated inequalities using the Need for Cognition scale.",
      "finding": "Cognitive forcing significantly reduced overreliance compared with the simple explainable-AI approaches. Participants gave the least favourable subjective ratings to the designs that reduced overreliance the most. On average the interventions benefited participants higher in Need for Cognition more, so human cognitive motivation moderates the effectiveness of explainable AI. The authors argue people rarely engage analytically with each individual recommendation and instead develop general heuristics about when to follow the AI.",
      "supports": "That deliberately adding friction to an AI-assisted decision measurably reduces acceptance of wrong suggestions, and that the intervention people dislike most is the one that works best.",
      "doesNotSupport": "That slower decisions produce better organisational outcomes. This is a controlled task with 199 participants, not a field study, and it measures overreliance rather than downstream results. The uneven benefit by Need for Cognition means the effect will not be uniform across a workforce.",
      "terms": [
        "cognitive forcing functions",
        "overreliance",
        "automation bias",
        "explainability"
      ],
      "relatedPages": [
        "/research/which-decisions-should-become-slower-because-of-ai",
        "/research/how-do-you-design-a-stop-button-people-will-use"
      ]
    },
    {
      "id": "chan-visibility-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#chan-visibility-2024",
      "section": "institutional",
      "authors": "Chan, A., Ezell, C., Kaufmann, M., Wei, K., Hammond, L., Bradley, H., Bluemke, E., Rajkumar, N., Krueger, D., Kolt, N., Heim, L. and Anderljung, M.",
      "year": 2024,
      "title": "Visibility into AI Agents",
      "publication": "Proceedings of the 2024 ACM Conference on Fairness, Accountability, and Transparency (FAccT 24), Rio de Janeiro, 3-6 June 2024. DOI 10.1145/3630106.3658948",
      "url": "https://facctconference.org/static/papers24/facct24-63.pdf",
      "grade": "peer-reviewed",
      "method": "Conference paper assessing three categories of measure for increasing visibility into deployed AI agents, across a spectrum of centralised to decentralised deployment contexts and accounting for hardware and software service providers in the supply chain. Analysis and proposal rather than measurement.",
      "finding": "Defines visibility as information about where, why, how and by whom AI agents are used, and assesses agent identifiers, real-time monitoring and activity logging as measures. Names five agent-specific risks: malicious use, overreliance and disempowerment, delayed and diffuse impacts, multi-agent risks, and sub-agents. On the last, the authors state that stopping an agent may require intervening on its sub-agents and that this may be difficult because, in their words, we lack methods for determining when an agent has created a sub-agent. The paper explicitly does not advocate immediate implementation of the measures and discusses their privacy and concentration-of-power costs.",
      "supports": "That the basic precondition of managing an agent, knowing what is running and what it has spawned, is an unsolved technical problem rather than a governance oversight.",
      "doesNotSupport": "That any of the proposed measures works, or is proportionate. The authors state they are describing options for further study rather than recommending deployment, and the paper contains no empirical evaluation.",
      "terms": [
        "AI agents",
        "visibility",
        "sub-agents",
        "monitoring",
        "accountability"
      ],
      "relatedPages": [
        "/research/who-manages-ai-agents"
      ]
    },
    {
      "id": "kolt-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#kolt-2025",
      "section": "institutional",
      "authors": "Kolt, N.",
      "year": 2025,
      "title": "Governing AI Agents",
      "publication": "Notre Dame Law Review, Vol. 101, forthcoming. Preprint arXiv:2501.07913, submitted 14 January 2025, revised 11 February 2025",
      "url": "https://arxiv.org/abs/2501.07913",
      "grade": "compiled-review",
      "method": "Legal and economic analysis applying the economic theory of principal-agent problems and the common law doctrine of agency relationships to AI agents. No empirical component.",
      "finding": "Characterises three problems arising from AI agents in agency terms: information asymmetry, discretionary authority and loyalty. Argues that the conventional solutions to agency problems, incentive design, monitoring and enforcement, might not be effective for governing AI agents that make uninterpretable decisions and operate at unprecedented speed and scale. Concludes that new technical and legal infrastructure is needed to support governance principles of inclusivity, visibility and liability.",
      "supports": "That a mature body of law and theory already exists for the question of who is accountable when something acts on your behalf, and that its standard remedies have identifiable failure points when the agent is artificial.",
      "doesNotSupport": "Anything about how agents behave in practice. It is a law review article arguing a position, cited here for its framework rather than as evidence of an outcome, and at the time of writing it is forthcoming rather than published.",
      "terms": [
        "AI agents",
        "principal-agent",
        "accountability",
        "agency law"
      ],
      "relatedPages": [
        "/research/who-manages-ai-agents"
      ]
    },
    {
      "id": "eu-ai-act-art-26",
      "citeAs": "https://thesuperskills.com/research/evidence#eu-ai-act-art-26",
      "section": "institutional",
      "authors": "European Union",
      "year": 2024,
      "title": "Regulation (EU) 2024/1689, Article 26: Obligations of deployers of high-risk AI systems",
      "publication": "Official Journal of the European Union. Chapter III, Section 3",
      "url": "https://artificialintelligenceact.eu/article/26/",
      "grade": "institutional-modelling",
      "method": "Binding regulation. Legal requirement rather than empirical finding. Text read at source.",
      "finding": "Paragraph 2 requires that deployers assign human oversight to natural persons who have the necessary competence, training and authority, as well as the necessary support. Paragraph 5 requires deployers to monitor operation on the basis of the instructions for use, to inform the provider and the relevant market surveillance authority without undue delay where the system presents a risk, and to suspend use of the system. Paragraph 6 requires retention of the automatically generated logs under the deployer's control for a period appropriate to the intended purpose and at least six months. Paragraph 7 requires employers, before putting a high-risk system into service at the workplace, to inform workers' representatives and the affected workers that they will be subject to its use. Paragraph 11 requires that natural persons subject to decisions made or assisted by an Annex III system be told.",
      "supports": "That the obligation to name a competent, authorised human overseer of a deployed system, and to be able to stop it, is law rather than good practice for systems in scope.",
      "doesNotSupport": "That it applies to most commercial agent deployments, which will fall outside the high-risk classification. Nor that any of it happens: the Regulation creates duties and does not evidence compliance. The application timetable has been subject to amendment and the published texts consulted for this entry did not agree on the dates, so no date is stated here.",
      "terms": [
        "human oversight",
        "deployer obligations",
        "AI governance",
        "accountability"
      ],
      "relatedPages": [
        "/research/who-manages-ai-agents",
        "/research/which-decisions-should-become-slower-because-of-ai"
      ]
    },
    {
      "id": "leo-xiv-magnifica-humanitas-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#leo-xiv-magnifica-humanitas-2026",
      "section": "institutional",
      "authors": "Leo XIV",
      "year": 2026,
      "title": "Magnifica Humanitas: Encyclical Letter on Safeguarding the Human Person in the Time of Artificial Intelligence",
      "publication": "The Holy See, given at Saint Peter's, 15 May 2026. 245 numbered paragraphs, 224 footnotes, five chapters",
      "url": "https://www.vatican.va/content/leo-xiv/en/encyclicals/documents/20260515-magnifica-humanitas.html",
      "grade": "institutional-modelling",
      "method": "Papal encyclical. Doctrinal and moral argument, read in full at source. Presents no original data and reports no study. It reasons from Catholic social teaching, explicitly continuing the line from Rerum Novarum (1891), whose 135th anniversary the signing date marks, through Laudato Si' and the 2025 Vatican note Antiqua et Nova.",
      "finding": "States that AI use can weaken human capability, in terms specific enough to quote. Paragraph 100: heavy reliance and the search for ready-made answers can \"weaken personal creativity and judgment\". Paragraph 140: \"every technology shapes those who use it\", educating people about AI \"involves teaching them to decide when and for what purpose it ought not to be used\", and the ease of obtaining answers or summaries risks \"extinguishing the desire to ask questions\". Paragraph 150, quoting Antiqua et Nova, holds that \"current approaches to technology can paradoxically de-skill workers, subject them to automated surveillance and relegate them to rigid and repetitive tasks\". Paragraph 106 argues that \"a slower pace in adopting AI does not mean opposing progress\". Paragraph 156: \"it is not enough to react only when jobs disappear; we must oversee the transformation in advance\". Paragraph 198: \"moral judgment cannot be reduced to calculation\", and lethal or otherwise irreversible decisions may not be entrusted to artificial systems.",
      "supports": "That deskilling, the loss of the impulse to ask a question, and the case for deliberately slowing adoption are now stated in a magisterial document addressed to a global audience, rather than only in the research literature and the trade press. It is evidence about the standing of the argument, not about the world.",
      "doesNotSupport": "Nothing empirical whatsoever. It measures nothing, samples nobody and tests no hypothesis, and its deskilling claim is a quotation from an earlier Vatican note which is itself not an empirical study. Citing it as evidence that AI de-skills workers would be a category error. Its authority is moral and institutional.",
      "terms": [
        "deskilling",
        "judgement",
        "human dignity",
        "AI governance",
        "education"
      ],
      "relatedPages": [
        "/research/the-best-writing-on-ai"
      ]
    },
    {
      "id": "liu-persistence-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#liu-persistence-2026",
      "section": "learning",
      "authors": "Liu, G., Christian, B., Dumbalska, T., Bakker, M. A. and Dubey, R.",
      "year": 2026,
      "title": "AI Assistance Reduces Persistence and Hurts Independent Performance",
      "publication": "arXiv:2604.04721, submitted 6 April 2026, revised 5 August 2026 (v4). Preprint, not peer reviewed",
      "url": "https://arxiv.org/abs/2604.04721",
      "grade": "working-paper",
      "method": "A series of randomised controlled trials on human-AI interactions, N = 1,222, across mathematical reasoning and reading comprehension. AI assistance is available during a practice phase and then withdrawn, and performance is measured unassisted.",
      "finding": "AI assistance improves performance in the short term, and people then perform significantly worse without AI and are more likely to give up. The authors report that these effects emerge after only brief interactions, approximately 10 minutes. They attribute the loss of persistence to AI conditioning people to expect immediate answers, denying them the experience of working through challenges on their own, and note that persistence is one of the strongest predictors of long-term learning.",
      "supports": "That a withdrawal effect can be produced causally, in a randomised design, and that it appears far faster than anyone had assumed. It also moves the mechanism from knowledge to persistence, which is a different and more portable claim.",
      "doesNotSupport": "Anything about sustained professional practice. These are short online tasks and the measured effect is a within-session carry-over rather than skill decay, so it cannot show whether the effect compounds, persists beyond the session or transfers to complex work. A preprint, not peer reviewed.",
      "terms": [
        "deskilling",
        "persistence",
        "unaided performance",
        "learning",
        "productive struggle"
      ],
      "relatedPages": [
        "/research/the-best-writing-on-ai"
      ]
    },
    {
      "id": "stromberg-lei-wu-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#stromberg-lei-wu-2026",
      "section": "learning",
      "authors": "Stromberg, D., Lei, V. and Wu, Y.",
      "year": 2026,
      "title": "The Generative AI Learning Penalty: Evidence from Chinese Secondary Education",
      "publication": "CEPR Discussion Paper 21577, CEPR Press, published 2 June 2026. Working paper, not peer reviewed",
      "url": "https://cepr.org/publications/dp21577",
      "grade": "working-paper",
      "method": "Thirty months of panel data on 26,811 Chinese students in grades 7 to 12, combining monthly closed-book exams, high-school and college entrance exams, and homework scores and completion time across nine subjects. Staggered AI adoption in a difference-in-differences design. The outcome measures are closed-book and invigilated, so they record unaided performance by construction.",
      "finding": "AI adoption raises homework scores by 18 per cent and reduces completion time by 30 per cent, and lowers monthly exam scores by 20 per cent within six months. High-stakes entrance-exam scores fall by 18 and 24 per cent, with the full penalty emerging only after about two years. Losses are largest in social science, then STEM, then languages, and are especially large for junior students, high-achieving students and boys. They concentrate among roughly 80 per cent of AI users whose behaviour is consistent with homework outsourcing, indicated by very short completion time coupled with high homework scores. Users who maintain similar completion time to non-users experience small losses.",
      "supports": "That the gap between assisted output and unaided capability can be measured at scale over years rather than minutes, and that it widens rather than closing. It also locates the damage in delegation rather than in access: the students who kept working at their normal pace were largely spared.",
      "doesNotSupport": "Causation with the confidence of a randomised trial. Adoption is self-selected and staggered rather than assigned, the outsourcing split is inferred from time-on-homework rather than observed, and a two-year lag makes contemporaneous confounds harder to exclude. One country, one school system, secondary students. Nothing about professional work. Not peer reviewed.",
      "terms": [
        "deskilling",
        "unaided performance",
        "learning",
        "productive struggle",
        "assessment"
      ],
      "relatedPages": [
        "/research/the-best-writing-on-ai"
      ]
    },
    {
      "id": "drew-2014",
      "citeAs": "https://thesuperskills.com/research/evidence#drew-2014",
      "section": "frontline",
      "authors": "Drew, B. J., Harris, P., Zegre-Hemsey, J. K., Mammone, T., Schindler, D., Salas-Boni, R. et al.",
      "year": 2014,
      "title": "Insights into the Problem of Alarm Fatigue with Physiologic Monitor Devices: A Comprehensive Observational Study of Consecutive Intensive Care Unit Patients",
      "publication": "PLoS ONE, 9(10), e110274. DOI 10.1371/journal.pone.0110274, published 22 October 2014",
      "url": "https://journals.plos.org/plosone/article?id=10.1371%2Fjournal.pone.0110274",
      "grade": "peer-reviewed",
      "method": "Observational study of consecutive adults in five adult intensive care units at the University of California, San Francisco, over the 31 days of March 2013. All monitor data, seven ECG leads, pressure, SpO2 and respiration waveforms, user settings and alarms, were stored for 461 patients. Nurse scientists annotated 12,671 arrhythmia alarms against a defined protocol, with inter-rater agreement of 95 per cent on true against false and a Cohen kappa of 0.86. Funded by GE Healthcare.",
      "finding": "2,558,760 unique alarms occurred in 31 days: 1,154,201 arrhythmia, 612,927 parameter and 791,632 technical. 381,560 were audible, an audible alarm burden of 187 per bed per day. 88.8 per cent of the 12,671 annotated arrhythmia alarms were false positives, and 93 per cent of the 168 true ventricular tachycardia alarms were not sustained long enough to warrant treatment.",
      "supports": "That a safety control firing at this rate trains the person holding it to ignore it, and that the training is rational rather than negligent. It is the best measured case anywhere of the base-rate problem that makes a stop control nominal.",
      "doesNotSupport": "Anything about AI. The 88.8 per cent applies only to the 12,671 annotated arrhythmia alarms, NOT to all 2.56 million alarms and not to clinical alarms in general; a page stating that 88.8 per cent of clinical alarms are false has misread it. Single centre, one month, five units, industry funded.",
      "terms": [
        "alarm fatigue",
        "false alarms",
        "human oversight",
        "base rate",
        "disuse"
      ],
      "relatedPages": [
        "/research/how-do-you-design-a-stop-button-people-will-use"
      ]
    },
    {
      "id": "joint-commission-sea50-2013",
      "citeAs": "https://thesuperskills.com/research/evidence#joint-commission-sea50-2013",
      "section": "institutional",
      "authors": "The Joint Commission",
      "year": 2013,
      "title": "Sentinel Event Alert 50: Medical device alarm safety in hospitals",
      "publication": "The Joint Commission, Issue 50, 8 April 2013",
      "url": "https://digitalassets.jointcommission.org/api/public/content/f65e5c9df2b94000a99445e0a7877007",
      "grade": "institutional-survey",
      "method": "Analysis of the Joint Commission's own Sentinel Event database, supplemented by FDA MAUDE reports and ECRI hazard rankings. Voluntary reporting; no sampling frame.",
      "finding": "98 alarm-related events between January 2009 and June 2012, of which 80 resulted in death, 13 in permanent loss of function and five in unexpected additional care or extended stay. 94 of the events occurred in hospitals. Contributing factors recorded as alarm signals inappropriately turned off (36), absent or inadequate alarm system (30), alarm signals not audible in all areas (25) and improper alarm settings (21). Reports that clinicians may turn the volume down, turn the alarm off, or set it outside safe limits in response to the volume of signals. Cites 566 alarm-related patient deaths in FDA MAUDE between January 2005 and June 2010. Led to National Patient Safety Goal NPSG.06.01.01, phased from 1 July 2014.",
      "supports": "That a regulator has documented, with named contributing factors, people disabling a safety control because it fired too often, and has had to legislate who holds the authority to change or silence it.",
      "doesNotSupport": "The size of the problem. The Commission's own footnote states that reporting is voluntary, represents only a small proportion of actual events, and that no conclusions should be drawn about relative frequency or trend. The widely quoted estimate that 85 to 99 per cent of alarm signals do not require clinical intervention is quoted BY the Commission from AAMI Horizons, Spring 2011, which is not a Joint Commission measurement and was not read for this entry.",
      "terms": [
        "alarm fatigue",
        "human oversight",
        "safety controls",
        "regulation"
      ],
      "relatedPages": [
        "/research/how-do-you-design-a-stop-button-people-will-use"
      ]
    },
    {
      "id": "ntsb-tempe-2019",
      "citeAs": "https://thesuperskills.com/research/evidence#ntsb-tempe-2019",
      "section": "institutional",
      "authors": "National Transportation Safety Board",
      "year": 2019,
      "title": "Collision Between Vehicle Controlled by Developmental Automated Driving System and Pedestrian, Tempe, Arizona, March 18, 2018",
      "publication": "NTSB Highway Accident Report NTSB/HAR-19/03, PB2019-101402, case HWY18MH010",
      "url": "https://www.ntsb.gov/investigations/AccidentReports/Reports/HAR1903.pdf",
      "grade": "statutory-investigation",
      "method": "Statutory accident investigation of a single fatal collision, with access to vehicle system data, operator records and the developer's internal procedures.",
      "finding": "The automated driving system detected the pedestrian 5.6 seconds before impact and tracked her to the crash without ever classifying her correctly or predicting her path. The developer had disengaged the Volvo XC90's factory forward collision warning and automatic emergency braking during automated operation. On detecting an emergency the system entered a one-second period of action suppression, withholding braking while it verified the hazard or the operator took control, and NTSB record that no alert was given to the operator when action suppression was initiated. The system recognised an imminent collision 1.2 seconds before impact. Probable cause was determined as the operator's failure to monitor the driving environment while visually distracted by a personal phone, with contributing factors including inadequate safety risk assessment procedures, ineffective oversight of vehicle operators and lack of adequate mechanisms for addressing operators' automation complacency.",
      "supports": "That a design can name a human as the primary countermeasure in an emergency and simultaneously withhold the alert that would let them act as one, and that the stated reason for doing so was concern about false alarms.",
      "doesNotSupport": "That the absence of an alert caused the crash. NTSB determined the probable cause to be the operator's inattention and found she would likely have had sufficient time to react had she been attentive. One vehicle, one developer, one jurisdiction, developmental software from 2018.",
      "terms": [
        "automation complacency",
        "human oversight",
        "stop controls",
        "alerting",
        "automation bias"
      ],
      "relatedPages": [
        "/research/how-do-you-design-a-stop-button-people-will-use"
      ]
    },
    {
      "id": "weber-stop-work-2018",
      "citeAs": "https://thesuperskills.com/research/evidence#weber-stop-work-2018",
      "section": "frontline",
      "authors": "Weber, D. E., MacGregor, S. C., Provan, D. J. and Rae, A.",
      "year": 2018,
      "title": "'We can stop work, but then nothing gets done.' Factors that support and hinder a workforce to discontinue work for safety",
      "publication": "Safety Science, 108, 149-160. Safety Science Innovation Lab, Griffith University",
      "url": "https://forgeworks.com/wp-content/uploads/2022/09/Authority-to-Stop-Work.pdf",
      "grade": "peer-reviewed",
      "method": "Qualitative study. Ten focus groups with workers in a range of roles in the liquefied petroleum gas industry, examining an explicit organisational Authority to Stop an Unsafe Task.",
      "finding": "Stopping work for safety was reported as challenging at the sharp operational end despite the authority existing and carrying no formal penalty. The authors conclude that stopping an unsafe task 'does not solely hinge on the willingness of individual workers to stop, but also depends on contextual factors surrounding the stop work decision'. A participant's account supplies the title: the authority exists, using it produces no drama, and then nothing gets done, so the work resumes as before.",
      "supports": "That granting an authority is not the same as making it usable, and that the decisive factor is what happens to the work and the worker afterwards rather than the existence of the permission.",
      "doesNotSupport": "Any rate or frequency. Ten focus groups in one industry in one country, qualitative by design, with no measurement of how often stops occurred or should have.",
      "terms": [
        "stop-work authority",
        "safety culture",
        "human oversight",
        "organisational design"
      ],
      "relatedPages": [
        "/research/how-do-you-design-a-stop-button-people-will-use"
      ]
    },
    {
      "id": "garicano-2000",
      "citeAs": "https://thesuperskills.com/research/evidence#garicano-2000",
      "section": "work",
      "authors": "Garicano, L.",
      "year": 2000,
      "title": "Hierarchies and the Organization of Knowledge in Production",
      "publication": "Journal of Political Economy, 108(5), 874-904. University of Chicago Press",
      "url": "https://www.journals.uchicago.edu/doi/abs/10.1086/317671",
      "grade": "peer-reviewed",
      "method": "Theoretical model of knowledge acquisition and problem-solving in production, with communication costs and knowledge acquisition costs traded off against each other. No empirical estimation.",
      "finding": "A knowledge-based hierarchy is a natural way to organise the acquisition of knowledge when matching problems with those who know how to solve them is costly. Production workers acquire knowledge of the most common or easiest problems and refer exceptions upward to specialist problem solvers, with problems passed on until somebody solves them or the conditional probability of a solution is too low to justify continuing. Adding layers of problem solvers raises the utilisation rate of knowledge and economises on knowledge acquisition, at the cost of increasing the communication required.",
      "supports": "That the number of layers in an organisation is a function of two costs rather than of custom, which gives a testable prediction for what happens when either cost falls. It is the model every subsequent claim about AI flattening organisations implicitly relies on.",
      "doesNotSupport": "Anything measured. It is a model, published a quarter of a century before generative AI, and it contains no data about firms, layers or technology adoption.",
      "terms": [
        "knowledge hierarchy",
        "organisational design",
        "span of control",
        "problem solving"
      ],
      "relatedPages": [
        "/research/what-happens-to-work-that-moves-information"
      ]
    },
    {
      "id": "bloom-ict-2014",
      "citeAs": "https://thesuperskills.com/research/evidence#bloom-ict-2014",
      "section": "work",
      "authors": "Bloom, N., Garicano, L., Sadun, R. and Van Reenen, J.",
      "year": 2014,
      "title": "The Distinct Effects of Information Technology and Communication Technology on Firm Organization",
      "publication": "Management Science, 60(12), 2859-2885. DOI 10.1287/mnsc.2014.2013. Earlier version NBER Working Paper 14975, May 2009",
      "url": "https://researchonline.lse.ac.uk/id/eprint/79246/",
      "grade": "peer-reviewed",
      "method": "Survey data on worker and plant manager autonomy and span of control, combined with measures of technology adoption, and instrumented using distance from ERP's place of origin and heterogeneous telecommunication costs arising from regulation. The working-paper version describes approximately 1,000 manufacturing firms with 100 to 5,000 employees across the US, France, Germany, Italy, Poland, Portugal, Sweden and the UK, drawn from the CEP double-blind management survey, with technology data from the Harte-Hanks ICT panel.",
      "finding": "Information technology is a decentralising force and communication technology is a centralising force. Better information technologies, ERP for plant managers and computer-assisted design or manufacturing for production workers, are associated with more autonomy and a wider span of control. Technologies that improve communication, such as data intranets, decrease autonomy for workers and plant managers. Instrumenting strengthens the result.",
      "supports": "That 'technology' has no single organisational effect, and that predicting what a tool does to authority requires knowing whether it lowers the cost of knowing or the cost of telling.",
      "doesNotSupport": "Anything about generative AI, which is both kinds of technology in one interface. Manufacturing plants only, data predating 2009. The sample description above is verified in the working paper; the published article abstract says only 'American and European manufacturing firms', and the typeset article could not be opened at the publisher.",
      "terms": [
        "decentralisation",
        "organisational design",
        "span of control",
        "autonomy"
      ],
      "relatedPages": [
        "/research/what-happens-to-work-that-moves-information"
      ]
    },
    {
      "id": "ewens-giroud-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#ewens-giroud-2025",
      "section": "work",
      "authors": "Ewens, M. and Giroud, X.",
      "year": 2025,
      "title": "Corporate Hierarchy",
      "publication": "NBER Working Paper 34162, issued August 2025, revised October 2025. DOI 10.3386/w34162. Not peer reviewed",
      "url": "https://www.nber.org/papers/w34162",
      "grade": "working-paper",
      "method": "A measure of corporate hierarchy for over 3,100 US public firms, built from online resumes of 7 million US-based workers, 2016 to 2023, using a network estimation technique to identify hierarchical layers. AI adoption is proxied by AI job postings following Babina, Fedyk, He and Hodson.",
      "finding": "Firms average ten hierarchical layers and a pyramidal structure, with the average and median number of layers declining across the sample period. More hierarchical firms show a more educated workforce, higher internal promotion rates, longer tenure, higher operating performance and higher administrative costs. Companies flattened their hierarchies following adoption of AI technologies, while pharmaceutical companies added layers after Covid-19. The authors state the AI tests are under-powered and that point estimates are significant at the 10 per cent level regardless of the adoption metric.",
      "supports": "That the Garicano prediction has now been tested against firm-level data and points in the predicted direction, which is more than the flattening discourse previously had.",
      "doesNotSupport": "That AI causes flattening. The authors call their own tests under-powered at the 10 per cent level, adoption is measured by job postings rather than by use, hierarchy is inferred from self-reported resumes, and the sample is US public firms. Not peer reviewed. Figures of 2,500 firms and 16 million employees circulate from an earlier draft and are wrong for this version.",
      "terms": [
        "organisational design",
        "flattening",
        "middle management",
        "AI adoption"
      ],
      "relatedPages": [
        "/research/what-happens-to-work-that-moves-information"
      ]
    },
    {
      "id": "babina-hierarchy-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#babina-hierarchy-2025",
      "section": "work",
      "authors": "Babina, T., Fedyk, A., He, A. and Hodson, J.",
      "year": 2025,
      "title": "Firm Investments in Artificial Intelligence Technologies and Changes in Workforce Composition",
      "publication": "Chapter 3 in Technology, Productivity, and Economic Growth, NBER Studies in Income and Wealth 83, University of Chicago Press, July 2025. Earlier version NBER Working Paper 31325",
      "url": "https://www.nber.org/books-and-chapters/technology-productivity-and-economic-growth/firm-investments-artificial-intelligence-technologies-and-changes-workforce-composition",
      "grade": "peer-reviewed",
      "method": "Worker resume and job-posting datasets combined to measure firm-level AI investment against workforce composition variables including educational attainment, specialisation and hierarchy.",
      "finding": "AI investments are associated with a flattening of firms' hierarchical structure, with significant increases in the share of workers at the junior level and decreases in the shares in middle-management and senior roles.",
      "supports": "That the composition of the flattening, and not only its existence, is measurable, and that the layer shrinking is the one that has historically developed juniors into seniors.",
      "doesNotSupport": "Any magnitude. The abstract read for this entry states direction and significance and no percentage, so none should be attributed to it. Association rather than causation, and firms that invest in AI differ from those that do not in many other ways.",
      "terms": [
        "middle management",
        "workforce composition",
        "flattening",
        "AI investment"
      ],
      "relatedPages": [
        "/research/what-happens-to-work-that-moves-information"
      ]
    },
    {
      "id": "yang-remote-2022",
      "citeAs": "https://thesuperskills.com/research/evidence#yang-remote-2022",
      "section": "work",
      "authors": "Yang, L., Holtz, D., Jaffe, S., Suri, S., Sinha, S., Weston, J., Joyce, C., Shah, N., Sherman, K., Hecht, B. and Teevan, J.",
      "year": 2022,
      "title": "The effects of remote work on collaboration among information workers",
      "publication": "Nature Human Behaviour, 6, 43-54. DOI 10.1038/s41562-021-01196-4, published online 9 September 2021",
      "url": "https://www.nature.com/articles/s41562-021-01196-4",
      "grade": "peer-reviewed",
      "method": "Observed telemetry on emails, calendars, instant messages, video and audio calls and workweek hours of 61,182 US Microsoft employees over the first six months of 2020, using workers already remote before the pandemic as a comparison to separate firm-wide remote work from other pandemic effects.",
      "finding": "Firm-wide remote work caused the collaboration network to become more static and siloed, with fewer bridges between disparate parts of the organisation, a decrease in synchronous and an increase in asynchronous communication. The authors state these effects may make it harder for employees to acquire and share new information across the network.",
      "supports": "That changing how information moves reorganises who knows what, without anybody redesigning a role, and that the change is measurable in the network rather than only in self-report.",
      "doesNotSupport": "Anything about AI, and nothing about outcomes. It measures communication structure rather than performance, in one very large technology company, during a pandemic. Full text is paywalled; this entry rests on the published abstract.",
      "terms": [
        "collaboration",
        "information flow",
        "organisational network",
        "tacit knowledge"
      ],
      "relatedPages": [
        "/research/what-happens-to-work-that-moves-information"
      ]
    },
    {
      "id": "noy-zhang-2023",
      "citeAs": "https://thesuperskills.com/research/evidence#noy-zhang-2023",
      "section": "collaboration",
      "authors": "Noy, S. and Zhang, W.",
      "year": 2023,
      "title": "Experimental evidence on the productivity effects of generative artificial intelligence",
      "publication": "Science, 381(6654), 187-192, 13 July 2023. DOI 10.1126/science.adh2586. Preregistered, AEA RCT Registry trial 10882",
      "url": "https://www.science.org/doi/10.1126/science.adh2586",
      "grade": "peer-reviewed",
      "method": "Preregistered online randomised experiment. 453 college-educated professionals, including marketers, grant writers, consultants, data analysts, HR professionals and managers, given occupation-specific incentivised writing tasks of 20 to 30 minutes. Half were randomly exposed to ChatGPT. Run 27 January to 21 February 2023 with GPT-3.5.",
      "finding": "Average time taken fell by 40 per cent and output quality rose by 18 per cent. Time on the post-treatment task dropped by 11 minutes, 0.75 standard deviations, against a control mean of 27 minutes, and evaluator grades rose by 0.45 standard deviations. Inequality between workers decreased: in the treatment group initial inequalities were more than half-erased, with the correlation between first-task and second-task grades falling to 0.14.",
      "supports": "That the tool compresses the performance distribution on tasks it does well, raising the floor much more than the ceiling. That is a pricing fact about expertise before it is a productivity fact.",
      "doesNotSupport": "That the effect generalises. The authors say they examined a limited range of occupations and tasks in which ChatGPT may be unusually useful, and speculate that real-economy effects will be somewhat lower. Short one-off tasks, online sample, an early model. Nothing about capability retention.",
      "terms": [
        "productivity",
        "compression",
        "inequality",
        "writing",
        "expertise"
      ],
      "relatedPages": [
        "/research/what-becomes-more-valuable-as-ai-gets-cheaper",
        "/research/does-ai-actually-make-people-more-productive"
      ]
    },
    {
      "id": "peng-copilot-2023",
      "citeAs": "https://thesuperskills.com/research/evidence#peng-copilot-2023",
      "section": "collaboration",
      "authors": "Peng, S., Kalliamvakou, E., Cihon, P. and Demirer, M.",
      "year": 2023,
      "title": "The Impact of AI on Developer Productivity: Evidence from GitHub Copilot",
      "publication": "arXiv:2302.06590, 13 February 2023. DOI 10.48550/arXiv.2302.06590. Authors at Microsoft Research, GitHub and MIT Sloan. Not peer reviewed",
      "url": "https://arxiv.org/abs/2302.06590",
      "grade": "working-paper",
      "method": "Randomised controlled trial. 95 professional programmers recruited through Upwork from 166 offers, randomised to 45 treated and 50 control, given a standardised task of implementing an HTTP server in JavaScript. 35 in each group completed the task and survey.",
      "finding": "Conditional on completion, the treated group averaged 71.17 minutes against 160.89 for control, a 55.8 per cent reduction in completion time, p = 0.0017, with a 95 per cent confidence interval on the improvement of 21 to 89 per cent. Participants in both groups estimated a 35 per cent productivity increase, which the authors describe as an underestimation of the 55.8 per cent revealed increase.",
      "supports": "A large measured speed gain on a well-specified task, and, separately, that self-reported productivity can UNDERSTATE the measured effect. Any claim that self-report systematically overstates gains has to answer this result.",
      "doesNotSupport": "Anything about quality. The authors state the study does not examine the effects of AI on code quality. The effect is estimated on 35 completers per arm rather than 95 participants, the task is standardised and greenfield rather than work inside a mature codebase, the authors are employed by the vendor and its parent. Not peer reviewed.",
      "terms": [
        "productivity",
        "software development",
        "self-report",
        "perception gap"
      ],
      "relatedPages": [
        "/research/what-becomes-more-valuable-as-ai-gets-cheaper",
        "/research/does-ai-actually-make-people-more-productive",
        "/research/will-ai-replace-programmers"
      ]
    },
    {
      "id": "census-btos-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#census-btos-2024",
      "section": "work",
      "authors": "Bonney, K., Breaux, C., Buffington, C., Dinlersoz, E., Foster, L., Goldschlag, N., Haltiwanger, J., Kroff, Z. and Savage, K.",
      "year": 2024,
      "title": "Tracking Firm Use of AI in Real Time: A Snapshot from the Business Trends and Outlook Survey",
      "publication": "US Census Bureau, Center for Economic Studies Working Paper CES 24-16, March 2024; also NBER Working Paper 32319. Not peer reviewed",
      "url": "https://www.census.gov/hfp/btos/downloads/CES-WP-24-16.pdf",
      "grade": "working-paper",
      "method": "Analysis of the AI questions in the Business Trends and Outlook Survey, a high-frequency nationally representative survey of US firms, over the collection period covered by the paper.",
      "finding": "Bi-weekly estimates of the AI use rate rose from 3.7 to 5.4 per cent, with an expected rate of about 6.6 per cent by early autumn 2024. 94.6 per cent of AI-using businesses reported no net change in employment in the previous six months attributable to AI use. On discontinuation, which the survey does not ask about directly, 67.9 per cent of current AI users expect to use it in future, 14.5 per cent do not, and 17.6 per cent do not know, so about one in seven current users may de-adopt; the authors attribute this to experimentation that does not yield anticipated benefits or organisational synergies. The authors state explicitly that the analysis does not seek to identify a causal link between AI use and firm performance, only whether use is associated with better performance in general, and that causal analysis awaits integrated data and repeat collections after about three to four years.",
      "supports": "That the body with the best firm-level data in the world declines to make the causal claim that consultancy and vendor reporting makes routinely, and says in its own text what would be required to make it.",
      "doesNotSupport": "Any effect of AI on firm performance, by the authors' own statement. Adoption rates here are not comparable with later Census figures, because the question was broadened in November 2025 from use in producing goods or services to use in any business function. Census working papers carry the standard disclaimer that they have not undergone the review accorded Census Bureau publications.",
      "terms": [
        "AI adoption",
        "measurement",
        "causal inference",
        "firm performance"
      ],
      "relatedPages": [
        "/research/what-becomes-more-valuable-as-ai-gets-cheaper",
        "/research/which-ai-investments-should-we-stop"
      ]
    },
    {
      "id": "metr-selfreport-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#metr-selfreport-2026",
      "section": "collaboration",
      "authors": "Becker, J. (METR)",
      "year": 2026,
      "title": "Measuring the Self-Reported Impact of Early-2026 AI on Technical Worker Productivity",
      "publication": "METR, 11 May 2026. Not peer reviewed",
      "url": "https://metr.org/blog/2026-05-11-ai-usage-survey/",
      "grade": "working-paper",
      "method": "Survey of 349 technical workers on self-reported change in speed and in value of work attributable to AI tools, set against the published field-experiment literature.",
      "finding": "Median self-reported change in the VALUE of work is 1.4 to 2 times, and median self-reported SPEED change is 3 times. METR state that to their knowledge only one study has gathered survey and field experiment results on the same population and metric, Becker et al. (2025), which finds developers overestimate productivity gains by over 40 percentage points. They add that public survey estimates have tended to exceed field-experiment estimates, while stating it is difficult to determine the extent to which surveys overestimate gains relative to experimental data, and that speed measures are likely biased upwards relative to value measures.",
      "supports": "That the evidence base for the most repeated claim in enterprise AI, that surveys overstate productivity, rests on a single population, and that the people who own that finding say so themselves. It also separates speed from value, which almost no business case does.",
      "doesNotSupport": "That self-report always overstates. The GitHub Copilot randomised trial found the opposite direction, with participants estimating 35 per cent against a measured 55.8 per cent. A survey of a self-selected technical population, not peer reviewed, and its own authors decline the general claim.",
      "terms": [
        "self-report",
        "productivity",
        "measurement",
        "perception gap"
      ],
      "relatedPages": [
        "/research/what-becomes-more-valuable-as-ai-gets-cheaper",
        "/research/does-ai-actually-make-people-more-productive"
      ]
    },
    {
      "id": "keil-1995",
      "citeAs": "https://thesuperskills.com/research/evidence#keil-1995",
      "section": "institutional",
      "authors": "Keil, M.",
      "year": 1995,
      "title": "Pulling the Plug: Software Project Management and the Problem of Project Escalation",
      "publication": "MIS Quarterly, 19(4), 421-447",
      "url": "https://misq.umn.edu/pulling-the-plug-software-project-management-and-the-problem-of-project-escalation.html",
      "grade": "peer-reviewed",
      "method": "Longitudinal exploratory single-case study of one IT project inside a large computer manufacturer, pseudonymised as CompuSys. 111 interviews across eight job functions, 19 observed meetings, more than 350 collected documents.",
      "finding": "CONFIG, an expert system built to help sales representatives produce error-free configurations before quoting, ran for over a decade and was terminated at the end of 1992 after, in the author's words, tens of millions of dollars. Successive business cases put net present value at 43.9 million dollars in 1982, 55.7 million in 1985 and at least 41.1 million in 1987. Keil concludes escalation is promoted by a combination of project, psychological, social and organisational factors rather than by any one of them.",
      "supports": "That the canonical study of a technology programme nobody could stop is a study of an artificial intelligence programme. The de-escalation literature is not being applied to AI by analogy; it began there and the field forgot.",
      "doesNotSupport": "Any prevalence. N is one, the organisation is pseudonymised, and there is no comparison group. An Academia.edu machine-generated summary of this paper asserts the project absorbed 250 million dollars; the article says tens of millions, and the 250 figure in it is 250 billion, being 1994 total US IT applications spending.",
      "terms": [
        "escalation of commitment",
        "project failure",
        "expert systems",
        "sunk cost"
      ],
      "relatedPages": [
        "/research/which-ai-investments-should-we-stop"
      ]
    },
    {
      "id": "keil-mann-rai-2000",
      "citeAs": "https://thesuperskills.com/research/evidence#keil-mann-rai-2000",
      "section": "institutional",
      "authors": "Keil, M., Mann, J. and Rai, A.",
      "year": 2000,
      "title": "Why Software Projects Escalate: An Empirical Analysis and Test of Four Theoretical Models",
      "publication": "MIS Quarterly, 24(4), 631-664",
      "url": "https://misq.umn.edu/why-software-projects-escalate-an-empirical-analysis-and-test-of-four-theoretical-models.html",
      "grade": "peer-reviewed",
      "method": "Survey of information systems audit and control professionals, designed to gather data on projects that did not escalate as well as those that did, testing four theories: self-justification, prospect, agency and approach-avoidance.",
      "finding": "The authors state that between 30 and 40 per cent of all IS projects exhibit some degree of escalation. The completion effect derived from approach-avoidance theory gave the best classification, correctly classifying over 70 per cent of both escalated and non-escalated projects.",
      "supports": "That escalation is common enough to be a base condition rather than an exception, and that the best-supported explanation is the pull of finishing rather than the psychology of self-justification alone.",
      "doesNotSupport": "Anything about AI, and nothing about whether escalated projects should have been stopped: it shows their outcomes were worse. Some degree of escalation is a soft threshold, and the respondents are auditors reporting retrospectively rather than a random sample of projects. The published sample size sits in the full text, which could not be opened during the 4 September 2026 build; the prevalence figure is quoted from the abstract and pages citing it say so.",
      "terms": [
        "escalation of commitment",
        "project failure",
        "prevalence",
        "governance"
      ],
      "relatedPages": [
        "/research/which-ai-investments-should-we-stop"
      ]
    },
    {
      "id": "montealegre-keil-2000",
      "citeAs": "https://thesuperskills.com/research/evidence#montealegre-keil-2000",
      "section": "institutional",
      "authors": "Montealegre, R. and Keil, M.",
      "year": 2000,
      "title": "De-escalating Information Technology Projects: Lessons from the Denver International Airport",
      "publication": "MIS Quarterly, 24(3), 417-447",
      "url": "https://misq.umn.edu/de-escalating-information-technology-projects-lessons-from-the-denver-international-airport.html",
      "grade": "peer-reviewed",
      "method": "Longitudinal qualitative case study of the automated baggage handling system at Denver International Airport, used to induce a process model of de-escalation.",
      "finding": "De-escalation runs as a four-phase process: problem recognition, re-examination of prior course of action, search for alternative course of action, and implementing an exit strategy. The authors note that while escalation is well researched, there has been comparatively little research on the process of breaking the cycle.",
      "supports": "That climbing back down has a describable structure, and that the hard step is the second one, because re-examining the prior course of action requires the person who chose it to permit the question.",
      "doesNotSupport": "How often de-escalation is attempted or succeeds. N is one, the model is inductive rather than tested, and the case is a physical baggage system in the 1990s. The authors describe four phases, not stages.",
      "terms": [
        "de-escalation",
        "project termination",
        "exit strategy",
        "governance"
      ],
      "relatedPages": [
        "/research/which-ai-investments-should-we-stop"
      ]
    },
    {
      "id": "staw-1976",
      "citeAs": "https://thesuperskills.com/research/evidence#staw-1976",
      "section": "judgement",
      "authors": "Staw, B. M.",
      "year": 1976,
      "title": "Knee-Deep in the Big Muddy: A Study of Escalating Commitment to a Chosen Course of Action",
      "publication": "Organizational Behavior and Human Performance, 16(1), 27-44",
      "url": "https://doi.org/10.1016/0030-5073(76)90005-2",
      "grade": "peer-reviewed",
      "method": "Role-played corporate financial decision, 240 business students, two-by-two factorial crossing personal responsibility for the initial investment against positive or negative decision consequences.",
      "finding": "Participants personally responsible for the earlier investment allocated an average of 11.08 million dollars to the division they had chosen, against 8.89 million where another officer had chosen it. Negative consequences drew 11.20 million against 8.77 million for positive ones. Where a participant's own earlier choice had subsequently declined, the figure rose to 13.07 million. Both main effects and the interaction were significant.",
      "supports": "That responsibility for the original decision changes the next one, in a known direction, before any question of competence arises. This is why asking a programme's sponsor whether to continue it is not an assessment.",
      "doesNotSupport": "Anything about real organisations or real money. A single-session paper exercise with undergraduates, no longitudinal element, and no prevalence claim. Staw himself flags the ambiguity between self-justification and self-perception as the mechanism.",
      "terms": [
        "escalation of commitment",
        "sunk cost",
        "decision-making",
        "accountability"
      ],
      "relatedPages": [
        "/research/which-ai-investments-should-we-stop"
      ]
    },
    {
      "id": "rand-ai-failure-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#rand-ai-failure-2024",
      "section": "institutional",
      "authors": "Ryseff, J., De Bruhl, B. and Newberry, S. J.",
      "year": 2024,
      "title": "The Root Causes of Failure for Artificial Intelligence Projects and How They Can Succeed",
      "publication": "RAND Corporation, RR-A2680-1",
      "url": "https://www.rand.org/content/dam/rand/pubs/research_reports/RRA2600/RRA2680-1/RAND_RRA2680-1.pdf",
      "grade": "institutional-survey",
      "method": "Qualitative root-cause study. Interviews with 65 data scientists and engineers with at least five years building AI and machine-learning models in industry or academia.",
      "finding": "A set of organisational anti-patterns behind AI project failure, chiefly misunderstood or miscommunicated intent, inadequate data, focus on the technology rather than the problem, missing infrastructure, and problems the technology cannot solve. The report opens by stating that by some estimates more than 80 per cent of AI projects fail, twice the rate of non-AI corporate IT projects.",
      "supports": "That the most repeated statistic about AI project failure does not come from a study. RAND footnote the 80 per cent to a press article and the comparison to a business magazine piece, and hedge it as by some estimates. Everyone quoting RAND drops the hedge.",
      "doesNotSupport": "Any base rate. This is 65 expert interviews about causes, not a measurement of frequency, and RAND do not claim otherwise. The two academic sources it cites on failure factors are themselves expert-interview studies rather than prevalence studies.",
      "terms": [
        "AI project failure",
        "base rates",
        "citation provenance",
        "governance"
      ],
      "relatedPages": [
        "/research/which-ai-investments-should-we-stop"
      ]
    },
    {
      "id": "dekker-woods-2002",
      "citeAs": "https://thesuperskills.com/research/evidence#dekker-woods-2002",
      "section": "collaboration",
      "authors": "Dekker, S. W. A. and Woods, D. D.",
      "year": 2002,
      "title": "MABA-MABA or Abracadabra? Progress on Human-Automation Co-ordination",
      "publication": "Cognition, Technology & Work, 4(4), 240-244",
      "url": "https://doi.org/10.1007/s101110200022",
      "grade": "peer-reviewed",
      "method": "Conceptual analysis of function allocation methods in human factors. No data, no participants, no experiment.",
      "finding": "Substitution-based function allocation, of which the Fitts list is the archetype, cannot deliver human-automation coordination, because the effects of automation are qualitative rather than quantitative. The authors name the underlying assumption the substitution myth and write that capitalising on a strength of automation does not replace a human weakness but creates new human strengths and weaknesses, often in unanticipated ways. Allocating a function also creates new functions for the other partner that did not exist before. They add that neither the list nor much of the supervisory control literature explains the cognitive work involved in deciding how and when to intervene or how to switch from level to level.",
      "supports": "That splitting the tasks is the wrong unit of design, and that the coordination at the boundary has been the acknowledged gap in this literature since 2002.",
      "doesNotSupport": "How large the effect is, or anything measurable at all. It is an argument. Its supporting accident examples are cited rather than analysed.",
      "terms": [
        "function allocation",
        "substitution myth",
        "human-automation coordination",
        "handoffs"
      ],
      "relatedPages": [
        "/research/how-do-humans-and-agents-divide-work"
      ]
    },
    {
      "id": "cook-render-woods-2000",
      "citeAs": "https://thesuperskills.com/research/evidence#cook-render-woods-2000",
      "section": "frontline",
      "authors": "Cook, R. I., Render, M. and Woods, D. D.",
      "year": 2000,
      "title": "Gaps in the continuity of care and progress on patient safety",
      "publication": "BMJ, 320(7237), 791-794",
      "url": "https://doi.org/10.1136/bmj.320.7237.791",
      "grade": "peer-reviewed",
      "method": "Conceptual paper in the education and debate section, developed against two publicly documented US clinical accidents.",
      "finding": "Gaps are defined as discontinuities in care, appearing as losses of information or momentum or interruptions in delivery. Most gaps are anticipated and bridged by practitioners, so invisibly that neither outsiders nor insiders recognise the activity as distinct work. Bridging is not elimination: some bridges are frail and easily undone. Accidents occur when conditions overwhelm the mechanisms practitioners use to detect and bridge gaps, from which the authors conclude that efforts to forestall errors by isolating practitioners from the system will misfire. Their worked example is the division of nursing work with less credentialed technicians, which delivers a real economic benefit and restricts the nurse's ability to anticipate gaps.",
      "supports": "That the boundary between steps in a process is a named object with its own failure modes, and that the work of bridging it is invisible to every measure of output. The nursing example transfers directly to delegating work to agents.",
      "doesNotSupport": "Anything quantified. There is no sample, no rate and no measurement, and nothing in it concerns automation or AI. It is a framing paper.",
      "terms": [
        "handoffs",
        "continuity",
        "patient safety",
        "invisible work",
        "oversight"
      ],
      "relatedPages": [
        "/research/how-do-humans-and-agents-divide-work"
      ]
    },
    {
      "id": "starmer-ipass-2014",
      "citeAs": "https://thesuperskills.com/research/evidence#starmer-ipass-2014",
      "section": "frontline",
      "authors": "Starmer, A. J., Spector, N. D., Srivastava, R., West, D. C., Rosenbluth, G., Allen, A. D. et al. for the I-PASS Study Group",
      "year": 2014,
      "title": "Changes in Medical Errors after Implementation of a Handoff Program",
      "publication": "New England Journal of Medicine, 371(19), 1803-1812",
      "url": "https://doi.org/10.1056/NEJMsa1405556",
      "grade": "peer-reviewed",
      "method": "Prospective systems-based intervention study across nine paediatric residency programmes in the US and Canada, January 2011 to May 2013, in three staggered waves with season-matched six-month pre and post periods. 875 consenting residents, 10,740 patient admissions. Active surveillance five days a week, incidents classified by two blinded physician reviewers. Not randomised, no control group.",
      "finding": "The medical-error rate fell 23 per cent, from 24.5 to 18.8 per 100 admissions, and preventable adverse events fell 30 per cent, from 4.7 to 3.3 per 100 admissions, both P<0.001. Near misses and non-harmful errors fell 21 per cent. Non-preventable adverse events did not change, 3.0 against 2.8, P=0.79. Oral handoff duration did not change, 2.4 against 2.5 minutes per patient, P=0.55. Error rates did not change significantly at three of the nine sites, although written and oral handoff processes improved at all nine.",
      "supports": "That a handoff can be made substantially safer without being made longer, by changing what is transferred rather than how much time is spent transferring it. The unchanged non-preventable rate is the strongest internal evidence the effect is real.",
      "doesNotSupport": "Causation, by the authors' own statement, and not which element of the bundle did the work. Paediatric inpatient units only; generalisation to other specialties is untested. Reviewer agreement was moderate, kappa 0.47 for error classification. Data collectors could not be blinded to period. Funding included an unrestricted medical education grant from Pfizer.",
      "terms": [
        "handoffs",
        "patient safety",
        "process design",
        "communication"
      ],
      "relatedPages": [
        "/research/how-do-humans-and-agents-divide-work"
      ]
    },
    {
      "id": "cemri-mast-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#cemri-mast-2025",
      "section": "collaboration",
      "authors": "Cemri, M., Pan, M. Z., Yang, S., Agrawal, L. A., Chopra, B., Tiwari, R., Keutzer, K., Parameswaran, A., Klein, D., Ramchandran, K., Zaharia, M., Gonzalez, J. E. and Stoica, I.",
      "year": 2025,
      "title": "Why Do Multi-Agent LLM Systems Fail?",
      "publication": "NeurIPS 2025 Datasets and Benchmarks Track. arXiv:2503.13657",
      "url": "https://arxiv.org/abs/2503.13657",
      "grade": "peer-reviewed",
      "method": "Empirical failure taxonomy built from expert annotation of 150 execution traces at inter-annotator kappa 0.88, then applied at scale with LLM-as-judge across more than 1,600 traces from seven multi-agent frameworks. Failure distribution computed on 210 traces.",
      "finding": "Failures divide into system design issues 41.8 per cent, inter-agent misalignment 36.9 per cent and task verification 21.3 per cent. Inter-agent misalignment is defined as a breakdown in critical information flow during interaction and coordination, and comprises conversation resets 2.20 per cent, proceeding on wrong assumptions rather than seeking clarification 6.80 per cent, task derailment 7.40 per cent, withholding information another agent needed 0.85 per cent, ignoring another agent's input 1.90 per cent, and mismatch between reasoning and action 13.2 per cent. The authors state that context and communication protocols are often insufficient, because the errors occur even when agents in the same framework communicate in natural language. Improving role specification alone raised ChatDev success by 9.4 percentage points.",
      "supports": "That more than a third of observed multi-step agent failure sits at the joins rather than in any single agent's competence, and that standardising the message format does not fix it.",
      "doesNotSupport": "Anything about human-agent boundaries. Every handoff in the dataset is agent to agent and no humans appear anywhere. The authors caveat that the 210-trace distribution illustrates system-specific profiles rather than comparing performance across systems, so the percentages are not cross-benchmark comparable.",
      "terms": [
        "AI agents",
        "multi-agent systems",
        "handoffs",
        "coordination failure"
      ],
      "relatedPages": [
        "/research/how-do-humans-and-agents-divide-work"
      ]
    },
    {
      "id": "ntsb-asiana-2014",
      "citeAs": "https://thesuperskills.com/research/evidence#ntsb-asiana-2014",
      "section": "frontline",
      "authors": "National Transportation Safety Board",
      "year": 2014,
      "title": "Descent Below Visual Glidepath and Impact With Seawall, Asiana Airlines Flight 214, Boeing 777-200ER, HL7742, San Francisco, California, July 6, 2013",
      "publication": "NTSB/AAR-14/01, adopted 24 June 2014",
      "url": "https://www.ntsb.gov/investigations/AccidentReports/Reports/AAR1401.pdf",
      "grade": "statutory-investigation",
      "method": "Statutory accident investigation with access to flight recorders, crew interviews, manufacturer documentation and operator training records. N is one.",
      "finding": "The pilot flying selected a mode that caused the autoflight system to climb, then disconnected the autopilot and moved the thrust levers to idle, which put the autothrottle into HOLD, a mode in which it does not control airspeed. The board records that neither the pilot flying, the pilot monitoring, nor the observer noted the change in autothrottle mode to HOLD. Contributing factors in the probable cause include the complexities of the autothrottle and autopilot flight director systems being inadequately described in the manufacturer's documentation and the operator's training, which increased the likelihood of mode error. The board attributes insufficient airspeed monitoring in part to automation reliance, and notes the operator's automation policy emphasised full use of automation and did not encourage manual flight in line operations.",
      "supports": "That a correct automation state change can constitute a failed handoff. Nothing malfunctioned: authority moved from machine to crew and the crew were not told in a way that reached them.",
      "doesNotSupport": "Any frequency of mode confusion. One accident, three crew, one aircraft type. It cannot support a rate and is used here as a demonstration of a mechanism.",
      "terms": [
        "mode confusion",
        "handoffs",
        "automation reliance",
        "aviation"
      ],
      "relatedPages": [
        "/research/how-do-humans-and-agents-divide-work"
      ]
    },
    {
      "id": "joint-commission-sea58-2017",
      "citeAs": "https://thesuperskills.com/research/evidence#joint-commission-sea58-2017",
      "section": "frontline",
      "authors": "The Joint Commission",
      "year": 2017,
      "title": "Sentinel Event Alert 58: Inadequate hand-off communication",
      "publication": "The Joint Commission, Issue 58, 12 September 2017",
      "url": "https://www.jointcommission.org/en-us/knowledge-library/newsletters/sentinel-event-alert/issue-58",
      "grade": "compiled-review",
      "method": "Advisory bulletin aggregating third-party findings. No original data, no sample, no denominator.",
      "finding": "States that inadequate hand-off communication contributes to adverse events including wrong-site surgery, delay in treatment, falls and medication errors. Reports, from cited third parties, that communication failures were responsible at least in part for 30 per cent of US malpractice claims over five years, 1,744 deaths and 1.7 billion dollars in costs; that a typical teaching hospital may experience more than 4,000 hand-offs a day; and that 69 per cent of clinical learning environments had no standardised hand-off process against 20 per cent with some standardisation. Its only quantified outcome evidence is the I-PASS trial.",
      "supports": "That handoff has been named as a systemic failure point by a national accreditation body, and that the profession's own quantified evidence for it is thinner than the advisory framing implies.",
      "doesNotSupport": "The most quoted claim attached to it. The statement that 80 per cent of serious medical errors involve miscommunication during handoff does not appear anywhere in this alert; the full six pages were read on 4 September 2026. The nearest real figure is Starmer and colleagues citing a Joint Commission statistics page for two of every three sentinel events involving communication failures, which is a different denominator and a broader category. Every figure in the alert is footnoted to a document not read here.",
      "terms": [
        "handoffs",
        "patient safety",
        "citation provenance",
        "communication"
      ],
      "relatedPages": [
        "/research/how-do-humans-and-agents-divide-work"
      ]
    },
    {
      "id": "eu-ai-act-art-13",
      "citeAs": "https://thesuperskills.com/research/evidence#eu-ai-act-art-13",
      "section": "institutional",
      "authors": "European Parliament and Council of the European Union",
      "year": 2024,
      "title": "Regulation (EU) 2024/1689, Article 13: Transparency and provision of information to deployers",
      "publication": "Official Journal of the European Union, official version of 13 June 2024. Text read via the European Commission AI Act Service Desk",
      "url": "https://ai-act-service-desk.ec.europa.eu/en/ai-act/article-13",
      "grade": "institutional-modelling",
      "method": "Primary legal text.",
      "finding": "Requires high-risk systems to be designed so their operation is sufficiently transparent for deployers to interpret the output and use it appropriately, and to be accompanied by instructions for use stating the level of accuracy and its metrics, robustness and cybersecurity against which the system was tested and validated, plus known and foreseeable circumstances affecting that expected level, and where applicable information enabling deployers to interpret the output. Article 15(3) requires declared accuracy levels in those instructions.",
      "supports": "That European law requires documentary, aggregate, ex ante disclosure of accuracy and limitations to the deployer.",
      "doesNotSupport": "Any requirement to communicate uncertainty at the point of use. The word uncertainty appears in neither Article 13, Article 15 nor Article 50, and nothing requires a system to tell the person in front of it how confident it is in the specific output. EUR-Lex returned an empty document to every route tried on 4 September 2026, so the text was read at the Commission's own service desk; the Article 50 page there carries a notice that the provision has been amended by the Digital Omnibus and the displayed text not yet updated.",
      "terms": [
        "EU AI Act",
        "transparency",
        "uncertainty",
        "regulation"
      ],
      "relatedPages": [
        "/research/how-should-an-ai-agent-communicate-uncertainty"
      ]
    },
    {
      "id": "steyvers-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#steyvers-2025",
      "section": "collaboration",
      "authors": "Steyvers, M., Tejeda, H., Kumar, A., Belem, C., Karny, S., Hu, X., Mayer, L. W. and Smyth, P.",
      "year": 2025,
      "title": "What large language models know and what people think they know",
      "publication": "Nature Machine Intelligence, 7, 221-231",
      "url": "https://doi.org/10.1038/s42256-024-00976-7",
      "grade": "peer-reviewed",
      "method": "Two behavioural experiments, 301 participants recruited through Prolific, each assigned 40 questions from pools of 350 multiple-choice and 336 short-answer items. Model confidence from GPT-3.5, PaLM2 and GPT-4o compared against participant confidence after reading model explanations.",
      "finding": "Model confidence discriminates correct from incorrect answers at AUC 0.751 for GPT-3.5, 0.746 for PaLM2 and 0.781 for GPT-4o. Participants reading default explanations reached 0.589, 0.602 and 0.592, which the authors describe as only slightly better than random guessing. They name the shortfalls the calibration gap and the discrimination gap, and attribute human miscalibration primarily to overconfidence, people believing LLMs are more accurate than they are. Longer explanations significantly raised participant confidence without improving discrimination, mean participant AUC 0.54 for long explanations.",
      "supports": "That the reader is a worse judge of an answer's correctness than the model is, on the same items, and that adding explanation length makes readers surer rather than righter.",
      "doesNotSupport": "That closing the gap improves task accuracy. The modified-explanation result is a simulation via post-hoc filtering rather than a live deployment. Participants had no domain expertise and their own accuracy was 33 per cent against the model's 39. Statistics are Bayes factors only, with no p values or effect sizes, and the ECE values sit inside a figure rather than in the text.",
      "terms": [
        "calibration",
        "confidence",
        "over-reliance",
        "explanation"
      ],
      "relatedPages": [
        "/research/how-should-an-ai-agent-communicate-uncertainty",
        "/research/ai-and-human-disagreement"
      ]
    },
    {
      "id": "xiong-confidence-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#xiong-confidence-2024",
      "section": "collaboration",
      "authors": "Xiong, M., Hu, Z., Lu, X., Li, Y., Fu, J., He, J. and Hooi, B.",
      "year": 2024,
      "title": "Can LLMs Express Their Uncertainty? An Empirical Evaluation of Confidence Elicitation in LLMs",
      "publication": "ICLR 2024. arXiv:2306.13063",
      "url": "https://arxiv.org/abs/2306.13063",
      "grade": "peer-reviewed",
      "method": "Systematic evaluation of confidence elicitation across five models, Vicuna 13B, GPT-3, GPT-3.5-turbo, GPT-4 and LLaMA 2 70B, on eight datasets spanning arithmetic, commonsense, symbolic and professional reasoning. Metrics: expected calibration error, AUROC and AUPRC. No human subjects.",
      "finding": "Average expected calibration error for plain verbalised confidence is 0.520 for GPT-3, 0.461 for Vicuna, 0.436 for LLaMA 2, 0.377 for GPT-3.5 and 0.180 for GPT-4. GPT-4's average AUROC is 62.7 per cent against a 50 per cent chance baseline. Stated confidences cluster in the 80 to 100 per cent range in multiples of five, which the authors suggest means models may be imitating human expressions when verbalising confidence.",
      "supports": "That a number a model gives when asked how sure it is is close to unusable as a calibrated quantity, and that its shape suggests it is generated as plausible text rather than measured.",
      "doesNotSupport": "Anything about how users read these numbers, since no humans were involved. Mitigation strategies reduce ECE substantially, to 0.028 in the best case, while still failing to predict incorrect answers on knowledge-heavy tasks. Models are of the GPT-4 and LLaMA 2 generation.",
      "terms": [
        "calibration",
        "confidence",
        "hallucination",
        "AI agents"
      ],
      "relatedPages": [
        "/research/how-should-an-ai-agent-communicate-uncertainty"
      ]
    },
    {
      "id": "kim-uncertainty-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#kim-uncertainty-2024",
      "section": "collaboration",
      "authors": "Kim, S. S. Y., Liao, Q. V., Vorvoreanu, M., Ballard, S. and Wortman Vaughan, J.",
      "year": 2024,
      "title": "I'm Not Sure, But...: Examining the Impact of Large Language Models' Uncertainty Expression on User Reliance and Trust",
      "publication": "Proceedings of the 2024 ACM Conference on Fairness, Accountability, and Transparency (FAccT 24). Pre-registered at osf.io/mnrp9",
      "url": "https://doi.org/10.1145/3630106.3658941",
      "grade": "peer-reviewed",
      "method": "Pre-registered between-subjects experiment, four conditions, eight yes-or-no medical questions. 656 responses collected, 252 excluded on pre-registered criteria, final sample 404. Responses pre-generated and presented as a fictional system whose answers were correct on exactly half the questions. Conditions varied only the presence and perspective of uncertainty expression.",
      "finding": "Access to the system raised agreement to 80.9 per cent against 58.4 per cent without, and lowered accuracy to 63.9 per cent against 74.2 per cent without. First-person uncertainty expression significantly reduced agreement to 74.8 per cent and significantly raised accuracy to 72.8 per cent. Impersonal expression moved both in the same direction without reaching significance. Intention to use fell significantly under first-person hedging, 2.91 against 3.25 in control and 3.36 for the impersonal version. Hedging reduced accuracy slightly when the system was correct and raised it more when the system was wrong.",
      "supports": "That hedging changes reliance in the direction that helps, that the perspective of the hedge matters, and that the version which helps most is the version users least want to keep using.",
      "doesNotSupport": "That uncertainty expression should be mandated. The authors state directly that regulators should avoid blanket requirements until more research is done, and flag that their system had low accuracy and expressed uncertainty often in a poorly calibrated manner. It did not eliminate over-reliance: participants without AI access still performed best. Reported as model-estimated means with significance stars, no test statistics, exact p values or effect sizes, and 38.4 per cent of collected responses were excluded.",
      "terms": [
        "uncertainty",
        "over-reliance",
        "trust calibration",
        "interface design"
      ],
      "relatedPages": [
        "/research/how-should-an-ai-agent-communicate-uncertainty",
        "/research/ai-and-human-disagreement"
      ]
    },
    {
      "id": "zhou-unreliable-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#zhou-unreliable-2024",
      "section": "collaboration",
      "authors": "Zhou, K., Hwang, J. D., Ren, X. and Sap, M.",
      "year": 2024,
      "title": "Relying on the Unreliable: The Impact of Language Models' Reluctance to Express Uncertainty",
      "publication": "Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (ACL 2024), Long Papers, 3623-3643. Also arXiv:2401.06730",
      "url": "https://aclanthology.org/2024.acl-long.198/",
      "grade": "peer-reviewed",
      "method": "Model study: nine models prompted with 49 prompts over 284 MMLU questions, 125,244 queries. Human study: Prolific participants across a control setting and three interactive settings with calibrated, overconfident and underconfident systems, 25 recruited per setting.",
      "finding": "Only about 5 per cent of generated answers include any epistemic marker. Among confidently expressed responses the error rate averages 47 per cent, and only 53 per cent of generations expressing certainty are correct. In the human study, hedged answers were relied on around 10 per cent of the time and confident ones around 90, but plain unmarked statements were also relied on nearly 90 per cent of the time, which the authors read as users interpreting the absence of a marker as certainty. Exposure to an overconfident system left mental models uncorrected, with participants averaging 76 per cent during miscalibrated rounds and 86 afterwards, while an underconfident system produced 66 and then 98. Reward modelling scores plain statements at 4.03, expressions of certainty at 0.82 and expressions of doubt at minus 1.86.",
      "supports": "That silence about uncertainty is read as confidence rather than as neutrality, and that the confident register is a product of a preference model that penalises hedging more than it rewards assurance.",
      "doesNotSupport": "Any of it with statistical rigour on the human side. No total human N is stated in the text, and no p values, confidence intervals or effect sizes are reported for any human result. Twenty-five participants per setting. US-only, which the authors themselves call a narrow and US-centric view. Body text and Table 1 disagree slightly on the marker rate, 5 against 6 per cent. Peer-reviewed venue corrected to ACL 2024 on 4 September 2026 after an earlier draft of this entry read the arXiv preprint as unpublished.",
      "terms": [
        "uncertainty",
        "over-reliance",
        "RLHF",
        "confidence"
      ],
      "relatedPages": [
        "/research/how-should-an-ai-agent-communicate-uncertainty",
        "/research/ai-and-human-disagreement"
      ]
    },
    {
      "id": "budescu-2012",
      "citeAs": "https://thesuperskills.com/research/evidence#budescu-2012",
      "section": "judgement",
      "authors": "Budescu, D. V., Por, H.-H. and Broomell, S. B.",
      "year": 2012,
      "title": "Effective communication of uncertainty in the IPCC reports",
      "publication": "Climatic Change, 113, 181-200",
      "url": "https://doi.org/10.1007/s10584-011-0330-3",
      "grade": "peer-reviewed",
      "method": "Nationally representative US survey experiment via TESS and the Knowledge Networks panel, December 2009 to January 2010. 841 invited, 556 completed, 66 per cent. Three between-subjects conditions: control 193, translation table supplied 175, verbal terms with numerical ranges in the text 188. Eight sentences from IPCC reports, four probability terms tested.",
      "finding": "Mean estimates were 41 for very unlikely, 44 for unlikely, 54 for likely and 62 for very likely, against IPCC guidelines of below 10, below 33, above 66 and above 90. Consistency with the guidelines was 20.76 per cent in the control, 18.81 per cent when the translation table was supplied and 30.12 per cent when numerical ranges appeared alongside the words. 24 per cent of respondents gave no response consistent with the guidelines and only 6 per cent gave six or more. The authors describe the pattern as regressive, with the median respondent reading an intended 0.90 as about 0.65 to 0.75.",
      "supports": "That publishing a glossary does not fix a probability vocabulary, and that only putting the number in the sentence helps, and then only to under a third consistency.",
      "doesNotSupport": "That this settles the design. Only four of the seven IPCC terms were tested, no lower or upper bound data were collected, and the authors decline to read the result as a criticism of the IPCC, noting there is no optimal method. The widely cited 2009 Psychological Science paper by the same authors could not be opened during the 4 September 2026 build, so its figures appear nowhere on this estate.",
      "terms": [
        "uncertainty",
        "probability",
        "communication",
        "interpretation"
      ],
      "relatedPages": [
        "/research/how-should-an-ai-agent-communicate-uncertainty"
      ]
    },
    {
      "id": "budescu-2014",
      "citeAs": "https://thesuperskills.com/research/evidence#budescu-2014",
      "section": "judgement",
      "authors": "Budescu, D. V., Por, H.-H., Broomell, S. B. and Smithson, M.",
      "year": 2014,
      "title": "The interpretation of IPCC probabilistic statements around the world",
      "publication": "Nature Climate Change, 4(6), 508-512",
      "url": "https://doi.org/10.1038/nclimate2194",
      "grade": "peer-reviewed",
      "method": "Multi-national survey experiment, 25 samples across 24 countries and 17 languages, testing four target terms under a translation condition and a verbal-numerical condition.",
      "finding": "Laypeople interpret IPCC statements as conveying probabilities closer to 50 per cent than intended by the IPCC authors. Supplementing verbal terms with numerical ranges increases correspondence with the guidelines and improves differentiation between terms. The authors describe the qualitative patterns as remarkably stable across all samples and languages, and note that interpretations across languages become more similar under the numerical format.",
      "supports": "That the regressive reading of probability words is not an artefact of English or of American respondents. It is general.",
      "doesNotSupport": "Any magnitude quotable from this estate. The article is paywalled and only the abstract was read on 4 September 2026, so no participant count, per-country sample or consistency percentage is given anywhere here.",
      "terms": [
        "uncertainty",
        "probability",
        "communication",
        "cross-cultural"
      ],
      "relatedPages": [
        "/research/how-should-an-ai-agent-communicate-uncertainty"
      ]
    },
    {
      "id": "zhang-liao-2020",
      "citeAs": "https://thesuperskills.com/research/evidence#zhang-liao-2020",
      "section": "collaboration",
      "authors": "Zhang, Y., Liao, Q. V. and Bellamy, R. K. E.",
      "year": 2020,
      "title": "Effect of confidence and explanation on accuracy and trust calibration in AI-assisted decision making",
      "publication": "Proceedings of the 2020 Conference on Fairness, Accountability, and Transparency (FAT* 20)",
      "url": "https://doi.org/10.1145/3351095.3372852",
      "grade": "peer-reviewed",
      "method": "Two online experiments on an income-prediction task from US census data, 40 trials each. Experiment 1: 72 Mechanical Turk participants, nine per cell in a two by two by two design. Experiment 2: nine further participants, analysed against two Experiment 1 cells for a total of 27. Human unaided accuracy 65 per cent, model accuracy 75 per cent.",
      "finding": "Showing confidence scores significantly increased trust, F(1,64)=4.64, p=.035, and significantly improved trust calibration when model confidence was above 80 per cent, F(4,256)=15.8, p<.001. There was no significant difference in AI-assisted accuracy across the prediction and confidence conditions, a result the authors report as rejecting their own hypothesis. Local explanations produced no significant change against baseline on switching behaviour or accuracy, with a reverse trend on accuracy.",
      "supports": "That a confidence score moves how a person feels about a system more reliably than it moves what they catch, and that explanation alone did neither in this setting.",
      "doesNotSupport": "Much on its own. Nine participants per cell in Experiment 1 and 27 in Experiment 2, no effect sizes, confidence intervals or standard deviations reported, no exclusions or attention checks described, non-expert participants, and a contrived task carrying no responsibility. The authors also note the approach depends on the model's probabilities being well calibrated in the first place, which the calibration literature says they usually are not.",
      "terms": [
        "confidence",
        "explanation",
        "trust calibration",
        "over-reliance"
      ],
      "relatedPages": [
        "/research/how-should-an-ai-agent-communicate-uncertainty"
      ]
    },
    {
      "id": "orben-przybylski-2019",
      "citeAs": "https://thesuperskills.com/research/evidence#orben-przybylski-2019",
      "section": "learning",
      "authors": "Orben, A. and Przybylski, A. K.",
      "year": 2019,
      "title": "The association between adolescent well-being and digital technology use",
      "publication": "Nature Human Behaviour, 3(2), 173-182",
      "url": "https://doi.org/10.1038/s41562-018-0506-1",
      "grade": "peer-reviewed",
      "method": "Specification curve analysis across three nationally representative datasets: the US Youth Risk and Behaviour Survey 2007-2015 at 74,814 adolescents, Monitoring the Future 2008-2016 at 268,672, and the UK Millennium Cohort Study at 11,872, a total of 355,358. 372 justifiable specifications identified for YRBS, 40,966 for MTF and 603,979,752 for MCS, of which 20,004 were run.",
      "finding": "The association between digital technology use and adolescent wellbeing is negative but small, explaining at most 0.4 per cent of the variation in wellbeing, which the authors state is too small to warrant policy change. In YRBS, regularly eating potatoes was associated with wellbeing 0.9 times as negatively as technology use; in MCS, wearing glasses was 1.5 times as negatively. Bullying ran 4.3 times more negative and marijuana 2.7 times, both YRBS. Sleep and breakfast ranged from 1.7 to 44.2 times more positive across all three datasets.",
      "supports": "That an entire public debate was conducted on an effect too small to act on, and that analytic flexibility rather than data availability was doing the work in the studies that found more.",
      "doesNotSupport": "Causation in either direction. The authors state it is possible the associations they document, and those previously documented, are spurious, and that noisy self-report measurement could itself have diminished a real effect. Two figures often attached to this paper are not in it: 1.45 times for glasses, and 3,221,225,472 analyses, which comes from Odgers and Jensen describing it. Read in the Oxford ORA accepted manuscript, the published full text being paywalled.",
      "terms": [
        "screen time",
        "adolescents",
        "wellbeing",
        "measurement",
        "specification curve"
      ],
      "relatedPages": [
        "/research/is-screen-time-the-same-argument-as-ai-use"
      ]
    },
    {
      "id": "orben-przybylski-diaries-2019",
      "citeAs": "https://thesuperskills.com/research/evidence#orben-przybylski-diaries-2019",
      "section": "learning",
      "authors": "Orben, A. and Przybylski, A. K.",
      "year": 2019,
      "title": "Screens, Teens, and Psychological Well-Being: Evidence From Three Time-Use-Diary Studies",
      "publication": "Psychological Science, 30(5), 682-696",
      "url": "https://doi.org/10.1177/0956797619830329",
      "grade": "peer-reviewed",
      "method": "Exploratory and confirmatory specification analyses across three nationally representative datasets from Ireland, the United States and the United Kingdom, N = 17,247 after exclusions, using time-use diaries as well as retrospective self-report.",
      "finding": "Little evidence of substantial negative associations between digital screen engagement and adolescent wellbeing, whether measured across the day or before bedtime. Correlations between diary-recorded and retrospectively self-reported engagement were 0.18 in Ireland, 0.08 and 0.05 on US weekdays and weekend days, and 0.18 in the UK. Retrospective self-report consistently produced the most negative correlations. Extrapolating from median effects, an adolescent would need to report 63 hours 31 minutes more technology use a day to lower wellbeing by half a standard deviation, or 11 hours 14 minutes taking the maximum effect size in the specification set.",
      "supports": "That the two standard instruments for measuring screen exposure barely agree with each other, and that the more negative results come from the weaker instrument, which is what common method variance would predict.",
      "doesNotSupport": "That diaries are ground truth. They remain recall-based and the authors note brief or concurrent uses may not be recorded. Still cross-sectional, and it measures totals and timing rather than the kind of activity displaced.",
      "terms": [
        "screen time",
        "measurement",
        "self-report",
        "adolescents"
      ],
      "relatedPages": [
        "/research/is-screen-time-the-same-argument-as-ai-use"
      ]
    },
    {
      "id": "parry-2021",
      "citeAs": "https://thesuperskills.com/research/evidence#parry-2021",
      "section": "learning",
      "authors": "Parry, D. A., Davidson, B. I., Sewall, C. J. R., Fisher, J. T., Mieczkowski, H. and Quintana, D. S.",
      "year": 2021,
      "title": "A systematic review and meta-analysis of discrepancies between logged and self-reported digital media use",
      "publication": "Nature Human Behaviour, 5(11), 1535-1547",
      "url": "https://doi.org/10.1038/s41562-021-01117-5",
      "grade": "peer-reviewed",
      "method": "Systematic review and meta-analysis using robust variance estimation. 106 effect sizes included overall; the self-report against logged comparison draws on 66 effect sizes from 44 studies with a total sample of 52,007.",
      "finding": "The correlation between self-reported and logged digital media use is positive but medium, r = 0.38, 95 per cent CI 0.33 to 0.42. For problematic use it falls to r = 0.25 across 40 effect sizes from 19 studies. Over-reporting and under-reporting occur in similar proportions, and fewer than 10 per cent of self-reports fall within 5 per cent of the equivalent logged value. The authors conclude that self-report measures may not be a valid stand-in for more objective measures and ask for pause in drawing wide-reaching knowledge or policy conclusions from studies relying solely on them.",
      "supports": "That the independent variable in most of the screen-time literature is wrong by a measured amount, which is the strongest single reason not to carry that method into questions about AI use.",
      "doesNotSupport": "That self-report is useless, or that logs are ground truth: the authors note potential biases in log data too. They state it is an open question whether the discrepancy is random or systematic error. Nothing in it concerns AI. Read in the University of Bath accepted manuscript alongside the published abstract.",
      "terms": [
        "measurement",
        "self-report",
        "screen time",
        "validity"
      ],
      "relatedPages": [
        "/research/is-screen-time-the-same-argument-as-ai-use"
      ]
    },
    {
      "id": "mansfield-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#mansfield-2025",
      "section": "learning",
      "authors": "Mansfield, K. L., Ghai, S., Hakman, T., Ballou, N., Vuorre, M. and Przybylski, A. K.",
      "year": 2025,
      "title": "From social media to artificial intelligence: improving research on digital harms in youth",
      "publication": "The Lancet Child and Adolescent Health, 9(3), 194-204. Personal View, not primary research",
      "url": "https://doi.org/10.1016/S2352-4642(24)00332-8",
      "grade": "compiled-review",
      "method": "Personal View setting out the methodological failures of the social media harms literature and what should be done differently for AI. No new data.",
      "finding": "Self-reported screen time is problematic as a measure, being imprecise and prone to bias, and as a construct, being unidimensional, homogenous and of little validity, failing to distinguish social, educational, entertainment, work and informational uses. On AI, the authors write that using self-reported frequency or duration of adolescent AI use as the exposure measure of interest is perhaps even more concerning than counting total time spent on social media, and that only behavioural data on exposure to a range of AI applications would provide the detail needed. They record that health policy decisions have been implemented on inconsistent, non-causal or ungeneralisable evidence of online harms.",
      "supports": "That the researchers who built the screen-time evidence base have published the warning against transplanting its method to AI, in advance and in a clinical journal.",
      "doesNotSupport": "Anything empirical. It is a commentary with no new data and no head-to-head methodological comparison. Its prescription is finer-grained measurement of which AI in what context, which is adjacent to but not the same as asking which human practice was displaced. Publisher full texts returned empty bodies; read in the Oxford ORA accepted version.",
      "terms": [
        "screen time",
        "AI and children",
        "measurement",
        "research method"
      ],
      "relatedPages": [
        "/research/is-screen-time-the-same-argument-as-ai-use"
      ]
    },
    {
      "id": "odgers-jensen-2020",
      "citeAs": "https://thesuperskills.com/research/evidence#odgers-jensen-2020",
      "section": "learning",
      "authors": "Odgers, C. L. and Jensen, M. R.",
      "year": 2020,
      "title": "Annual Research Review: Adolescent mental health in the digital age: facts, fears, and future directions",
      "publication": "Journal of Child Psychology and Psychiatry, 61(3), 336-348",
      "url": "https://doi.org/10.1111/jcpp.13190",
      "grade": "compiled-review",
      "method": "Annual research review of the evidence on adolescent digital technology use and mental health, covering 29 studies in its main table.",
      "finding": "Most research to date has been correlational, focused on adults rather than adolescents, and has produced a mix of conflicting small positive, negative and null associations. The most recent and rigorous large-scale preregistered studies report small associations that offer no way of distinguishing cause from effect and are unlikely to be of clinical or practical significance, explaining less than 0.5 per cent of the variance. Of the 29 studies reviewed, only two included objective or informant-rated measures of screen use, and the correlation between objectively measured and retrospectively reported screen time is estimated at about 0.20.",
      "supports": "That the field's own review verdict is that the evidence does not support causal claims or even strong consistent correlational patterns.",
      "doesNotSupport": "Anything new. It is a review, not primary data, it does not test displacement of any specific activity, and it does not address AI. Read in the NIH author manuscript.",
      "terms": [
        "screen time",
        "adolescents",
        "mental health",
        "evidence quality"
      ],
      "relatedPages": [
        "/research/is-screen-time-the-same-argument-as-ai-use"
      ]
    },
    {
      "id": "przybylski-weinstein-2017",
      "citeAs": "https://thesuperskills.com/research/evidence#przybylski-weinstein-2017",
      "section": "learning",
      "authors": "Przybylski, A. K. and Weinstein, N.",
      "year": 2017,
      "title": "A Large-Scale Test of the Goldilocks Hypothesis: Quantifying the Relations Between Digital-Screen Use and the Mental Well-Being of Adolescents",
      "publication": "Psychological Science, 28(2), 204-215",
      "url": "https://doi.org/10.1177/0956797616678438",
      "grade": "peer-reviewed",
      "method": "Preregistered analysis of a representative sample of English 15-year-olds. Sampling frame 298,080; 120,115 provided usable data, 100,850 on paper and 19,265 online.",
      "finding": "Links between digital screen time and mental wellbeing are described by quadratic rather than linear functions, with inflection points at 1 hour 40 minutes for weekday video-game play and 1 hour 57 minutes for weekday smartphone use, rising to 3 hours 41 minutes and 4 hours 17 minutes for other measures and to between 3 hours 35 minutes and 4 hours 50 minutes at weekends. Average Cohen's d for engagement beyond the inflection points was minus 0.18, accounting for 1 per cent or less of variance, against d = 0.54 for regularly eating breakfast and 0.58 for regular sleep. The authors state that moderate use is not intrinsically harmful and may be advantageous.",
      "supports": "That dose-response in this literature is not linear, and that the negative effects of heavy use are less than a third the size of the positive associations with sleep and breakfast.",
      "doesNotSupport": "Causation, and not displacement. The paper names the displacement hypothesis as the field's dominant assumption and calls for future work systematically analysing what is being displaced or amplified, which the field did not then do. Cross-sectional, self-reported exposure, English 15-year-olds only.",
      "terms": [
        "screen time",
        "adolescents",
        "displacement",
        "dose-response"
      ],
      "relatedPages": [
        "/research/is-screen-time-the-same-argument-as-ai-use"
      ]
    },
    {
      "id": "uk-cmo-screentime-2019",
      "citeAs": "https://thesuperskills.com/research/evidence#uk-cmo-screentime-2019",
      "section": "institutional",
      "authors": "Davies, S. C., Atherton, F., Calderwood, C. and McBride, M.",
      "year": 2019,
      "title": "United Kingdom Chief Medical Officers' commentary on screen-based activities and children and young people's mental health and psychosocial wellbeing",
      "publication": "Department of Health and Social Care, Office of the Chief Medical Officer, 7 February 2019",
      "url": "https://assets.publishing.service.gov.uk/government/uploads/system/uploads/attachment_data/file/777026/UK_CMO_commentary_on_screentime_and_social_media_map_of_reviews.pdf",
      "grade": "compiled-review",
      "method": "Official commentary by the four UK Chief Medical Officers on a commissioned systematic map of reviews. No new data.",
      "finding": "States that scientific research is currently insufficiently conclusive to support UK CMO evidence-based guidelines on optimal amounts of screen use or online activities, and that the research does not present evidence of a causal relationship between screen-based activities and mental health problems, noting the possibility that young people who already have mental health problems spend more time on social media. It recommends a precautionary approach anyway, separates screen time from internet content and from persuasive design as three distinct issues, and advises families on the basis that screen time can displace health-promoting activities.",
      "supports": "That the UK's most senior clinical advisers concluded in 2019 that the dose measure could not support guidance, and named displacement rather than duration as the mechanism worth managing.",
      "doesNotSupport": "That screen use is harmless. The CMOs explicitly state that the absence of evident causal effect does not mean there is no effect. It is a commentary on a map of reviews rather than primary research, and it predates the AI question entirely.",
      "terms": [
        "screen time",
        "public health",
        "displacement",
        "policy"
      ],
      "relatedPages": [
        "/research/is-screen-time-the-same-argument-as-ai-use"
      ]
    },
    {
      "id": "patel-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#patel-2024",
      "section": "frontline",
      "authors": "Patel, V. R., Liu, M., Worsham, C. M. and Jena, A. B.",
      "year": 2024,
      "title": "Alzheimer's disease mortality among taxi and ambulance drivers: population based cross sectional study",
      "publication": "The BMJ, 387, Christmas issue, 17 December 2024. DOI 10.1136/bmj-2024-082194. PROVENANCE: bmj.com could not be reached from either fetcher on 5 September 2026, so the figures below are taken from BMJ Group's own press release for the paper and the Science Media Centre briefing, both read at source that day, rather than from the full text. Re-read and confirm at bmj.com when access allows.",
      "url": "https://bmjgroup.com/alzheimers-disease-deaths-lowest-among-taxi-and-ambulance-drivers/",
      "grade": "peer-reviewed",
      "method": "Population-based cross-sectional study of US death certificates from the National Vital Statistics System, 1 January 2020 to 31 December 2022, covering 443 occupations and nearly 9 million deaths with occupational information. Usual occupation is the one in which the decedent spent most of their working life. Adjusted for age at death and sociodemographic factors. Published in the BMJ's Christmas issue, which is peer reviewed and deliberately light in subject matter.",
      "finding": "3.9 per cent of deaths (348,328) had Alzheimer's disease listed as a cause. Among 16,658 taxi drivers, 171 did (1.03 per cent); among 1,348 ambulance drivers, 10 did (0.74 per cent). After adjustment, taxi and ambulance drivers had the lowest proportion of any occupation examined (1.03 and 0.91 per cent) against 1.69 per cent for the general population. The pattern did not appear among bus drivers or pilots, who follow predetermined routes, nor for other dementias.",
      "supports": "That an occupational association exists at national scale, and that it is specific to route-generating rather than route-following driving. The authors' own summary is the right one: 'We view these findings not as conclusive, but as hypothesis generating.'",
      "doesNotSupport": "That navigation protects anyone. Three limitations do most of the damage, two of them raised by Tara Spires-Jones for the Science Media Centre. The drivers died at around 64 to 67 against 74 for other occupations, and Alzheimer's onset is typically after 65, so some may not have lived long enough to develop it. Women were 10 to 22 per cent of the drivers against 48 per cent elsewhere, on a disease women are more likely to develop. And no brain imaging was involved, so the hippocampal mechanism is hypothesis rather than measurement. Robert Howard, reviewing it alongside her, judged it premature to suggest drivers turn off their satnavs to prevent dementia.",
      "terms": [
        "spatial memory",
        "The Knowledge",
        "cognitive reserve",
        "deskilling"
      ],
      "relatedPages": [
        "/research/does-gps-damage-your-brain"
      ]
    },
    {
      "id": "massenkoff-mccrory-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#massenkoff-mccrory-2026",
      "section": "work",
      "authors": "Massenkoff, M. and McCrory, P.",
      "year": 2026,
      "title": "Labor market impacts of AI: A new measure and early evidence",
      "publication": "Anthropic, 5 March 2026. Corrected 8 March 2026. Read at source 5 September 2026",
      "url": "https://www.anthropic.com/research/labor-market-impacts",
      "grade": "vendor-research",
      "method": "A new occupational exposure measure, observed exposure, built from three inputs: O*NET task lists for roughly 800 US occupations, Eloundou and colleagues' theoretical LLM task-exposure scores, and Anthropic's own Claude usage data from the Anthropic Economic Index, weighting automated over augmentative use. Employment outcomes come from the Current Population Survey, in a difference-in-differences comparison of the top exposure quartile against the 30 per cent of workers with zero measured exposure.",
      "finding": "No systematic increase in unemployment for highly exposed workers since late 2022; the pooled difference-in-differences estimate is small and indistinguishable from zero. On hiring, the monthly job-finding rate for 22 to 25 year olds entering the most exposed occupations fell by about 14 per cent in the post-ChatGPT period against 2022. The stable comparator runs at about 2 per cent per month in less exposed occupations, and entry into the most exposed jobs falls by roughly half a percentage point. The authors' own qualification, in their words, is that this is 'just barely statistically significant'. No such decrease appears for workers over 25.",
      "supports": "That a slowdown in youth hiring into exposed occupations shows up in a second dataset and a second method, alongside Brynjolfsson and colleagues on ADP payroll data. Also that the harder claim, a rise in unemployment among exposed workers, does not appear: the authors estimate they could detect a differential increase of about one percentage point and see nothing.",
      "doesNotSupport": "The 14 per cent is routinely restated as a fall against low-exposure peers. It is not. The paper's own words are 'compared to that in 2022 in the exposed occupations', so it is a change over time within one group, which is a weaker claim than the comparison usually reported. Nor does it establish cause: the authors list three benign readings, that unhired young workers may be staying in existing jobs, taking different ones, or returning to study, and note that survey-measured job transitions are prone to mismeasurement. TWO FURTHER CAUTIONS. The treatment variable is built from Anthropic's own product telemetry and cannot be reconstructed from outside the company. And on 8 March 2026 Anthropic corrected Figure 7, the job-finding figure this number is taken from, which had reversed the labels between the top-quartile and zero-exposure groups. Anyone citing a version of this chart captured between 5 and 8 March 2026 is citing the reversed one.",
      "terms": [
        "early careers",
        "jobs",
        "exposure",
        "hiring"
      ],
      "relatedPages": [
        "/research/will-ai-replace-entry-level-jobs",
        "/research/should-juniors-use-ai",
        "/research/the-most-quoted-ai-statistics-checked",
        "/research/what-is-the-ai-employment-gap"
      ]
    },
    {
      "id": "lo-2023",
      "citeAs": "https://thesuperskills.com/research/evidence#lo-2023",
      "section": "learning",
      "authors": "Lo, L. S.",
      "year": 2023,
      "title": "The CLEAR path: A framework for enhancing information literacy through prompt engineering",
      "publication": "The Journal of Academic Librarianship, 49(4), 102720",
      "url": "https://digitalrepository.unm.edu/ulls_fsp/211/",
      "grade": "argued-perspective",
      "method": "A proposed framework, not an experiment. Sets out five principles for writing prompts, Concise, Logical, Explicit, Adaptive and Reflective, for use in information literacy teaching.",
      "finding": "Offers the most widely cited peer-reviewed prompt framework in education. Every element concerns the quality of the instruction given to the model: brevity, order, specificity, iteration and review of the output.",
      "supports": "That the published prompt frameworks optimise the output. Useful here as the comparison case: CLEAR, CO-STAR and Google's TCREI all ask what the model should be told, and none asks what the person should keep doing themselves.",
      "doesNotSupport": "That following CLEAR improves learning, or output quality. It is a framework proposal in a library science journal, with no trial behind it, and its author does not claim one.",
      "terms": [
        "prompting",
        "information literacy",
        "frameworks"
      ],
      "relatedPages": [
        "/research/goal-context-friction-standard"
      ]
    },
    {
      "id": "wineburg-mcgrew-2019",
      "citeAs": "https://thesuperskills.com/research/evidence#wineburg-mcgrew-2019",
      "section": "learning",
      "authors": "Wineburg, S. and McGrew, S.",
      "year": 2019,
      "title": "Lateral Reading and the Nature of Expertise: Reading Less and Learning More When Evaluating Digital Information",
      "publication": "Teachers College Record, 121(11), 1-40",
      "url": "https://journals.sagepub.com/doi/10.1177/016146811912101102",
      "grade": "peer-reviewed",
      "method": "Think-aloud study of 45 experienced internet users evaluating unfamiliar websites: 10 PhD historians, 10 professional fact-checkers and 25 Stanford undergraduates.",
      "finding": "Historians and undergraduates read vertically, staying on the page and judging it by its own appearance. Fact-checkers left almost immediately, opened new tabs and read laterally, judging the source by what the rest of the web said about it. The fact-checkers reached sounder conclusions in less time.",
      "supports": "That evaluating a source is a behaviour rather than a checklist, that the behaviour is learnable, and that domain expertise does not confer it: PhD historians performed like undergraduates.",
      "doesNotSupport": "Anything about AI output specifically. It predates generative models, and a model's answer has no source page to leave. The transferable part is the move away from judging a text by its own surface.",
      "terms": [
        "verification",
        "sources",
        "critical thinking"
      ],
      "relatedPages": [
        "/research/the-source-rule"
      ]
    },
    {
      "id": "caulfield-2019",
      "citeAs": "https://thesuperskills.com/research/evidence#caulfield-2019",
      "section": "learning",
      "authors": "Caulfield, M.",
      "year": 2019,
      "title": "SIFT (The Four Moves)",
      "publication": "Hapgood, 19 June 2019. An earlier version, Four Moves and a Habit, appeared in Web Literacy for Student Fact Checkers (2017)",
      "url": "https://hapgood.us/2019/06/19/sift-the-four-moves/",
      "grade": "argued-perspective",
      "method": "A practitioner method, not a study. Four moves: Stop, Investigate the source, Find better coverage, Trace claims to the original context.",
      "finding": "Turns Wineburg and McGrew's lateral reading result into four actions a person can perform in under a minute, and is now taught in university libraries worldwide.",
      "supports": "That the fact-checker behaviour can be reduced to a teachable sequence. It is the most widely adopted of the source-evaluation frameworks and the closest external comparison to any source rule offered here.",
      "doesNotSupport": "Its own effectiveness. Caulfield offers it as a practical method and does not present a trial of it, and it inherits its evidence from the lateral reading literature rather than generating any.",
      "terms": [
        "verification",
        "sources",
        "misinformation"
      ],
      "relatedPages": [
        "/research/the-source-rule"
      ]
    },
    {
      "id": "hamilton-2016",
      "citeAs": "https://thesuperskills.com/research/evidence#hamilton-2016",
      "section": "learning",
      "authors": "Hamilton, E. R., Rosenberg, J. M. and Akcaoglu, M.",
      "year": 2016,
      "title": "The Substitution Augmentation Modification Redefinition (SAMR) Model: A Critical Review and Suggestions for Its Use",
      "publication": "TechTrends, 60(5), 433-441",
      "url": "https://eric.ed.gov/?id=EJ1110736",
      "grade": "peer-reviewed",
      "method": "Critical review of the SAMR model against the educational technology literature.",
      "finding": "Names three problems: the absence of context, a rigid hierarchical structure that implies higher is better, and an emphasis on product over process. The authors also record that SAMR is largely absent from the peer-reviewed literature despite heavy practitioner use, and that its theoretical and foundational evidence is thin.",
      "supports": "That a widely adopted framework can spread through practice with almost no evidence behind it, and that popularity is not a proxy for validity. Held here as the standard against which any framework on this estate, including this estate's own, should be judged.",
      "doesNotSupport": "That SAMR is useless in practice. The authors offer suggestions for its use rather than a case for abandoning it, and their objection is to how it is applied more than to what it contains.",
      "terms": [
        "frameworks",
        "education technology",
        "evidence"
      ],
      "relatedPages": [
        "/research/the-four-levels-of-ai-use"
      ]
    },
    {
      "id": "anderson-krathwohl-2001",
      "citeAs": "https://thesuperskills.com/research/evidence#anderson-krathwohl-2001",
      "section": "learning",
      "authors": "Anderson, L. W. and Krathwohl, D. R. (eds.)",
      "year": 2001,
      "title": "A Taxonomy for Learning, Teaching, and Assessing: A Revision of Bloom's Taxonomy of Educational Objectives",
      "publication": "Longman, New York",
      "url": "https://eric.ed.gov/?id=ED480802",
      "grade": "compiled-review",
      "method": "A revision of Bloom's 1956 taxonomy by a group of cognitive psychologists and curriculum specialists, restating the categories as verbs and adding a second knowledge dimension.",
      "finding": "Six cognitive process categories in ascending order: remember, understand, apply, analyse, evaluate, create.",
      "supports": "That there is a long-established and widely taught vocabulary for ordering cognitive demand, which any newer ladder of AI use is implicitly competing with.",
      "doesNotSupport": "That the order is strictly hierarchical in practice, or that it transfers to human-AI interaction. It describes what a learner is asked to do, not what a tool is asked to do, and the two come apart as soon as a model can perform the higher categories on request.",
      "terms": [
        "learning",
        "taxonomy",
        "frameworks"
      ],
      "relatedPages": [
        "/research/the-four-levels-of-ai-use"
      ]
    },
    {
      "id": "charlotin-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#charlotin-2026",
      "section": "institutional",
      "authors": "Charlotin, D.",
      "year": 2026,
      "title": "AI Hallucination Cases Database",
      "publication": "damiencharlotin.com/hallucinations. Updated daily; last updated 5 September 2026. Read at source 6 September 2026",
      "url": "https://www.damiencharlotin.com/hallucinations/",
      "grade": "compiled-review",
      "method": "A continuously updated register of legal DECISIONS in which a court found that generative AI had produced hallucinated content in material put before it. Compiled from published judgments worldwide since Q2 2023, with an automated reference checker used to find further examples. Not peer reviewed and explicitly a work in progress.",
      "finding": "2,022 decisions as at 6 September 2026. By jurisdiction: USA 1,379, Canada 217, Australia 110, UK 69, Israel 57, Brazil 41, India and Italy 15 each, France 13, Germany 11, and roughly thirty further countries. By party, and this is the figure most reporting misses: PRO SE LITIGANTS 1,163 against lawyers 805, with judges themselves recorded 31 times, experts 15, prosecutors 5 and paralegals 2. By nature: fabricated material 1,677, misrepresented 845, false quotes 547.",
      "supports": "That fabricated citation is a documented, counted, cross-jurisdictional phenomenon in the one setting where checking sources is a professional duty, and that it is not confined to lawyers: more than half the recorded parties were representing themselves. Useful as a floor and as a demonstration that the failure survives professional incentives to avoid it.",
      "doesNotSupport": "The scale of the problem, and the database says so itself: it tracks decisions where a court ruled on the matter, and 'does not track the (necessarily wider) universe of all fake citations or use of AI in court filings'. Undetected instances are absent by construction. Growth in the count also mixes real growth with better detection and better compilation. And because it updates daily, ANY figure taken from it must carry the date it was read.",
      "terms": [
        "hallucination",
        "verification",
        "sources",
        "legal"
      ],
      "relatedPages": [
        "/research/how-often-does-ai-invent-a-source",
        "/research/the-source-rule"
      ]
    },
    {
      "id": "mani-2013",
      "citeAs": "https://thesuperskills.com/research/evidence#mani-2013",
      "section": "judgement",
      "authors": "Mani, A., Mullainathan, S., Shafir, E. and Zhao, J.",
      "year": 2013,
      "title": "Poverty Impedes Cognitive Function",
      "publication": "Science, 341(6149), 976-980",
      "url": "https://www.science.org/doi/10.1126/science.1238041",
      "grade": "peer-reviewed",
      "method": "Two studies. A laboratory experiment with shoppers in a New Jersey shopping centre, in which thoughts about finances were experimentally induced before cognitive tasks. And a field study of sugarcane farmers in Tamil Nadu tested before harvest, when poor, and after harvest, when comparatively rich.",
      "finding": "Inducing financial concern reduced cognitive performance among poorer participants and not among better-off ones. The same farmers performed worse on cognitive tasks before harvest than after, with the authors putting the shortfall in a range comparable to a night without sleep.",
      "supports": "That scarcity itself consumes cognitive capacity, independently of who is experiencing it, and that the effect appears within the same individuals as their circumstances change. It is the origin of the bandwidth argument now applied to attention, time and other scarce resources.",
      "doesNotSupport": "Anything about AI, about rate limits or about any scarcity other than money and, by extension, time. A published Comment in Science (2013, doi 10.1126/science.1246680) disputes aspects of the analysis, and anyone citing this should say so. Applying it to a message quota is an analogy and should be labelled as one.",
      "terms": [
        "scarcity",
        "cognitive load",
        "judgement"
      ],
      "relatedPages": [
        "/research/running-out-of-messages-makes-you-worse"
      ]
    },
    {
      "id": "hoc-fees-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#hoc-fees-2026",
      "section": "institutional",
      "authors": "House of Commons Library",
      "year": 2026,
      "title": "Tuition fees in England: History, debates, and international comparisons",
      "publication": "Research Briefing CBP-10155. Fee cap set by The Higher Education (Fee Limits and Student Support) regulations for the 2026/27 academic year",
      "url": "https://commonslibrary.parliament.uk/research-briefings/cbp-10155/",
      "grade": "compiled-review",
      "method": "Parliamentary research briefing compiling the statutory fee limits and their history.",
      "finding": "The maximum tuition fee for a standard full-time undergraduate course in England is 9,790 pounds for 2026/27, a rise of 2.71 per cent from 9,535 pounds in 2025/26, uprated on the Office for Budget Responsibility's RPIX forecast published in November 2025.",
      "supports": "The per-year price of an English undergraduate degree, which is what makes a per-module cost calculable and turns an abstract argument about learning into a number a student can hold.",
      "doesNotSupport": "The cost of any individual module, which depends on how many a course runs in a year, nor anything about Scotland, Wales or Northern Ireland, which set fees separately. Nor does it include maintenance, rent or loan interest.",
      "terms": [
        "higher education",
        "tuition",
        "cost"
      ],
      "relatedPages": [
        "/research/what-happens-if-you-get-caught-using-ai"
      ]
    },
    {
      "id": "eu-ai-act-application-dates",
      "citeAs": "https://thesuperskills.com/research/evidence#eu-ai-act-application-dates",
      "section": "institutional",
      "authors": "European Commission, AI Act Service Desk",
      "year": 2026,
      "title": "Article 113: Entry into force and application, with the Digital Omnibus on AI amendments",
      "publication": "Regulation (EU) 2024/1689, official version of 13 June 2024, as amended by the Digital Omnibus on AI. Article 113 and the Service Desk Digital Omnibus FAQ both read at source 6 September 2026",
      "url": "https://ai-act-service-desk.ec.europa.eu/en/ai-act/article-113",
      "grade": "compiled-review",
      "method": "The enacted text of Article 113 on the Commission's own AI Act Explorer, read alongside the Commission's Digital Omnibus FAQ on the same site.",
      "finding": "As enacted, the Regulation applies from 2 August 2026, with Chapters I and II from 2 February 2025, Chapter III Section 4 and Chapters V, VII and XII from 2 August 2025, and Article 6(1) with its corresponding obligations from 2 August 2027. The Digital Omnibus on AI amends this, and the mechanism is NOT a fixed new date: the Commission ties the high-risk rules to the availability of standards and other support tools, so they begin once the Commission confirms those are sufficiently available, after a transition period. The flexibility carries a backstop. High-risk rules in Annex III areas such as employment and law enforcement apply AT MOST 16 months later than originally envisaged, and high-risk AI embedded in Annex I products such as medical devices at most 12 months later. Those backstops are widely reported as 2 December 2027 and 2 August 2028.",
      "supports": "That the high-risk obligations, including the Article 14 human oversight duties, have a LATEST date rather than a start date, and could begin sooner. Also that two Commission pages disagreed on 6 September 2026: the Article 113 page displays the unamended text and carries a disclaimer saying it has not been updated for the Omnibus, while the FAQ describing the Omnibus is still written in the language of a proposal.",
      "doesNotSupport": "The date on which any obligation will actually bite, which depends on a Commission confirmation about standards that had not been made when this was read. ANY PAGE SAYING 'December 2027 at the earliest' HAS IT BACKWARDS: December 2027 is the longstop for Annex III, not the opening. Nor does this establish the Omnibus's own entry into force from a primary source; that date is reported by secondary legal commentary as 27 July 2026 and is not confirmed here.",
      "terms": [
        "EU AI Act",
        "human oversight",
        "regulation",
        "high-risk"
      ],
      "relatedPages": [
        "/research/who-can-override-an-ai-system",
        "/research/what-board-oversight-of-ai-looks-like",
        "/research/how-should-ai-decision-rights-be-allocated",
        "/research/which-decisions-should-become-slower-because-of-ai"
      ]
    },
    {
      "id": "ohno-1988",
      "citeAs": "https://thesuperskills.com/research/evidence#ohno-1988",
      "section": "capability",
      "authors": "Ohno, T.",
      "year": 1988,
      "title": "Toyota Production System: Beyond Large-Scale Production",
      "publication": "Productivity Press. First English printing 1988; the Japanese original, Toyota seisan hoshiki, is 1978. Publisher record and DOI confirmed at Taylor and Francis 10 September 2026, where the current reissue is dated 2019",
      "url": "https://www.taylorfrancis.com/books/mono/10.4324/9780429273018/toyota-production-system-taiichi-ohno",
      "grade": "practitioner-method",
      "method": "A plant director's account of a production system he built, setting out the five-times-why procedure with a worked example of a machine that stopped. No experiment, no sample, no control.",
      "finding": "Ohno sets out asking why five times in succession as the standard procedure for reaching the cause of a fault rather than its symptom, and credits the underlying approach to Sakichi Toyoda. The procedure is a discipline for not stopping at the first answer.",
      "supports": "That the five-times-why has a dated, named industrial origin and belongs to Toyota rather than to whoever most recently put it on a slide.",
      "doesNotSupport": "That the procedure produces better causes than any other method, that five is the right number, or anything at all about curiosity as a psychological disposition. It is a shop-floor convention with sixty years of adoption and no controlled test behind it.",
      "terms": [
        "curiosity",
        "root cause",
        "questioning"
      ],
      "relatedPages": [
        "/research/superskill-curiosity"
      ]
    },
    {
      "id": "eberle-1971",
      "citeAs": "https://thesuperskills.com/research/evidence#eberle-1971",
      "section": "capability",
      "authors": "Eberle, R. F.",
      "year": 1971,
      "title": "Scamper: Games for Imagination Development",
      "publication": "D.O.K. Publishers. The 1971 first edition could not be opened: the Internet Archive scan of this title is the 1996 Prufrock Press edition, and the resolvable object below is the 2023 Routledge combined edition, DOI 10.4324/9781003423560. Both checked at source 10 September 2026",
      "url": "https://www.taylorfrancis.com/books/mono/10.4324/9781003423560/scamper-bob-eberle",
      "grade": "practitioner-method",
      "method": "A classroom activity book. Eberle assembles a seven-part mnemonic, Substitute, Combine, Adapt, Modify or Magnify, Put to other uses, Eliminate, Rearrange or Reverse, from the idea-spurring checklist in Alex Osborn's Applied Imagination of 1953. No study design.",
      "finding": "SCAMPER is a rewriting of an older checklist into a mnemonic for children, and its content belongs to Osborn before Eberle.",
      "supports": "That SCAMPER is an established teaching device with a traceable lineage, and that neither the letters nor the method are anybody's recent invention.",
      "doesNotSupport": "That running the seven prompts makes a person more creative or more curious. The 1971 edition was not read for this entry, and the date is taken from the bibliographic record rather than from the book.",
      "terms": [
        "curiosity",
        "creativity",
        "questioning"
      ],
      "relatedPages": [
        "/research/superskill-curiosity"
      ]
    }
  ]
}