{
  "name": "The SuperSkills evidence base",
  "description": "Graded evidence on what increasingly capable AI does to human capability. Each entry states method, finding, what it supports and what it does not.",
  "url": "https://thesuperskills.com/research/evidence",
  "compiledBy": "Rahim Hirji, The SuperSkills Intelligence Company",
  "licence": "CC BY 4.0",
  "lastReviewed": "28 August 2026",
  "count": 161,
  "grades": {
    "peer-reviewed": "Published in a peer-reviewed journal or archival conference.",
    "working-paper": "Circulated for comment; not yet peer-reviewed.",
    "institutional-survey": "Survey by an organisation, usually self-selected respondents, not peer-reviewed.",
    "institutional-modelling": "Projection or secondary analysis by an organisation, assumption-driven.",
    "compiled-review": "Aggregation of third-party data rather than original research."
  },
  "entries": [
    {
      "id": "risko-gilbert-2016",
      "citeAs": "https://thesuperskills.com/research/evidence#risko-gilbert-2016",
      "section": "judgement",
      "authors": "Risko, E. F. and Gilbert, S. J.",
      "year": 2016,
      "title": "Cognitive Offloading",
      "publication": "Trends in Cognitive Sciences, 20(9)",
      "url": "https://www.cell.com/trends/cognitive-sciences/abstract/S1364-6613(16)30098-5",
      "grade": "peer-reviewed",
      "method": "Review of the experimental literature on offloading.",
      "finding": "Defines cognitive offloading as using physical action or an external tool to reduce the mental demand of a task, and shows people offload not only when a task is hard but when they judge it to be hard.",
      "supports": "That the decision to offload is metacognitive, and frequently mistaken.",
      "doesNotSupport": "Nothing about generative AI specifically; it predates it.",
      "terms": [
        "cognitive offloading"
      ],
      "relatedPages": [
        "/research/what-is-cognitive-offloading",
        "/research/ai-and-human-judgement"
      ]
    },
    {
      "id": "sparrow-2011",
      "citeAs": "https://thesuperskills.com/research/evidence#sparrow-2011",
      "section": "judgement",
      "authors": "Sparrow, B., Liu, J. and Wegner, D. M.",
      "year": 2011,
      "title": "Google Effects on Memory: Cognitive Consequences of Having Information at Our Fingertips",
      "publication": "Science, 333(6043)",
      "url": "https://www.science.org/doi/10.1126/science.1207745",
      "grade": "peer-reviewed",
      "method": "Four laboratory experiments.",
      "finding": "When people expect information to remain available, they remember where to find it rather than the thing itself.",
      "supports": "That expected availability changes what gets encoded.",
      "doesNotSupport": "That total memory capability declines, or that the trade is net negative.",
      "terms": [
        "the Google effect",
        "cognitive offloading"
      ],
      "relatedPages": [
        "/research/what-is-cognitive-offloading"
      ]
    },
    {
      "id": "dahmani-bohbot-2020",
      "citeAs": "https://thesuperskills.com/research/evidence#dahmani-bohbot-2020",
      "section": "judgement",
      "authors": "Dahmani, L. and Bohbot, V. D.",
      "year": 2020,
      "title": "Habitual use of GPS negatively impacts spatial memory during self-guided navigation",
      "publication": "Scientific Reports, 10, 6310",
      "url": "https://www.nature.com/articles/s41598-020-62877-0",
      "grade": "peer-reviewed",
      "method": "Cross-sectional plus a three-year longitudinal follow-up.",
      "finding": "Habitual satnav users had worse spatial memory when navigating unaided, and heavier use over the following three years was associated with steeper decline.",
      "supports": "That a reliably performed external function is associated with weakening of the human equivalent, over years.",
      "doesNotSupport": "That the same holds for reasoning. Spatial memory is not judgement, and the design is correlational.",
      "terms": [
        "cognitive offloading",
        "deskilling"
      ],
      "relatedPages": [
        "/research/what-is-cognitive-offloading",
        "/research/using-ai-without-dependency"
      ]
    },
    {
      "id": "lee-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#lee-2025",
      "section": "judgement",
      "authors": "Lee, H.-P. et al.",
      "year": 2025,
      "title": "The Impact of Generative AI on Critical Thinking: Self-Reported Reductions in Cognitive Effort and Confidence Effects from a Survey of Knowledge Workers",
      "publication": "Microsoft Research and Carnegie Mellon, CHI 2025",
      "url": "https://www.microsoft.com/en-us/research/publication/the-impact-of-generative-ai-on-critical-thinking-self-reported-reductions-in-cognitive-effort-and-confidence-effects-from-a-survey-of-knowledge-workers/",
      "grade": "peer-reviewed",
      "method": "Survey of 319 knowledge workers about 936 real uses of AI at work.",
      "finding": "Higher confidence in the tool was associated with less critical thinking, and the thinking that remains shifts from producing to verifying, from solving to integrating.",
      "supports": "That the character of professional thinking changes with AI use, by self-report.",
      "doesNotSupport": "Causation. People who think differently may use AI differently, and a survey cannot separate the two.",
      "terms": [
        "critical thinking",
        "verification",
        "cognitive offloading"
      ],
      "relatedPages": [
        "/research/ai-and-critical-thinking",
        "/research/ai-and-human-judgement"
      ]
    },
    {
      "id": "gerlich-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#gerlich-2025",
      "section": "judgement",
      "authors": "Gerlich, M.",
      "year": 2025,
      "title": "AI Tools in Society: Impacts on Cognitive Offloading and the Future of Critical Thinking",
      "publication": "Societies, 15(1), 6",
      "url": "https://www.mdpi.com/2075-4698/15/1/6",
      "grade": "peer-reviewed",
      "method": "Survey and interviews, 666 participants.",
      "finding": "A negative correlation between frequent AI use and critical-thinking scores, mediated by cognitive offloading, strongest among the youngest users.",
      "supports": "An association, with a plausible mechanism.",
      "doesNotSupport": "Causation, and it carries a published correction (Societies 2025, 15(9), 252) which anyone citing it should read alongside.",
      "correction": "https://www.mdpi.com/2075-4698/15/9/252",
      "terms": [
        "critical thinking",
        "cognitive offloading"
      ],
      "relatedPages": [
        "/research/ai-and-critical-thinking"
      ]
    },
    {
      "id": "kosmyna-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#kosmyna-2025",
      "section": "judgement",
      "authors": "Kosmyna, N. et al.",
      "year": 2025,
      "title": "Your Brain on ChatGPT: Accumulation of Cognitive Debt when Using an AI Assistant for Essay Writing Task",
      "publication": "MIT Media Lab preprint, arXiv:2506.08872",
      "url": "https://arxiv.org/abs/2506.08872",
      "grade": "working-paper",
      "method": "EEG study, 54 participants, essay writing with an LLM, a search engine, or unaided.",
      "finding": "The LLM group showed the weakest brain connectivity and the lowest sense of ownership over their own writing.",
      "supports": "Very little on its own. It is suggestive and widely over-quoted.",
      "doesNotSupport": "Anything settled. 54 participants, a preprint, and reproducibility flagged by its own commentators. Treat claims of proof with suspicion.",
      "terms": [
        "cognitive debt",
        "critical thinking"
      ],
      "relatedPages": [
        "/research/ai-and-critical-thinking"
      ]
    },
    {
      "id": "parasuraman-manzey-2010",
      "citeAs": "https://thesuperskills.com/research/evidence#parasuraman-manzey-2010",
      "section": "judgement",
      "authors": "Parasuraman, R. and Manzey, D. H.",
      "year": 2010,
      "title": "Complacency and Bias in Human Use of Automation: An Attentional Integration",
      "publication": "Human Factors, 52(3)",
      "url": "https://journals.sagepub.com/doi/10.1177/0018720810376055",
      "grade": "peer-reviewed",
      "method": "Review across aviation, medicine and military domains.",
      "finding": "Automation bias and complacency appear in novices and experts alike, resist training, and worsen under workload.",
      "supports": "That under-questioning automated advice is a robust, decades-old finding, not a novelty of the AI era.",
      "doesNotSupport": "The size of the effect for generative AI, which is far less predictable than the automation studied here.",
      "terms": [
        "automation bias",
        "automation complacency"
      ],
      "relatedPages": [
        "/research/human-ai-decision-making",
        "/research/ai-and-human-judgement"
      ]
    },
    {
      "id": "bastani-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#bastani-2025",
      "section": "learning",
      "authors": "Bastani, H., Bastani, O., Sungu, A., Ge, H., Kabakci, O. and Mariman, R.",
      "year": 2025,
      "title": "Generative AI Without Guardrails Can Harm Learning: Evidence from High School Mathematics",
      "publication": "Proceedings of the National Academy of Sciences, 122(26)",
      "url": "https://papers.ssrn.com/sol3/papers.cfm?abstract_id=4895486",
      "grade": "peer-reviewed",
      "method": "Field experiment, nearly 1,000 high-school students, three arms: unrestricted GPT-4, a hints-only tutor, and a control.",
      "finding": "Grades rose 48 percent with unrestricted access and 127 percent with the tutor while the tool was present. With access removed, the unrestricted group scored 17 percent LOWER than students who never had it. The guardrailed tutor largely removed the harm.",
      "supports": "That the design of the interface, not the presence of AI, decides whether people learn. The single most useful result in this literature.",
      "doesNotSupport": "What a guardrailed interface should look like for professional work. It was school mathematics over a bounded period.",
      "terms": [
        "learning",
        "deskilling",
        "missed reps"
      ],
      "relatedPages": [
        "/research/how-humans-learn-with-ai",
        "/research/chro-guide-to-ai"
      ]
    },
    {
      "id": "ericsson-1993",
      "citeAs": "https://thesuperskills.com/research/evidence#ericsson-1993",
      "section": "learning",
      "authors": "Ericsson, K. A., Krampe, R. T. and Tesch-Romer, C.",
      "year": 1993,
      "title": "The Role of Deliberate Practice in the Acquisition of Expert Performance",
      "publication": "Psychological Review, 100(3), 363-406",
      "url": "https://eric.ed.gov/?id=EJ471947",
      "grade": "peer-reviewed",
      "method": "Two studies of violinists and pianists in Berlin.",
      "finding": "Sets out deliberate practice: effortful, targeted activity at the edge of current ability, with feedback, sustained over years.",
      "supports": "That expert performance is built through a specific kind of effortful practice rather than exposure.",
      "doesNotSupport": "How much of the difference between performers practice explains. See the Macnamara and Maitra re-examination below.",
      "terms": [
        "deliberate practice",
        "expertise"
      ],
      "relatedPages": [
        "/research/how-humans-learn-with-ai"
      ]
    },
    {
      "id": "macnamara-maitra-2019",
      "citeAs": "https://thesuperskills.com/research/evidence#macnamara-maitra-2019",
      "section": "learning",
      "authors": "Macnamara, B. N. and Maitra, M.",
      "year": 2019,
      "title": "The role of deliberate practice in expert performance: revisiting Ericsson, Krampe and Tesch-Romer (1993)",
      "publication": "Royal Society Open Science, 6, 190327",
      "url": "https://royalsocietypublishing.org/doi/10.1098/rsos.190327",
      "grade": "peer-reviewed",
      "method": "Direct replication and re-analysis of the 1993 study.",
      "finding": "Accumulated practice explained considerably less of the difference between performers than the original is usually taken to claim.",
      "supports": "That the quality and design of practice matters more than the count. Included here deliberately, because it complicates the argument this research relies on.",
      "doesNotSupport": "That practice does not matter. It does; the simple dose-response reading is what fails.",
      "terms": [
        "deliberate practice",
        "expertise"
      ],
      "relatedPages": [
        "/research/how-humans-learn-with-ai"
      ]
    },
    {
      "id": "bjork-desirable-difficulties",
      "citeAs": "https://thesuperskills.com/research/evidence#bjork-desirable-difficulties",
      "section": "learning",
      "authors": "Bjork, E. L. and Bjork, R. A.",
      "year": 2011,
      "title": "Making Things Hard on Yourself, But in a Good Way: Creating Desirable Difficulties to Enhance Learning",
      "publication": "In Psychology and the Real World, Worth Publishers",
      "url": "https://bjorklab.psych.ucla.edu/wp-content/uploads/sites/13/2016/04/EBjork_RBjork_2011.pdf",
      "grade": "peer-reviewed",
      "method": "Synthesis of decades of laboratory work on spacing, interleaving and retrieval practice.",
      "finding": "Conditions that make study feel harder improve long-term retention; conditions that make it feel fluent improve immediate performance and worsen retention. Learners systematically mistake fluency for learning.",
      "supports": "That the subjective sense of learning is an unreliable guide to whether learning occurred. Directly relevant, because AI makes work feel fluent.",
      "doesNotSupport": "That AI-assisted work is equivalent to a fluent study condition. That is an inference, and it is ours.",
      "terms": [
        "desirable difficulties",
        "learning"
      ],
      "relatedPages": [
        "/research/how-humans-learn-with-ai"
      ]
    },
    {
      "id": "brynjolfsson-2023",
      "citeAs": "https://thesuperskills.com/research/evidence#brynjolfsson-2023",
      "section": "learning",
      "authors": "Brynjolfsson, E., Li, D. and Raymond, L.",
      "year": 2023,
      "title": "Generative AI at Work",
      "publication": "NBER Working Paper 31161; Quarterly Journal of Economics, 2025",
      "url": "https://www.nber.org/papers/w31161",
      "grade": "peer-reviewed",
      "method": "Field study of 5,179 customer-support agents with staged rollout of an AI assistant.",
      "finding": "Productivity rose 14 percent on average, 34 percent for the newest and least experienced staff, and barely at all for the most skilled.",
      "supports": "That AI transfers expert patterns to novices, immediately and at scale, and that it raises the floor far more than the ceiling.",
      "doesNotSupport": "Whether those novices became experts. It measures output, not development, over months rather than years.",
      "terms": [
        "productivity",
        "expertise",
        "synthetic seniority"
      ],
      "relatedPages": [
        "/research/how-humans-learn-with-ai",
        "/research/staying-valuable-in-the-age-of-ai"
      ]
    },
    {
      "id": "vaccaro-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#vaccaro-2024",
      "section": "collaboration",
      "authors": "Vaccaro, M., Almaatouq, A. and Malone, T.",
      "year": 2024,
      "title": "When combinations of humans and AI are useful: a systematic review and meta-analysis",
      "publication": "Nature Human Behaviour, 8, 2293-2303",
      "url": "https://www.nature.com/articles/s41562-024-02024-1",
      "grade": "peer-reviewed",
      "method": "Preregistered systematic review and meta-analysis: 106 experimental studies, 370 effect sizes, published January 2020 to June 2023.",
      "finding": "Human-AI combinations performed significantly WORSE on average than the better of human or AI alone (Hedges' g = -0.23). Losses concentrated in decision-making; gains in content creation. Pairing gained where humans beat the AI and lost where the AI beat humans.",
      "supports": "That adding a human is not a control, and that undesigned pairing can subtract. The most under-absorbed result in the field.",
      "doesNotSupport": "That human-AI teams are useless. The benchmark is an oracle-selected best performer, which you rarely know in advance. Also predates current frontier models.",
      "terms": [
        "human in the loop",
        "human-AI collaboration",
        "oversight"
      ],
      "relatedPages": [
        "/research/human-ai-decision-making",
        "/research/how-should-leaders-respond-to-ai"
      ]
    },
    {
      "id": "dellacqua-2023",
      "citeAs": "https://thesuperskills.com/research/evidence#dellacqua-2023",
      "section": "collaboration",
      "authors": "Dell'Acqua, F. et al.",
      "year": 2023,
      "title": "Navigating the Jagged Technological Frontier: Field Experimental Evidence of the Effects of AI on Knowledge Worker Productivity and Quality",
      "publication": "Harvard Business School and BCG working paper",
      "url": "https://papers.ssrn.com/sol3/papers.cfm?abstract_id=4573321",
      "grade": "working-paper",
      "method": "Field experiment, 758 BCG consultants, tasks inside and just outside GPT-4's competence.",
      "finding": "Inside the frontier, AI-assisted consultants were dramatically better and faster. Outside it, they performed worse than consultants with no AI at all.",
      "supports": "That model competence is jagged rather than smooth, and that confident output suppresses scrutiny at exactly the wrong moment.",
      "doesNotSupport": "Where the frontier runs in your domain. That is local and must be learned.",
      "terms": [
        "jagged frontier",
        "automation bias",
        "verification"
      ],
      "relatedPages": [
        "/research/human-ai-decision-making",
        "/research/why-learn-to-prompt-is-weak-career-advice"
      ]
    },
    {
      "id": "dietvorst-2015",
      "citeAs": "https://thesuperskills.com/research/evidence#dietvorst-2015",
      "section": "collaboration",
      "authors": "Dietvorst, B. J., Simmons, J. P. and Massey, C.",
      "year": 2015,
      "title": "Algorithm Aversion: People Erroneously Avoid Algorithms After Seeing Them Err",
      "publication": "Journal of Experimental Psychology: General, 144(1)",
      "url": "https://marketing.wharton.upenn.edu/wp-content/uploads/2016/10/Dietvorst-Simmons-Massey-2014.pdf",
      "grade": "peer-reviewed",
      "method": "Five experiments.",
      "finding": "After seeing an algorithm err, people abandon it even when it demonstrably outperforms them.",
      "supports": "That trust in a model moves for reasons unrelated to its accuracy.",
      "doesNotSupport": "That this holds for conversational AI, which is far more recent and feels different to use.",
      "terms": [
        "algorithm aversion",
        "trust"
      ],
      "relatedPages": [
        "/research/human-ai-decision-making"
      ]
    },
    {
      "id": "logg-2019",
      "citeAs": "https://thesuperskills.com/research/evidence#logg-2019",
      "section": "collaboration",
      "authors": "Logg, J. M., Minson, J. A. and Moore, D. A.",
      "year": 2019,
      "title": "Algorithm Appreciation: People Prefer Algorithmic to Human Judgment",
      "publication": "Organizational Behavior and Human Decision Processes, 151, 90-103",
      "url": "https://www.jennlogg.com/uploads/2/8/9/2/2892148/algorithm_appreciation__logg_minson_moore_2019_.pdf",
      "grade": "peer-reviewed",
      "method": "Six experiments on estimates and forecasts.",
      "finding": "People often weight algorithmic advice MORE heavily than human advice. Domain experts are the notable exception.",
      "supports": "Together with Dietvorst, that miscalibration runs in both directions and cannot be fixed by telling people to use judgement.",
      "doesNotSupport": "Which tendency dominates in any given workplace.",
      "terms": [
        "algorithm appreciation",
        "trust"
      ],
      "relatedPages": [
        "/research/human-ai-decision-making"
      ]
    },
    {
      "id": "zamfirescu-pereira-2023",
      "citeAs": "https://thesuperskills.com/research/evidence#zamfirescu-pereira-2023",
      "section": "collaboration",
      "authors": "Zamfirescu-Pereira, J. D., Wong, R. Y., Hartmann, B. and Yang, Q.",
      "year": 2023,
      "title": "Why Johnny Can't Prompt: How Non-AI Experts Try (and Fail) to Design LLM Prompts",
      "publication": "CHI 2023",
      "url": "https://dl.acm.org/doi/10.1145/3544548.3581388",
      "grade": "peer-reviewed",
      "method": "Design probe study with non-experts using a purpose-built prompt design tool.",
      "finding": "Non-experts approached prompting opportunistically rather than systematically, over-generalised from single successes and failures, and struggled to form an accurate model of the system.",
      "supports": "That prompting is genuinely harder than it looks, which is the strongest case FOR teaching it.",
      "doesNotSupport": "That this persists. It used 2023 models, and providers are actively engineering the difficulty away.",
      "terms": [
        "prompt engineering"
      ],
      "relatedPages": [
        "/research/why-learn-to-prompt-is-weak-career-advice"
      ]
    },
    {
      "id": "ayers-2023",
      "citeAs": "https://thesuperskills.com/research/evidence#ayers-2023",
      "section": "humanness",
      "authors": "Ayers, J. W. et al.",
      "year": 2023,
      "title": "Comparing Physician and Artificial Intelligence Chatbot Responses to Patient Questions Posted to a Public Social Media Forum",
      "publication": "JAMA Internal Medicine, 183(6), 589-596",
      "url": "https://pure.johnshopkins.edu/en/publications/comparing-physician-and-artificial-intelligence-chatbot-responses/",
      "grade": "peer-reviewed",
      "method": "Cross-sectional study, 195 real patient questions from a public forum, blind-rated by licensed healthcare professionals.",
      "finding": "Chatbot responses were rated good or very good quality 78.5 percent of the time against 22.1 percent for physicians, and empathetic or very empathetic 45.1 percent against 4.6 percent.",
      "supports": "That on the observable, textual performance of empathy, the machine already wins comfortably. The claim that empathy is safe from AI is empirically wrong.",
      "doesNotSupport": "That a machine can care for anyone. Doctors answering strangers free of charge between patients are not doing the job they trained for.",
      "terms": [
        "empathy",
        "what stays human"
      ],
      "relatedPages": [
        "/research/what-stays-human",
        "/research/outsourced-recognition"
      ]
    },
    {
      "id": "yin-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#yin-2024",
      "section": "humanness",
      "authors": "Yin, Y., Jia, N. and Wakslak, C. J.",
      "year": 2024,
      "title": "AI can help people feel heard, but an AI label diminishes this impact",
      "publication": "PNAS, 121(14), e2319112121",
      "url": "https://pure.psu.edu/en/publications/ai-can-help-people-feel-heard-but-an-ai-label-diminishes-this-imp/",
      "grade": "peer-reviewed",
      "method": "Experiments comparing AI-generated and human-written responses, with and without disclosure.",
      "finding": "AI-generated replies made recipients feel MORE heard than replies from untrained humans, and labelling the reply as AI removed the advantage.",
      "supports": "That the value of recognition is not in the words but in the belief that a person chose to attend to you. The most clarifying study in this debate.",
      "doesNotSupport": "That the label effect is stable. Norms around disclosed AI assistance are moving, and nobody has measured this over time.",
      "terms": [
        "recognition",
        "outsourced recognition",
        "empathy"
      ],
      "relatedPages": [
        "/research/outsourced-recognition",
        "/research/what-stays-human"
      ]
    },
    {
      "id": "doshi-hauser-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#doshi-hauser-2024",
      "section": "humanness",
      "authors": "Doshi, A. R. and Hauser, O. P.",
      "year": 2024,
      "title": "Generative AI enhances individual creativity but reduces the collective diversity of novel content",
      "publication": "Science Advances, 10(28)",
      "url": "https://discovery.ucl.ac.uk/id/eprint/10195027/",
      "grade": "peer-reviewed",
      "method": "Online experiment, 293 writers producing short fiction and 600 evaluators.",
      "finding": "AI-assisted stories were rated more creative, better written and more enjoyable, with the largest gains for the least creative writers, and were markedly more similar to one another.",
      "supports": "That individual creative quality and collective creative range move in opposite directions. A social dilemma: every writer is right to use it, and the literature gets duller.",
      "doesNotSupport": "That this generalises beyond one short creative task with one form of assistance.",
      "terms": [
        "creativity",
        "homogenisation"
      ],
      "relatedPages": [
        "/research/what-stays-human"
      ]
    },
    {
      "id": "autor-thompson-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#autor-thompson-2025",
      "section": "work",
      "authors": "Autor, D. and Thompson, N.",
      "year": 2025,
      "title": "Expertise",
      "publication": "NBER Working Paper 33941; Journal of the European Economic Association, 23(4), 1203-1271",
      "url": "https://www.nber.org/papers/w33941",
      "grade": "peer-reviewed",
      "method": "Four decades of task data across 303 US occupations, 1980-2018, with a novel content-agnostic measure of task expertise.",
      "finding": "Automation that removed the LESS expert tasks raised wages and reduced employment. Automation that removed the EXPERT tasks lowered wages and increased employment.",
      "supports": "That which tasks are automated matters more than how many, and gives a testable way to ask whether a given role is appreciating or commoditising.",
      "doesNotSupport": "Anything measured about generative AI. The data ends in 2018, so this is a lens, not a forecast.",
      "terms": [
        "expertise",
        "task automation",
        "human capability"
      ],
      "relatedPages": [
        "/research/staying-valuable-in-the-age-of-ai",
        "/research/will-ai-replace-my-job"
      ]
    },
    {
      "id": "humlum-vestergaard-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#humlum-vestergaard-2025",
      "section": "work",
      "authors": "Humlum, A. and Vestergaard, E.",
      "year": 2025,
      "title": "Still Waters, Rapid Currents: Early Labor Market Transformation under Generative AI",
      "publication": "NBER Working Paper 33777, revised March 2026",
      "url": "https://www.nber.org/papers/w33777",
      "grade": "working-paper",
      "method": "Adoption surveys linked to administrative labour records, roughly 25,000 workers across 7,000 Danish workplaces in 11 exposed occupations.",
      "finding": "Precise null effects on earnings and hours two years after ChatGPT, ruling out effects larger than 2 percent, alongside substantial task reorganisation and new tasks in AI oversight and integration.",
      "supports": "That the structure of work moves well before earnings do, and that pay is the slowest available indicator.",
      "doesNotSupport": "That the same holds elsewhere. Denmark is high-trust, high-wage and heavily unionised, and two years is early.",
      "terms": [
        "labour market",
        "task reorganisation",
        "AI adoption"
      ],
      "relatedPages": [
        "/research/ai-workforce-strategy",
        "/research/will-ai-replace-my-job"
      ]
    },
    {
      "id": "eloundou-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#eloundou-2024",
      "section": "work",
      "authors": "Eloundou, T., Manning, S., Mishkin, P. and Rock, D.",
      "year": 2024,
      "title": "GPTs are GPTs: Labor market impact potential of LLMs",
      "publication": "Science, 384(6702), 1306-1308",
      "url": "https://arxiv.org/abs/2303.10130",
      "grade": "peer-reviewed",
      "method": "Human and model ratings of task exposure across occupational task descriptions.",
      "finding": "Around 80 percent of US workers could have at least 10 percent of tasks affected; about 19 percent could see at least half affected.",
      "supports": "Where pressure is likely to fall across the occupational structure.",
      "doesNotSupport": "That any job will be lost. This is exposure, not displacement, and the authors say so explicitly. It is the most misquoted number in the field.",
      "terms": [
        "task exposure",
        "labour market"
      ],
      "relatedPages": [
        "/research/will-ai-replace-my-job"
      ]
    },
    {
      "id": "bick-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#bick-2024",
      "section": "work",
      "authors": "Bick, A., Blandin, A. and Deming, D. J.",
      "year": 2024,
      "title": "The Rapid Adoption of Generative AI",
      "publication": "NBER Working Paper 32966",
      "url": "https://www.nber.org/papers/w32966",
      "grade": "working-paper",
      "method": "Nationally representative US surveys of generative AI use at work and at home.",
      "finding": "By late 2024, nearly 40 percent of US adults aged 18-64 used generative AI and 23 percent of employed respondents had used it for work in the previous week, but only 1 to 5 percent of all work hours were assisted.",
      "supports": "Enormous reach, thin penetration into actual hours. Adoption is not transformation.",
      "doesNotSupport": "Quality of use. Self-reported use counts any use at all.",
      "terms": [
        "AI adoption"
      ],
      "relatedPages": [
        "/research/ai-workforce-strategy",
        "/research/why-learn-to-prompt-is-weak-career-advice"
      ]
    },
    {
      "id": "wef-foj-2018",
      "citeAs": "https://thesuperskills.com/research/evidence#wef-foj-2018",
      "section": "institutional",
      "authors": "World Economic Forum",
      "year": 2018,
      "title": "The Future of Jobs Report 2018",
      "publication": "World Economic Forum, Geneva, September 2018",
      "url": "https://www.weforum.org/publications/the-future-of-jobs-report-2018/",
      "grade": "institutional-survey",
      "method": "Employer survey via the WEF membership community.",
      "finding": "Set out expected skill demand to 2022, with analytical thinking and innovation, active learning and creativity leading the list.",
      "supports": "What the field expected in 2018. Its predictions are now checkable, which is why it is here.",
      "doesNotSupport": "A representative picture of employers. Respondents are drawn from a self-selected membership network.",
      "terms": [
        "human capability",
        "skills"
      ],
      "relatedPages": [
        "/research/human-skills-in-the-age-of-ai"
      ]
    },
    {
      "id": "wef-foj-2020",
      "citeAs": "https://thesuperskills.com/research/evidence#wef-foj-2020",
      "section": "institutional",
      "authors": "World Economic Forum",
      "year": 2020,
      "title": "The Future of Jobs Report 2020",
      "publication": "World Economic Forum, Geneva, October 2020",
      "url": "https://www.weforum.org/publications/the-future-of-jobs-report-2020/",
      "grade": "institutional-survey",
      "method": "Employer survey, conducted during the first year of the pandemic.",
      "finding": "Named critical thinking and problem solving as leading skills, and forecast large-scale reskilling need.",
      "supports": "What the field expected in 2020, including a pandemic-shaped view of remote work.",
      "doesNotSupport": "A clean read on AI. The 2020 edition is dominated by COVID-era disruption.",
      "terms": [
        "human capability",
        "skills",
        "critical thinking"
      ],
      "relatedPages": [
        "/research/human-skills-in-the-age-of-ai"
      ]
    },
    {
      "id": "wef-foj-2023",
      "citeAs": "https://thesuperskills.com/research/evidence#wef-foj-2023",
      "section": "institutional",
      "authors": "World Economic Forum",
      "year": 2023,
      "title": "The Future of Jobs Report 2023",
      "publication": "World Economic Forum, Geneva, April 2023",
      "url": "https://www.weforum.org/publications/the-future-of-jobs-report-2023/",
      "grade": "institutional-survey",
      "method": "Employer survey on expectations to 2027.",
      "finding": "Analytical thinking leads, with creative thinking second, and a growing emphasis on self-efficacy skills.",
      "supports": "The first post-ChatGPT edition, published five months after launch.",
      "doesNotSupport": "Considered judgement on generative AI. It was fielded too early for that.",
      "terms": [
        "human capability",
        "skills"
      ],
      "relatedPages": [
        "/research/human-skills-in-the-age-of-ai"
      ]
    },
    {
      "id": "wef-foj-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#wef-foj-2025",
      "section": "institutional",
      "authors": "World Economic Forum",
      "year": 2025,
      "title": "The Future of Jobs Report 2025",
      "publication": "World Economic Forum, Geneva, January 2025",
      "url": "https://www.weforum.org/publications/the-future-of-jobs-report-2025/",
      "grade": "institutional-survey",
      "method": "Employer survey on expectations to 2030.",
      "finding": "Analytical thinking is the most valued core skill, and skills gaps are named the single biggest barrier to business transformation.",
      "supports": "What employers say they want, which is a real and useful signal about demand.",
      "doesNotSupport": "What employers actually do. Stated skill preference and hiring behaviour diverge routinely.",
      "terms": [
        "human capability",
        "skills",
        "analytical thinking"
      ],
      "relatedPages": [
        "/research/human-skills-in-the-age-of-ai",
        "/research/ai-workforce-strategy"
      ]
    },
    {
      "id": "oecd-skills-outlook-2023",
      "citeAs": "https://thesuperskills.com/research/evidence#oecd-skills-outlook-2023",
      "section": "institutional",
      "authors": "OECD",
      "year": 2023,
      "title": "OECD Skills Outlook 2023: Skills for a Resilient Green and Digital Transition",
      "publication": "OECD Publishing, Paris, November 2023",
      "url": "https://www.oecd.org/en/publications/oecd-skills-outlook-2023_27452f29-en.html",
      "grade": "institutional-modelling",
      "method": "Secondary analysis of OECD data including PISA and PIAAC, not a new survey.",
      "finding": "Analyses the skills required for green and digital transitions across member economies.",
      "supports": "A cross-national, methodologically transparent baseline on skills, from data collected to a documented standard.",
      "doesNotSupport": "Anything AI-specific and current. The underlying data collection predates the generative-AI period.",
      "terms": [
        "skills",
        "human capability"
      ],
      "relatedPages": [
        "/research/human-skills-in-the-age-of-ai"
      ]
    },
    {
      "id": "deloitte-hct-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#deloitte-hct-2025",
      "section": "institutional",
      "authors": "Deloitte",
      "year": 2025,
      "title": "2025 Global Human Capital Trends",
      "publication": "Deloitte Insights",
      "url": "https://www.deloitte.com/us/en/insights/topics/talent/human-capital-trends/2025.html",
      "grade": "institutional-survey",
      "method": "Around 10,000 business and HR leaders across 93 countries, plus separate worker, manager and executive surveys and 25 or more executive interviews.",
      "finding": "Frames the worker-organisation relationship as a set of unresolved tensions rather than a set of solved problems.",
      "supports": "Scale and breadth of practitioner sentiment.",
      "doesNotSupport": "Causal claims. It is a sentiment survey by a firm that sells the remedies it recommends, and should be read with that in view.",
      "terms": [
        "workforce",
        "human capability"
      ],
      "relatedPages": [
        "/research/ai-workforce-strategy",
        "/research/chro-guide-to-ai"
      ]
    },
    {
      "id": "microsoft-wti-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#microsoft-wti-2025",
      "section": "institutional",
      "authors": "Microsoft and LinkedIn",
      "year": 2025,
      "title": "2025 Work Trend Index Annual Report: The Year the Frontier Firm is Born",
      "publication": "Microsoft WorkLab, April 2025",
      "url": "https://www.microsoft.com/en-us/worklab/work-trend-index/2025-the-year-the-frontier-firm-is-born",
      "grade": "institutional-survey",
      "method": "31,000 knowledge workers across 31 markets, plus LinkedIn labour data and Microsoft 365 telemetry.",
      "finding": "Describes the emergence of firms organised around human-agent teams and a shift towards workers managing AI agents.",
      "supports": "Large-scale, current sentiment plus real product telemetry, which few others have.",
      "doesNotSupport": "Independence. Microsoft sells the tools whose adoption it is measuring, and telemetry measures usage rather than value.",
      "terms": [
        "AI agents",
        "AI adoption",
        "workforce"
      ],
      "relatedPages": [
        "/research/ai-agents-and-human-judgement",
        "/research/ai-workforce-strategy"
      ]
    },
    {
      "id": "stanford-hai-index-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#stanford-hai-index-2025",
      "section": "institutional",
      "authors": "Stanford HAI",
      "year": 2025,
      "title": "The 2025 AI Index Report",
      "publication": "Stanford Institute for Human-Centered AI, April 2025",
      "url": "https://hai.stanford.edu/ai-index/2025-ai-index-report",
      "grade": "compiled-review",
      "method": "Aggregation of many third-party sources across eight chapters, with public-opinion data from Ipsos and Pew.",
      "finding": "The most comprehensive annual account of AI capability, investment, adoption and public attitudes.",
      "supports": "An authoritative baseline on what AI systems can do and how they are spreading.",
      "doesNotSupport": "Much about human capability. It measures the machine side of the equation, which is precisely the gap this evidence base exists to fill.",
      "terms": [
        "AI capability",
        "AI adoption"
      ],
      "relatedPages": [
        "/research/ai-workforce-strategy"
      ]
    },
    {
      "id": "mckinsey-new-future-of-work-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#mckinsey-new-future-of-work-2024",
      "section": "institutional",
      "authors": "McKinsey Global Institute",
      "year": 2024,
      "title": "A new future of work: The race to deploy AI and raise skills in Europe and beyond",
      "publication": "McKinsey Global Institute, May 2024",
      "url": "https://www.mckinsey.com/mgi/our-research/a-new-future-of-work-the-race-to-deploy-ai-and-raise-skills-in-europe-and-beyond",
      "grade": "institutional-modelling",
      "method": "Modelling for 2022-2030 across nine EU countries, the UK and the US, plus a survey of 1,100 or more C-suite executives in five countries.",
      "finding": "Projects large-scale occupational transitions and rising demand for social, emotional and higher cognitive skills.",
      "supports": "A transparent, well-documented scenario model, useful for direction.",
      "doesNotSupport": "What will happen. Scenario models are assumption-driven, and McKinsey's prior transition estimates have moved substantially between editions.",
      "terms": [
        "labour market",
        "skills"
      ],
      "relatedPages": [
        "/research/ai-workforce-strategy",
        "/research/will-ai-replace-my-job"
      ]
    },
    {
      "id": "mckinsey-deltas-2021",
      "citeAs": "https://thesuperskills.com/research/evidence#mckinsey-deltas-2021",
      "section": "institutional",
      "authors": "McKinsey and Company",
      "year": 2021,
      "title": "Defining the skills citizens will need in the future world of work",
      "publication": "McKinsey Public and Social Sector Practice, June 2021",
      "url": "https://www.mckinsey.com/industries/public-sector/our-insights/defining-the-skills-citizens-will-need-in-the-future-world-of-work",
      "grade": "institutional-survey",
      "method": "Online psychometric survey of 18,000 people across 15 countries, fielded 2019; 56 elements in 13 skill groups.",
      "finding": "Identifies distinct elements of talent, the DELTAs, associated with employment, income and job satisfaction.",
      "supports": "An unusually large individual-level dataset on skills and outcomes, which is rare in this literature.",
      "doesNotSupport": "Anything about AI. It was fielded in 2019, before the generative-AI period entirely.",
      "terms": [
        "skills",
        "human capability"
      ],
      "relatedPages": [
        "/research/human-skills-in-the-age-of-ai"
      ]
    },
    {
      "id": "nfer-skills-imperative-2035",
      "citeAs": "https://thesuperskills.com/research/evidence#nfer-skills-imperative-2035",
      "section": "institutional",
      "authors": "NFER",
      "year": 2023,
      "title": "The Skills Imperative 2035: An analysis of the demand for skills in the labour market in 2035 (Working Paper 3)",
      "publication": "National Foundation for Educational Research, with the University of Sheffield, funded by the Nuffield Foundation, May 2023",
      "url": "https://www.nfer.ac.uk/publications/the-skills-imperative-2035-an-analysis-of-the-demand-for-skills-in-the-labour-market-in-2035/",
      "grade": "institutional-modelling",
      "method": "161 skills from the US O*NET database mapped to UK occupational codes, combined with UK employment projections.",
      "finding": "Projects rising demand for a set of essential employment skills in the UK to 2035.",
      "supports": "A rare UK-specific, independently funded, methodologically documented projection.",
      "doesNotSupport": "Precision. It maps US skill data onto UK occupations, and a revised working paper corrects coding errors in the underlying labour force survey.",
      "terms": [
        "skills",
        "labour market"
      ],
      "relatedPages": [
        "/research/human-skills-in-the-age-of-ai"
      ]
    },
    {
      "id": "mckinsey-agents-robots-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#mckinsey-agents-robots-2026",
      "section": "institutional",
      "authors": "McKinsey Global Institute",
      "year": 2026,
      "title": "Agents, robots, and us: How AI reshapes work and skills in Europe",
      "publication": "McKinsey Global Institute, May 2026",
      "url": "https://www.mckinsey.com/mgi/our-research/agents-robots-and-us-how-ai-reshapes-work-and-skills-in-europe",
      "grade": "institutional-modelling",
      "method": "Task-level automation modelling across ten European economies covering more than 75 percent of regional labour force and GDP, plus job-postings data.",
      "finding": "Models how agentic AI and robotics together reshape task composition and skill demand across Europe.",
      "supports": "The most current institutional modelling of the agentic shift, and useful for the direction of task change.",
      "doesNotSupport": "Outcomes. Task exposure modelling has consistently over-predicted the pace of realised change.",
      "terms": [
        "AI agents",
        "task automation",
        "skills"
      ],
      "relatedPages": [
        "/research/ai-agents-and-human-judgement",
        "/research/ai-workforce-strategy"
      ]
    },
    {
      "id": "unicef-ai-children-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#unicef-ai-children-2025",
      "section": "institutional",
      "authors": "UNICEF Innocenti",
      "year": 2025,
      "title": "Guidance on AI and Children, Version 3.0: Recommendations for AI policies and systems that uphold child rights",
      "publication": "UNICEF Innocenti, December 2025",
      "url": "https://www.unicef.org/innocenti/reports/policy-guidance-ai-children",
      "grade": "institutional-modelling",
      "method": "Expert advisory group, multi-stakeholder consultation, peer review and a twelve-country study with children and caregivers.",
      "finding": "Sets out ten requirements and 48 recommendations for AI systems and policy affecting children.",
      "supports": "A rights-based standard developed with children rather than about them, which almost nothing else in this space does.",
      "doesNotSupport": "Empirical claims about learning or capability effects. It is a normative guidance document.",
      "terms": [
        "children",
        "education",
        "human agency"
      ],
      "relatedPages": [
        "/research/how-humans-learn-with-ai"
      ]
    },
    {
      "id": "allianz-labour-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#allianz-labour-2026",
      "section": "institutional",
      "authors": "Allianz Research",
      "year": 2026,
      "title": "Happy Labor Day? How geopolitics, immigration and AI will reshape work",
      "publication": "Allianz Trade, 30 April 2026",
      "url": "https://www.allianz-trade.com/en_global/news-insights/economic-insights/Happy-labor-day-How-geopolitics-immigration-AI-reshape-work.html",
      "grade": "institutional-modelling",
      "method": "Sectoral AI task-exposure estimates combined with national employment structures across the US, UK, Germany, France, Italy and Spain.",
      "finding": "Models the combined effect of AI, demographics and migration on labour supply and task composition.",
      "supports": "A current, cross-country modelling view from outside the consultancy sector.",
      "doesNotSupport": "Measured effects. Like all exposure modelling, it is an estimate of what could be affected.",
      "terms": [
        "labour market",
        "task exposure"
      ],
      "relatedPages": [
        "/research/will-ai-replace-my-job"
      ]
    },
    {
      "id": "pwc-jobs-barometer-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#pwc-jobs-barometer-2025",
      "section": "institutional",
      "authors": "PwC",
      "year": 2025,
      "title": "PwC Global AI Jobs Barometer 2025",
      "publication": "PwC, 2025",
      "url": "https://www.pwc.com/gx/en/issues/artificial-intelligence/job-barometer/2025/report.pdf",
      "grade": "institutional-modelling",
      "method": "Analysis of close to a billion job advertisements across multiple countries.",
      "finding": "Reports wage premiums for AI skills and shifting skill requirements in AI-exposed occupations.",
      "supports": "Large-scale observed labour demand rather than stated preference, which is a genuine strength.",
      "doesNotSupport": "Causation, and job advertisements describe what employers ask for rather than what the work requires.",
      "terms": [
        "labour market",
        "skills"
      ],
      "relatedPages": [
        "/research/will-ai-replace-entry-level-jobs"
      ]
    },
    {
      "id": "budzyn-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#budzyn-2025",
      "section": "frontline",
      "authors": "Budzyn, K., Roman'czyk, M., Kitala, D. et al.",
      "year": 2025,
      "title": "Endoscopist deskilling risk after exposure to artificial intelligence in colonoscopy: a multicentre, observational study",
      "publication": "The Lancet Gastroenterology and Hepatology, 10(10), 896-903. DOI 10.1016/S2468-1253(25)00133-5",
      "url": "https://pubmed.ncbi.nlm.nih.gov/40816301/",
      "grade": "peer-reviewed",
      "method": "Retrospective observational study nested in the ACCEPT trial, four Polish endoscopy centres. 1,443 colonoscopies performed WITHOUT AI assistance (795 before and 648 after AI was introduced) by 19 endoscopists averaging 28 years of experience.",
      "finding": "Adenoma detection rate in unassisted colonoscopy fell from 28.4 percent before AI exposure to 22.4 percent after, a drop of 6.0 percentage points (p=0.0089; adjusted odds ratio 0.69).",
      "supports": "Measured deskilling in highly experienced professionals, in unassisted performance, within months of routine AI exposure. The strongest direct evidence that capability degrades when a tool takes over the judgement, rather than merely a plausible mechanism.",
      "doesNotSupport": "Causation with certainty: it is observational, not randomised, and other changes over the period cannot be fully excluded. It is also one procedure in one country, and detection rate is a proxy for skill rather than skill itself.",
      "terms": [
        "deskilling",
        "capability debt"
      ],
      "relatedPages": [
        "/research/capability-debt",
        "/research/ai-and-human-judgement",
        "/research/how-humans-learn-with-ai"
      ]
    },
    {
      "id": "yu-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#yu-2024",
      "section": "frontline",
      "authors": "Yu, F., Moehring, A., Banerjee, O., Salz, T., Agarwal, N. and Rajpurkar, P.",
      "year": 2024,
      "title": "Heterogeneity and predictors of the effects of AI assistance on radiologists",
      "publication": "Nature Medicine, 30(3), 837-849",
      "url": "https://pubmed.ncbi.nlm.nih.gov/38504016/",
      "grade": "peer-reviewed",
      "method": "140 radiologists, 15 chest X-ray diagnostic tasks, roughly 5,190 observations, randomised AI assistance, with empirical-Bayes shrinkage to separate genuine individual differences from noise.",
      "finding": "The effect of AI assistance diverged sharply between radiologists, from strongly positive to strongly negative. Experience, subspecialty and prior familiarity with AI all failed to predict who would benefit, and lower performers did not consistently gain.",
      "supports": "That the effect of AI assistance on expert performance is individual and currently unpredictable, so a policy of giving everyone the tool will help some professionals and harm others with no way to tell in advance which.",
      "doesNotSupport": "That AI assistance is bad on average, or that the pattern holds outside diagnostic imaging.",
      "terms": [
        "human-AI collaboration",
        "expertise"
      ],
      "relatedPages": [
        "/research/human-ai-decision-making",
        "/research/staying-valuable-in-the-age-of-ai"
      ]
    },
    {
      "id": "lee-nursing-homes-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#lee-nursing-homes-2024",
      "section": "frontline",
      "authors": "Lee, Y. S., Iizuka, T. and Eggleston, K.",
      "year": 2024,
      "title": "Robots and Labor in Nursing Homes",
      "publication": "NBER Working Paper 33116",
      "url": "https://www.nber.org/system/files/working_papers/w33116/w33116.pdf",
      "grade": "working-paper",
      "method": "Original facility-level panel of Japanese nursing homes, using regional robot subsidies as an instrument for adoption.",
      "finding": "Robot adoption RAISED employment and improved retention, most strongly for non-regular staff, reallocated worker effort towards direct care, and improved quality: less use of physical restraint and fewer pressure ulcers.",
      "supports": "That automation can absorb routine physical work and upgrade the human job rather than hollow it. The cleanest counter-case in this evidence base, and it is in care work rather than knowledge work.",
      "doesNotSupport": "That this generalises. Japanese long-term care faces acute labour shortage, so robots substituted for vacancies rather than for people, which is a very particular condition.",
      "terms": [
        "automation",
        "care work"
      ],
      "relatedPages": [
        "/research/what-stays-human",
        "/research/ai-workforce-strategy"
      ]
    },
    {
      "id": "dauth-2021",
      "citeAs": "https://thesuperskills.com/research/evidence#dauth-2021",
      "section": "frontline",
      "authors": "Dauth, W., Findeisen, S., Suedekum, J. and Woessner, N.",
      "year": 2021,
      "title": "The Adjustment of Labor Markets to Robots",
      "publication": "Journal of the European Economic Association, 19(6), 3104-3153",
      "url": "https://academic.oup.com/jeea/article-abstract/19/6/3104/6179884",
      "grade": "peer-reviewed",
      "method": "German administrative worker and plant data, 1994 to 2014, with a shift-share instrument for robot exposure.",
      "finding": "Incumbent workers largely kept their jobs and moved into new, higher-quality tasks within their original plants. The cost fell instead on young labour-market entrants, who shifted away from vocational manufacturing training towards university.",
      "supports": "That the damage from automation falls on skill FORMATION rather than skill possession. Twenty years of German manufacturing data making the missing-rungs argument before anyone applied it to knowledge work.",
      "doesNotSupport": "That generative AI will behave like industrial robots. The technologies and the tasks differ substantially.",
      "terms": [
        "missing rungs",
        "manufacturing",
        "early careers"
      ],
      "relatedPages": [
        "/research/missing-rungs",
        "/research/will-ai-replace-entry-level-jobs"
      ]
    },
    {
      "id": "kanazawa-2022",
      "citeAs": "https://thesuperskills.com/research/evidence#kanazawa-2022",
      "section": "frontline",
      "authors": "Kanazawa, K., Kawaguchi, D., Shigeoka, H. and Watanabe, Y.",
      "year": 2022,
      "title": "AI, Skill, and Productivity: The Case of Taxi Drivers",
      "publication": "NBER Working Paper 30612; published in Management Science, 72(2), 1376-1388 (2026)",
      "url": "https://www.nber.org/papers/w30612",
      "grade": "peer-reviewed",
      "method": "Driver-level data from a Japanese taxi fleet through the rollout of an AI demand-prediction system.",
      "finding": "Productivity gains accrued almost entirely to LOW-skilled drivers, narrowing the gap between best and worst by 14 percent.",
      "supports": "That the novice-boost pattern found in customer support and software also appears in manual frontline work.",
      "doesNotSupport": "Included deliberately as a disconfirming case for a tidy story. Anyone arguing that AI levels up white-collar workers while degrading frontline ones has to explain this result, which runs the other way.",
      "terms": [
        "productivity",
        "expertise"
      ],
      "relatedPages": [
        "/research/staying-valuable-in-the-age-of-ai"
      ]
    },
    {
      "id": "nilsson-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#nilsson-2025",
      "section": "frontline",
      "authors": "Nilsson, A. et al. (Karolinska Institutet)",
      "year": 2025,
      "title": "Algorithmic management is associated with psychological distress, musculoskeletal pain, and occupational accidents: a cross-sectional study in logistics",
      "publication": "International Archives of Occupational and Environmental Health, 98",
      "url": "https://link.springer.com/article/10.1007/s00420-025-02180-5",
      "grade": "peer-reviewed",
      "method": "Survey of Swedish logistics workers, February to July 2024, 978 respondents (592 drivers, 378 warehouse), using an eleven-item algorithmic-management exposure scale with adjusted models.",
      "finding": "Higher exposure to algorithmic management was associated with greater psychological distress, more occupational accidents and more musculoskeletal pain.",
      "supports": "That where AI meets frontline work the output is measured in bodies rather than in output quality, a category almost entirely absent from white-collar productivity research.",
      "doesNotSupport": "Causation: it is cross-sectional and self-reported. Workers under strain may also perceive management as more algorithmic.",
      "terms": [
        "algorithmic management",
        "logistics"
      ],
      "relatedPages": [
        "/research/ai-workforce-strategy",
        "/research/how-should-leaders-respond-to-ai"
      ]
    },
    {
      "id": "kesavan-2022",
      "citeAs": "https://thesuperskills.com/research/evidence#kesavan-2022",
      "section": "frontline",
      "authors": "Kesavan, S., Lambert, S. J., Williams, J. C. and Pendem, P. K.",
      "year": 2022,
      "title": "Doing Well by Doing Good: Improving Retail Store Performance with Responsible Scheduling Practices at the Gap, Inc.",
      "publication": "Management Science, 68(11), 7818-7836",
      "url": "https://pubsonline.informs.org/doi/10.1287/mnsc.2021.4291",
      "grade": "peer-reviewed",
      "method": "Randomised field experiment across 28 Gap stores in San Francisco and Chicago over nine months, November 2015 to August 2016, analysed as intent-to-treat.",
      "finding": "Restoring schedule predictability and worker control raised productivity 5.1 percent, with sales up 3.3 percent and labour hours down 1.8 percent.",
      "supports": "That algorithmic optimisation of scheduling is not merely harsh but operationally counterproductive: giving humans back control improved the numbers the algorithm was optimising.",
      "doesNotSupport": "Anything directly about generative AI. It concerns algorithmic scheduling, which is an older and different technology.",
      "terms": [
        "algorithmic management",
        "retail",
        "agency"
      ],
      "relatedPages": [
        "/research/how-should-leaders-respond-to-ai"
      ]
    },
    {
      "id": "iab-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#iab-2024",
      "section": "international",
      "authors": "Institut fur Arbeitsmarkt- und Berufsforschung (IAB), Germany",
      "year": 2024,
      "title": "Folgen des technologischen Wandels fur den Arbeitsmarkt (Consequences of technological change for the labour market: it is above all the highly qualified who feel digitalisation)",
      "publication": "IAB-Kurzbericht 5/2024, published in German",
      "url": "https://doku.iab.de/kurzber/2024/kb2024-05.pdf",
      "grade": "institutional-modelling",
      "method": "The Substituierbarkeitspotenziale series, 2022 wave. Three independent coders score more than 9,000 tasks in the BERUFENET expert database across roughly 4,600 occupations.",
      "finding": "Substitutability rose about ten percentage points for degree-level expert occupations between 2019 and 2022, and was roughly flat for helper occupations. IAB frames AI as relief for skills shortages rather than as displacement.",
      "supports": "That the German expert assessment puts the pressure on the highly qualified, which inverts the assumption that automation threatens the least skilled first.",
      "doesNotSupport": "Realised outcomes. Substitutability is technical potential assessed by coders, not what employers did.",
      "terms": [
        "skills",
        "deskilling",
        "international"
      ],
      "relatedPages": [
        "/research/will-ai-replace-my-job",
        "/research/human-skills-in-the-age-of-ai"
      ]
    },
    {
      "id": "laboria-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#laboria-2024",
      "section": "international",
      "authors": "LaborIA (French Ministry of Labour, Inria and Matrice)",
      "year": 2024,
      "title": "Etude des impacts de l'IA sur le travail: Rapport d'enquete LaborIA Explorer (Study of the impacts of AI on work)",
      "publication": "LaborIA, published in French",
      "url": "https://www.laboria.ai/wp-content/uploads/2024/05/Rapport-denquete-LaborIA-Explorer.pdf",
      "grade": "institutional-survey",
      "method": "Telephone survey of 250 decision-makers in firms with more than 50 staff (42 with AI deployed), longitudinal interviews with 10 decision-makers across three waves, and six ethnographic field sites.",
      "finding": "Names a conflit de rationalite, a clash of rationalities: managers justify AI by error reduction (81 percent), performance (75 percent) and removing drudgery (74 percent), while fieldwork shows workers becoming the system's de facto trainers.",
      "supports": "That what management believes AI is doing and what workers experience it doing can diverge systematically inside the same organisation. A work-psychology and ergonomics frame rather than task-exposure modelling.",
      "doesNotSupport": "Scale. The qualitative core rests on six sites and ten repeated interviews.",
      "terms": [
        "organisation design",
        "international"
      ],
      "relatedPages": [
        "/research/ai-workforce-strategy",
        "/research/how-should-leaders-respond-to-ai"
      ]
    },
    {
      "id": "diwabe-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#diwabe-2025",
      "section": "international",
      "authors": "BAuA, ZEW, IAB and BIBB, Germany",
      "year": 2025,
      "title": "Digitalisierung und Wandel der Beschaftigung, DiWaBe 2.0 (Digitalisation and the transformation of employment)",
      "publication": "Bundesanstalt fur Arbeitsschutz und Arbeitsmedizin, published in German",
      "url": "https://www.baua.de/EN/Service/Publications/Report/F2573",
      "grade": "institutional-survey",
      "method": "Representative 2024 survey of roughly 9,800 employees subject to social insurance, linkable to administrative employer and employee records.",
      "finding": "More than half already use AI at work but largely informally. Use ranges from about a third of unqualified workers to around 80 percent of those with a degree or Meister qualification. There was NO difference in training participation between AI users and non-users.",
      "supports": "That AI use is spreading through workplaces without any corresponding increase in training, which is the adoption-without-redesign pattern measured directly at national scale.",
      "doesNotSupport": "What that absence of training does to capability over time. The survey is a single 2024 snapshot.",
      "terms": [
        "AI adoption",
        "skills",
        "international"
      ],
      "relatedPages": [
        "/research/ai-workforce-strategy",
        "/research/chro-guide-to-ai"
      ]
    },
    {
      "id": "jilpt-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#jilpt-2025",
      "section": "international",
      "authors": "Japan Institute for Labour Policy and Training (JILPT)",
      "year": 2025,
      "title": "Survey on the impact of workplace AI adoption on working styles (Research Series No. 256)",
      "publication": "JILPT, published in Japanese, designed with the OECD",
      "url": "https://www.jil.go.jp/institute/research/2025/256.html",
      "grade": "institutional-survey",
      "method": "22,000 employees, stratified on 2020 Census occupation, employment type, sex and age. Fieldwork May to June 2024.",
      "finding": "Only 12.9 percent report any firm AI use and 8.4 percent use it themselves. Among users, reports of improved job quality and wellbeing outweighed reports of decline, and the gain was markedly larger where the employer had consulted staff and funded training.",
      "supports": "That the effect of AI on how work feels is conditional on how the employer introduced it, not determined by the technology. Direct empirical support for the design-rather-than-drift argument, from a 22,000-person sample.",
      "doesNotSupport": "Long-run capability effects, and Japanese adoption rates are far below US levels so the user group is early and unusual.",
      "terms": [
        "AI adoption",
        "agency",
        "international"
      ],
      "relatedPages": [
        "/research/ai-workforce-strategy",
        "/research/design-versus-drift"
      ]
    },
    {
      "id": "kdi-2023",
      "citeAs": "https://thesuperskills.com/research/evidence#kdi-2023",
      "section": "international",
      "authors": "Korea Development Institute (KDI)",
      "year": 2023,
      "title": "Changes in the labour market due to artificial intelligence and policy directions (Research Report 2023-03)",
      "publication": "KDI, published in Korean",
      "url": "https://www.kdi.re.kr/research/reportView?pub_no=18370",
      "grade": "institutional-modelling",
      "method": "Expert and GPT-4 capability ratings applied to Korean occupational profiles, a KDI survey of 800 firms in September 2023, and firm-panel econometrics.",
      "finding": "38.8 percent of jobs are technically automatable across more than 70 percent of their tasks, yet only 2.7 percent of firms with ten or more staff had adopted AI. Realised effects showed no aggregate employment change, lower earnings, and the impact concentrated on YOUNGER, tertiary-educated workers and women.",
      "supports": "That the gap between technical potential and actual adoption is enormous, and that where effects appear they fall on the young and educated rather than the low-skilled.",
      "doesNotSupport": "That the Korean pattern transfers. Korea has unusually high tertiary education rates and a distinctive labour market.",
      "terms": [
        "labour market",
        "early careers",
        "international"
      ],
      "relatedPages": [
        "/research/will-ai-replace-entry-level-jobs",
        "/research/will-ai-replace-my-job"
      ]
    },
    {
      "id": "cas-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#cas-2025",
      "section": "international",
      "authors": "Lu, Y. and Gui, L. (Bulletin of the Chinese Academy of Sciences)",
      "year": 2025,
      "title": "Analysis of the impact of artificial intelligence technology on employment and income in China",
      "publication": "Bulletin of the Chinese Academy of Sciences, 40(4), 642-651, published in Chinese",
      "url": "http://old2022.bulletin.cas.cn/publish_article/2025/4/20250408.htm",
      "grade": "institutional-modelling",
      "method": "Policy synthesis of the Chinese empirical literature and official statistics, under National Social Science Fund major project 23ZDA100.",
      "finding": "Between 2018 and 2023 the substitution effect outweighed complementarity: a one percent rise in industrial robots reduced firm labour demand by 0.18 percent. Roughly 200 million people, 27 percent of employment, are in flexible work with 37 percent social-insurance coverage.",
      "supports": "That in the largest manufacturing economy the measured balance so far has been substitution rather than augmentation, and the policy response centres on social security redesign rather than retraining.",
      "doesNotSupport": "Comparability with service-sector generative AI in high-income economies. This is largely industrial robotics.",
      "terms": [
        "labour market",
        "manufacturing",
        "international"
      ],
      "relatedPages": [
        "/research/will-ai-replace-my-job"
      ]
    },
    {
      "id": "cepal-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#cepal-2026",
      "section": "international",
      "authors": "Jung, J. and Katz, R. (CEPAL / ECLAC)",
      "year": 2026,
      "title": "Impacto economico de la inteligencia artificial en America Latina (The economic impact of artificial intelligence in Latin America)",
      "publication": "United Nations Economic Commission for Latin America and the Caribbean, published in Spanish",
      "url": "https://www.cepal.org/es/publicaciones/81909-impacto-economico-la-inteligencia-artificial-america-latina-transformacion",
      "grade": "institutional-modelling",
      "method": "Theoretical and econometric modelling of AI's macroeconomic effect through skilled-labour productivity across the region.",
      "finding": "The gains run through skilled labour, and the binding constraint across Latin America is human-capital formation and low investment. The regional risk is UNDER-adoption rather than displacement.",
      "supports": "That the framing dominant in rich economies, where the worry is AI doing too much, inverts in middle-income economies, where the worry is that it will not arrive at all.",
      "doesNotSupport": "Firm-level outcomes; it is a macro model.",
      "terms": [
        "labour market",
        "skills",
        "international"
      ],
      "relatedPages": [
        "/research/ai-workforce-strategy"
      ]
    },
    {
      "id": "funcas-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#funcas-2026",
      "section": "international",
      "authors": "Rodriguez-Fernandez, M. (Funcas)",
      "year": 2026,
      "title": "Inteligencia artificial y mercado de trabajo en Espana (Artificial intelligence and the labour market in Spain)",
      "publication": "Funcas working paper, published in Spanish",
      "url": "https://www.funcas.es/documentos_trabajo/inteligencia-artificial-y-mercado-de-trabajo-en-espana-exposicion-ocupacional-efectos-sobre-el-empleo-y-adopcion-empresarial/",
      "grade": "institutional-modelling",
      "method": "The Felten AI occupational exposure index remapped to Spanish occupational classifications, combined with the Q4 2025 Labour Force Survey.",
      "finding": "Spain shows medium-high exposure at 27.4 percent but low automation risk at 5.9 percent, against an OECD average near 12 percent, because of its interpersonal and physical occupational mix.",
      "supports": "That national occupational structure, not technology, determines exposure. An economy weighted towards interpersonal and physical work is structurally less automatable.",
      "doesNotSupport": "Outcomes. Exposure indices remain estimates of what could be affected.",
      "terms": [
        "task exposure",
        "international"
      ],
      "relatedPages": [
        "/research/will-ai-replace-my-job"
      ]
    },
    {
      "id": "azim-premji-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#azim-premji-2026",
      "section": "international",
      "authors": "Azim Premji University",
      "year": 2026,
      "title": "State of Working India 2026: Youth in the Labour Market",
      "publication": "Azim Premji University, Bengaluru",
      "url": "https://azimpremjiuniversity.edu.in/publications/2026/report/swi-2026",
      "grade": "institutional-modelling",
      "method": "Analysis of National Sample Survey employment data 1983 to 2011, all quarters of the Periodic Labour Force Survey 2017 to 2024, plus AISHE, NCVT-MIS and CMIE-CPHS.",
      "finding": "Roughly five million graduates enter the Indian labour market each year against about 2.8 million finding work, with graduate unemployment near 40 percent for 15 to 25 year olds. The report explicitly declines to attribute this to AI.",
      "supports": "That in the world's most populous labour market the early-career crisis is a demand-side bottleneck that long predates AI. A necessary corrective to reading every graduate hiring problem as an AI story.",
      "doesNotSupport": "Anything about AI's effect in India, which it deliberately does not claim.",
      "terms": [
        "early careers",
        "jobs",
        "international"
      ],
      "relatedPages": [
        "/research/will-ai-replace-entry-level-jobs"
      ]
    },
    {
      "id": "insee-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#insee-2026",
      "section": "international",
      "authors": "INSEE, France",
      "year": 2026,
      "title": "Note de conjoncture: digital investment, artificial intelligence and youth employment",
      "publication": "Institut national de la statistique et des etudes economiques, March 2026, published in French",
      "url": "https://www.insee.fr/fr/statistiques/fichier/8907419/ndc-mars-2026-ecl-AI.pdf",
      "grade": "institutional-modelling",
      "method": "National accounts and quarterly employment files, with an error-correction model estimated 1990Q1 to 2019Q4.",
      "finding": "French employment of 15 to 29 year olds, excluding apprentices, fell 7.4 percent year on year in IT services, 5.8 percent in publishing and 3.7 percent in management consulting in Q4 2025, against minus 0.7 percent across the market sector overall.",
      "supports": "A European national-statistics office finding the same entry-level pattern reported in the US, which makes the signal considerably harder to dismiss as an American artefact.",
      "doesNotSupport": "That AI caused it. INSEE explicitly cautions against attributing the fall to AI alone.",
      "terms": [
        "early careers",
        "jobs",
        "international"
      ],
      "relatedPages": [
        "/research/will-ai-replace-entry-level-jobs",
        "/research/will-ai-replace-my-job"
      ]
    },
    {
      "id": "brynjolfsson-canaries-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#brynjolfsson-canaries-2026",
      "section": "work",
      "authors": "Brynjolfsson, E., Chandar, B. and Chen, R.",
      "year": 2026,
      "title": "Canaries in the Coal Mine? Six Facts about the Recent Employment Effects of Artificial Intelligence",
      "publication": "Stanford Digital Economy Lab, updated August 2026",
      "url": "https://digitaleconomy.stanford.edu/publication/canaries-in-the-coal-mine-six-facts-about-the-recent-employment-effects-of-artificial-intelligence/",
      "grade": "working-paper",
      "method": "ADP payroll microdata covering millions of US workers, comparing employment by age and by occupational AI exposure since the release of ChatGPT.",
      "finding": "No widespread economy-wide displacement. But employment among 22 to 25 year olds in highly AI-exposed occupations sits about 19 percent below where it would be had it tracked similarly aged workers in less-exposed occupations. The divergence runs through reduced hiring rather than increased separations, and declines concentrate in occupations where AI substitutes for human tasks; where it complements, employment is flat or rising, especially for experienced workers.",
      "supports": "That the entry-level effect is real, measurable in payroll data rather than inferred, and specific to substitution rather than to AI exposure as such.",
      "doesNotSupport": "Economy-wide job destruction, which the authors explicitly rule out on current evidence. Nor does it establish causation: youth hiring is sensitive to interest rates, cohort size and hiring freezes, and the design is observational.",
      "terms": [
        "entry-level employment",
        "youth employment",
        "substitution versus complementarity",
        "hiring"
      ],
      "relatedPages": [
        "/research/will-ai-replace-entry-level-jobs",
        "/research/missing-rungs",
        "/research/the-best-writing-on-ai"
      ]
    },
    {
      "id": "metr-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#metr-2025",
      "section": "collaboration",
      "authors": "Model Evaluation and Threat Research (METR)",
      "year": 2025,
      "title": "Measuring the Impact of Early-2025 AI on Experienced Open-Source Developer Productivity",
      "publication": "METR, July 2025",
      "url": "https://metr.org/blog/2025-07-10-early-2025-ai-experienced-os-dev-study/",
      "grade": "working-paper",
      "method": "Randomised controlled trial. 16 experienced open-source developers, 246 real tasks on mature repositories, each task randomly assigned to permit or prohibit AI tools.",
      "finding": "Developers were measured as 19 percent SLOWER when permitted to use AI tools. They had forecast a 24 percent speed-up beforehand, and after completing the tasks and experiencing the slowdown, still estimated AI had made them about 20 percent faster.",
      "supports": "That self-reported productivity gain is an unreliable measure of actual productivity gain, and that the error can run in the opposite direction to the truth by a wide margin.",
      "doesNotSupport": "That AI slows all developers or all software work, and NOT the current position. Sixteen participants, all experienced, all working on large mature codebases they knew well, using early-2025 tooling. METR themselves withdrew this as a current signal on 24 February 2026: see metr-2026-update. The durable finding is the perception gap, not the 19 per cent.",
      "terms": [
        "productivity",
        "perception gap",
        "self-report",
        "software development"
      ],
      "relatedPages": [
        "/research/the-best-writing-on-ai",
        "/research/usage-theatre",
        "/research/what-is-human-ai-collaboration"
      ]
    },
    {
      "id": "pwc-jobs-barometer-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#pwc-jobs-barometer-2026",
      "section": "institutional",
      "authors": "PwC",
      "year": 2026,
      "title": "Global AI Jobs Barometer 2026",
      "publication": "PwC, June 2026",
      "url": "https://www.pwc.com/gx/en/news-room/press-releases/2026/pwc-2026-ai-jobs-barometer.html",
      "grade": "compiled-review",
      "method": "Analysis of close to a billion job advertisements across six continents, comparing skill requirements, wage premiums and job availability by occupational AI exposure.",
      "finding": "Describes the labour market splitting into two paths, with human skills increasingly rewarded. The 2025 edition found a 56 percent wage premium for AI skills, jobs still growing in the most exposed occupations, and skills requirements changing 66 percent faster in the most exposed jobs.",
      "supports": "That demand-side signals in job advertisements are moving fast, and that the picture is not uniformly negative for exposed occupations.",
      "doesNotSupport": "Wage or employment outcomes for actual workers. Job advertisements are stated employer demand, not revealed price or realised hiring, and PwC has a commercial interest in the AI-skills market it is measuring.",
      "terms": [
        "wage premium",
        "skills demand",
        "job advertisements",
        "human skills"
      ],
      "relatedPages": [
        "/research/the-best-writing-on-ai",
        "/research/human-skills-in-the-age-of-ai"
      ]
    },
    {
      "id": "bainbridge-1983",
      "citeAs": "https://thesuperskills.com/research/evidence#bainbridge-1983",
      "section": "collaboration",
      "authors": "Bainbridge, L.",
      "year": 1983,
      "title": "Ironies of Automation",
      "publication": "Automatica, 19(6)",
      "url": "https://doi.org/10.1016/0005-1098(83)90046-8",
      "grade": "peer-reviewed",
      "method": "Theoretical analysis of automated process control systems and the human roles left within them.",
      "finding": "Automating the routine parts of a task leaves the human with the hardest residue, monitoring and exception handling, while removing the routine practice that built the competence to do it. Automation makes the remaining human role harder, not easier.",
      "supports": "That monitoring is a demanding task rather than a light one, and that the design of automation determines whether the human retains the capability to supervise it.",
      "doesNotSupport": "Anything specific to AI. It is a process-control argument from 1983, and its application to generative systems is by analogy rather than by measurement.",
      "terms": [
        "ironies of automation",
        "monitoring",
        "deskilling",
        "oversight"
      ],
      "relatedPages": [
        "/research/human-in-the-loop-is-not-a-safeguard",
        "/research/what-is-deskilling",
        "/research/delegation-boundary-map"
      ]
    },
    {
      "id": "skitka-1999",
      "citeAs": "https://thesuperskills.com/research/evidence#skitka-1999",
      "section": "judgement",
      "authors": "Skitka, L. J., Mosier, K. L. and Burdick, M.",
      "year": 1999,
      "title": "Does automation bias decision-making?",
      "publication": "International Journal of Human-Computer Studies, 51(5)",
      "url": "https://doi.org/10.1006/ijhc.1999.0252",
      "grade": "peer-reviewed",
      "method": "Controlled experiments with a simulated flight task, comparing automated and non-automated decision aids.",
      "finding": "Automated aids produced two distinct error types: errors of omission, missing events the automation failed to flag, and errors of commission, following automated advice that was wrong.",
      "supports": "That automation bias has a measurable structure, and that the two error types require different countermeasures.",
      "doesNotSupport": "That the effect sizes transfer to generative AI or to non-simulated professional settings.",
      "terms": [
        "automation bias",
        "omission",
        "commission",
        "decision aids"
      ],
      "relatedPages": [
        "/research/what-is-automation-bias",
        "/research/human-ai-decision-making"
      ]
    },
    {
      "id": "dzindolet-2003",
      "citeAs": "https://thesuperskills.com/research/evidence#dzindolet-2003",
      "section": "judgement",
      "authors": "Dzindolet, M. T., Peterson, S. A., Pomranky, R. A., Pierce, L. G. and Beck, H. P.",
      "year": 2003,
      "title": "The role of trust in automation reliance",
      "publication": "International Journal of Human-Computer Studies, 58(6)",
      "url": "https://doi.org/10.1016/S1071-5819(03)00038-7",
      "grade": "peer-reviewed",
      "method": "Experiments manipulating what participants were told about an automated aid's reliability and failure modes.",
      "finding": "Explaining why an automated aid might err INCREASED reliance on it, restoring trust even where that trust was unwarranted.",
      "supports": "That awareness training is a weak control, and can move reliance in the opposite direction to the one intended.",
      "doesNotSupport": "That explanation is always counterproductive. The effect is about restoring trust after observed error, not about all forms of transparency.",
      "terms": [
        "trust in automation",
        "explanation",
        "reliance",
        "automation bias"
      ],
      "relatedPages": [
        "/research/what-is-automation-bias",
        "/research/why-does-ai-sound-so-confident",
        "/research/what-is-meaningful-human-oversight"
      ]
    },
    {
      "id": "parasuraman-riley-1997",
      "citeAs": "https://thesuperskills.com/research/evidence#parasuraman-riley-1997",
      "section": "judgement",
      "authors": "Parasuraman, R. and Riley, V.",
      "year": 1997,
      "title": "Humans and Automation: Use, Misuse, Disuse, Abuse",
      "publication": "Human Factors, 39(2)",
      "url": "https://journals.sagepub.com/doi/10.1518/001872097778543886",
      "grade": "peer-reviewed",
      "method": "Review and framework paper synthesising the human-factors literature on how people interact with automated systems.",
      "finding": "Establishes four distinct failure modes: use, misuse through over-reliance, disuse through under-reliance, and abuse through automating without regard for the human consequences.",
      "supports": "That over-reliance and under-reliance are separate problems requiring separate design responses, and that the failure can sit with the deploying organisation rather than the operator.",
      "doesNotSupport": "Quantified effect sizes. It is a framework paper rather than an experiment.",
      "terms": [
        "misuse",
        "disuse",
        "abuse",
        "over-reliance",
        "automation"
      ],
      "relatedPages": [
        "/research/what-is-automation-bias",
        "/research/human-ai-decision-making",
        "/research/design-versus-drift"
      ]
    },
    {
      "id": "acemoglu-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#acemoglu-2024",
      "section": "work",
      "authors": "Acemoglu, D.",
      "year": 2024,
      "title": "The Simple Macroeconomics of AI",
      "publication": "NBER Working Paper 32487; published in Economic Policy, 40(121), 2025",
      "url": "https://www.nber.org/papers/w32487",
      "grade": "peer-reviewed",
      "method": "Task-based macroeconomic model applying Hulten's theorem to existing estimates of AI task exposure and task-level cost savings.",
      "finding": "Estimates total factor productivity gains of no more than 0.66 percent over ten years, revised to under 0.53 percent once the difficulty of hard-to-learn tasks is accounted for. Argues AI is likely to widen the gap between capital and labour income rather than reduce labour income inequality.",
      "supports": "That plausible macroeconomic gains are an order of magnitude smaller than the headline value estimates in circulation.",
      "doesNotSupport": "That AI is unimportant. It models productivity through task-level cost savings, and would not capture effects running through new products, new tasks or capability change.",
      "terms": [
        "productivity",
        "macroeconomics",
        "total factor productivity",
        "inequality"
      ],
      "relatedPages": [
        "/research/how-should-leaders-respond-to-ai",
        "/research/ai-workforce-strategy",
        "/research/will-ai-replace-my-job"
      ]
    },
    {
      "id": "autor-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#autor-2024",
      "section": "work",
      "authors": "Autor, D.",
      "year": 2024,
      "title": "Applying AI to Rebuild Middle Class Jobs",
      "publication": "NBER Working Paper 32140",
      "url": "https://www.nber.org/papers/w32140",
      "grade": "working-paper",
      "method": "Argument and synthesis rather than empirical test, drawing on the author's prior work on task structure and labour demand.",
      "finding": "Argues that AI's distinctive opportunity is to extend the reach of expertise, letting a wider set of workers with complementary knowledge perform higher-stakes decision tasks currently reserved to elite experts.",
      "supports": "That there is a serious, well-argued case for AI as an expertise-widening technology rather than an expertise-replacing one.",
      "doesNotSupport": "That this will happen. The author is explicit that the thesis is an argument about what is possible rather than a forecast, and no evidence yet shows it occurring at scale.",
      "terms": [
        "expertise",
        "middle-skill work",
        "complementarity",
        "task structure"
      ],
      "relatedPages": [
        "/research/will-ai-replace-my-job",
        "/research/staying-valuable-in-the-age-of-ai"
      ]
    },
    {
      "id": "liang-2023",
      "citeAs": "https://thesuperskills.com/research/evidence#liang-2023",
      "section": "learning",
      "authors": "Liang, W., Yuksekgonul, M., Mao, Y., Wu, E. and Zou, J.",
      "year": 2023,
      "title": "GPT detectors are biased against non-native English writers",
      "publication": "Patterns, 4(7)",
      "url": "https://www.sciencedirect.com/science/article/pii/S2666389923001307",
      "grade": "peer-reviewed",
      "method": "Seven widely used GPT detectors evaluated against TOEFL essays by non-native English speakers and essays by US eighth-grade students.",
      "finding": "Detectors misclassified more than half of the non-native essays as AI-generated, an average false positive rate of 61.22 percent, while classifying US eighth-grade essays with near-perfect accuracy. The proposed mechanism is that detectors rely on perplexity, and second-language writing is more predictable.",
      "supports": "That AI detection carries a severe and systematic bias against second-language writers, and that the bias is structural rather than a tuning problem.",
      "doesNotSupport": "That every detector now on the market performs identically. The study tested tools available at the time, and vendors dispute the generalisation.",
      "terms": [
        "AI detection",
        "false positives",
        "academic integrity",
        "fairness"
      ],
      "relatedPages": [
        "/research/does-ai-detection-work",
        "/research/how-to-assess-students-when-ai-can-do-the-assignment"
      ]
    },
    {
      "id": "vonstumm-2011",
      "citeAs": "https://thesuperskills.com/research/evidence#vonstumm-2011",
      "section": "capability",
      "authors": "von Stumm, S., Hell, B. and Chamorro-Premuzic, T.",
      "year": 2011,
      "title": "The Hungry Mind: Intellectual Curiosity Is the Third Pillar of Academic Performance",
      "publication": "Perspectives on Psychological Science, 6(6)",
      "url": "https://journals.sagepub.com/doi/abs/10.1177/1745691611421204",
      "grade": "peer-reviewed",
      "method": "Path-model synthesis of prior meta-analytic correlation matrices. Component samples range from 608 to 28,471.",
      "finding": "Intellectual curiosity predicts academic performance independently of intelligence and effort, which the authors describe as a third pillar.",
      "supports": "That curiosity carries predictive weight that conscientiousness and ability do not account for.",
      "doesNotSupport": "Causation. It is a correlational synthesis. The widely quoted figure of roughly 50,000 students comes from the accompanying press release rather than from the paper itself.",
      "terms": [
        "curiosity"
      ],
      "relatedPages": [
        "/research/superskill-curiosity"
      ]
    },
    {
      "id": "harrison-2011",
      "citeAs": "https://thesuperskills.com/research/evidence#harrison-2011",
      "section": "capability",
      "authors": "Harrison, S. H., Sluss, D. M. and Ashforth, B. E.",
      "year": 2011,
      "title": "Curiosity adapted the cat: The role of trait curiosity in newcomer adaptation",
      "publication": "Journal of Applied Psychology, 96(1)",
      "url": "https://pubmed.ncbi.nlm.nih.gov/21244132/",
      "grade": "peer-reviewed",
      "method": "Longitudinal field study of 123 newcomers across 12 call-centre organisations.",
      "finding": "Specific curiosity predicted information-seeking from colleagues, which in turn was associated with more creative handling of customer problems.",
      "supports": "That curiosity operates through a behavioural mechanism, asking, rather than as a disposition on its own.",
      "doesNotSupport": "That curiosity can be trained into people, or that the effect holds outside newcomer adaptation.",
      "terms": [
        "curiosity"
      ],
      "relatedPages": [
        "/research/superskill-curiosity"
      ]
    },
    {
      "id": "swan-carmelli-1996",
      "citeAs": "https://thesuperskills.com/research/evidence#swan-carmelli-1996",
      "section": "capability",
      "authors": "Swan, G. E. and Carmelli, D.",
      "year": 1996,
      "title": "Curiosity and mortality in aging adults: A 5-year follow-up of the Western Collaborative Group Study",
      "publication": "Psychology and Aging, 11(3)",
      "url": "https://pubmed.ncbi.nlm.nih.gov/8893314/",
      "grade": "peer-reviewed",
      "method": "Prospective cohort. 1,118 men, mean age 70.6 at baseline, with an ancillary sample of 1,035 women.",
      "finding": "Higher curiosity at baseline was associated with survival at five-year follow-up, and state curiosity remained significant after adjustment for other risk factors.",
      "supports": "That the association survives controlling for medical risk factors in one large cohort.",
      "doesNotSupport": "That curiosity extends life. Residual confounding, in particular underlying health driving both curiosity and survival, cannot be excluded.",
      "terms": [
        "curiosity"
      ],
      "relatedPages": [
        "/research/superskill-curiosity"
      ]
    },
    {
      "id": "gino-2018",
      "citeAs": "https://thesuperskills.com/research/evidence#gino-2018",
      "section": "capability",
      "authors": "Gino, F.",
      "year": 2018,
      "title": "The Business Case for Curiosity",
      "publication": "Harvard Business Review, September to October 2018",
      "url": "https://hbr.org/2018/09/the-business-case-for-curiosity",
      "grade": "compiled-review",
      "method": "Survey of more than 3,000 employees across a range of firms, reported in a practitioner magazine.",
      "finding": "Around 92 per cent said curious people bring new ideas to their teams, while about 24 per cent reported feeling curious in their jobs regularly.",
      "supports": "That the stated value of curiosity and the felt experience of it diverge sharply inside organisations.",
      "doesNotSupport": "Any causal link between curiosity and a business outcome. Self-report, not peer-reviewed, and the sample is not nationally representative.",
      "terms": [
        "curiosity"
      ],
      "relatedPages": [
        "/research/superskill-curiosity"
      ]
    },
    {
      "id": "sadri-2011",
      "citeAs": "https://thesuperskills.com/research/evidence#sadri-2011",
      "section": "capability",
      "authors": "Sadri, G., Weber, T. J. and Gentry, W. A.",
      "year": 2011,
      "title": "Empathic emotion and leadership performance: An empirical analysis across 38 countries",
      "publication": "The Leadership Quarterly, 22(5)",
      "url": "https://www.sciencedirect.com/science/article/abs/pii/S1048984311001093",
      "grade": "peer-reviewed",
      "method": "360-degree ratings of 6,731 mid to upper-level managers across 38 countries. Subordinates rated empathy, superiors rated performance.",
      "finding": "Managers rated as more empathic by subordinates received higher performance ratings from their own superiors, with the effect moderated by national power distance.",
      "supports": "That the association holds at scale and across many national contexts.",
      "doesNotSupport": "That empathy training improves performance. The design is cross-sectional and correlational.",
      "terms": [
        "empathy"
      ],
      "relatedPages": [
        "/research/superskill-empathy"
      ]
    },
    {
      "id": "howick-2018",
      "citeAs": "https://thesuperskills.com/research/evidence#howick-2018",
      "section": "capability",
      "authors": "Howick, J., Moscrop, A., Mebius, A. et al.",
      "year": 2018,
      "title": "Effects of empathic and positive communication in healthcare consultations: a systematic review and meta-analysis",
      "publication": "Journal of the Royal Society of Medicine, 111(7)",
      "url": "https://journals.sagepub.com/doi/10.1177/0141076818769477",
      "grade": "peer-reviewed",
      "method": "Systematic review and meta-analysis of 28 randomised trials, 6,017 patients in total. Seven of the 28 tested empathic communication specifically; the remainder tested positive-expectation messaging.",
      "finding": "The seven empathy-specific trials showed a small improvement in pain, anxiety and satisfaction, SMD -0.18, 95 per cent CI -0.32 to -0.03.",
      "supports": "That empathic communication has a measurable effect on patient-reported outcomes.",
      "doesNotSupport": "A large clinical benefit. The authors describe the effect as small, and only a quarter of the pooled trials tested empathy rather than positive framing.",
      "terms": [
        "empathy"
      ],
      "relatedPages": [
        "/research/superskill-empathy"
      ]
    },
    {
      "id": "konrath-2011",
      "citeAs": "https://thesuperskills.com/research/evidence#konrath-2011",
      "section": "capability",
      "authors": "Konrath, S. H., O'Brien, E. H. and Hsing, C.",
      "year": 2011,
      "title": "Changes in Dispositional Empathy in American College Students Over Time: A Meta-Analysis",
      "publication": "Personality and Social Psychology Review, 15(2)",
      "url": "https://journals.sagepub.com/doi/10.1177/1088868310377395",
      "grade": "peer-reviewed",
      "method": "Cross-temporal meta-analysis of 72 samples of American college students, total N 13,737, from 1979 to 2009.",
      "finding": "Empathic Concern fell by 48 per cent and Perspective Taking by 34 per cent across the period, with most of the decline after 2000.",
      "supports": "That self-reported dispositional empathy declined measurably in this population over three decades.",
      "doesNotSupport": "A cause. The authors speculate about individualism and media but test no mechanism. It is also American college students only.",
      "terms": [
        "empathy"
      ],
      "relatedPages": [
        "/research/superskill-empathy"
      ]
    },
    {
      "id": "pulakos-2000",
      "citeAs": "https://thesuperskills.com/research/evidence#pulakos-2000",
      "section": "capability",
      "authors": "Pulakos, E. D., Arad, S., Donovan, M. A. and Plamondon, K. E.",
      "year": 2000,
      "title": "Adaptability in the Workplace: Development of a Taxonomy of Adaptive Performance",
      "publication": "Journal of Applied Psychology, 85(4)",
      "url": "https://psycnet.apa.org/doi/10.1037/0021-9010.85.4.612",
      "grade": "peer-reviewed",
      "method": "Content analysis of over 1,000 critical incidents drawn from 21 jobs, followed by scale validation.",
      "finding": "Eight dimensions of adaptive performance, including handling emergencies, managing stress, solving problems creatively and dealing with uncertain situations.",
      "supports": "That adaptability is decomposable into observable behaviours rather than being a single trait.",
      "doesNotSupport": "That employers reward these dimensions, or that they transfer across every occupation. The taxonomy was derived, not tested against outcomes.",
      "terms": [
        "change readiness",
        "adaptive performance"
      ],
      "relatedPages": [
        "/research/superskill-change-readiness"
      ]
    },
    {
      "id": "huang-2014",
      "citeAs": "https://thesuperskills.com/research/evidence#huang-2014",
      "section": "capability",
      "authors": "Huang, J. L., Ryan, A. M., Zabel, K. L. and Palmer, A.",
      "year": 2014,
      "title": "Personality and Adaptive Performance at Work: A Meta-Analytic Investigation",
      "publication": "Journal of Applied Psychology, 99(1)",
      "url": "https://psycnet.apa.org/doi/10.1037/a0034285",
      "grade": "peer-reviewed",
      "method": "Meta-analysis of 71 independent samples, total N 7,535.",
      "finding": "Emotional stability and ambition predict adaptive performance, and the pattern differs from predictors of routine task performance.",
      "supports": "That individual differences carry predictive weight for adaptive performance specifically.",
      "doesNotSupport": "That adaptive performance is a distinct construct. This paper assumes the construct from earlier taxonomy work rather than establishing it.",
      "terms": [
        "change readiness",
        "adaptive performance"
      ],
      "relatedPages": [
        "/research/superskill-change-readiness"
      ]
    },
    {
      "id": "edmondson-1999",
      "citeAs": "https://thesuperskills.com/research/evidence#edmondson-1999",
      "section": "capability",
      "authors": "Edmondson, A.",
      "year": 1999,
      "title": "Psychological Safety and Learning Behavior in Work Teams",
      "publication": "Administrative Science Quarterly, 44(2)",
      "url": "https://journals.sagepub.com/doi/10.2307/2666999",
      "grade": "peer-reviewed",
      "method": "Multi-method field study of 51 work teams in a single manufacturing company.",
      "finding": "Team psychological safety predicted learning behaviour, which in turn mediated the relationship with team performance.",
      "supports": "That the route from safety to performance runs through learning behaviour rather than directly.",
      "doesNotSupport": "Causation, or generalisation beyond one manufacturing firm. It is a correlational field study.",
      "terms": [
        "change readiness",
        "psychological safety"
      ],
      "relatedPages": [
        "/research/superskill-change-readiness"
      ]
    },
    {
      "id": "defreitas-2023",
      "citeAs": "https://thesuperskills.com/research/evidence#defreitas-2023",
      "section": "capability",
      "authors": "De Freitas, J., Uguralp, A. K., Oguz-Uguralp, Z., Paul, L. A., Tenenbaum, J. and Ullman, T. D.",
      "year": 2023,
      "title": "Self-orienting in human and machine learning",
      "publication": "Nature Human Behaviour, 7",
      "url": "https://www.nature.com/articles/s41562-023-01696-5",
      "grade": "peer-reviewed",
      "method": "Behavioural experiments with 124 human players across custom games, benchmarked against deep reinforcement learning agents.",
      "finding": "Humans were near optimal at working out their own position and capabilities after conditions were altered. The reinforcement learning baselines were far from optimal at the same task.",
      "supports": "That rapid self-orientation after an unexpected change is currently a human advantage over the algorithms tested.",
      "doesNotSupport": "General workplace adaptability. These are simple custom games, the sample is modest, and the comparison is against specific algorithms rather than all AI approaches.",
      "terms": [
        "change readiness"
      ],
      "relatedPages": [
        "/research/superskill-change-readiness"
      ]
    },
    {
      "id": "schlaegel-2021",
      "citeAs": "https://thesuperskills.com/research/evidence#schlaegel-2021",
      "section": "capability",
      "authors": "Schlaegel, C., Richter, N. F. and Taras, V.",
      "year": 2021,
      "title": "Cultural intelligence and work-related outcomes: A meta-analytic examination of joint effects and incremental predictive validity",
      "publication": "Journal of World Business, 56(4)",
      "url": "https://www.sciencedirect.com/science/article/abs/pii/S1090951621000225",
      "grade": "peer-reviewed",
      "method": "Meta-analysis of 70 studies providing 80 independent samples, total N 18,359.",
      "finding": "Cultural intelligence is moderately associated with work-related outcomes, with a reliability-corrected average effect of about .39.",
      "supports": "That the association is consistent across a large body of studies.",
      "doesNotSupport": "Causation, and it does not establish any single dimension as the strongest predictor. The paper is about joint effects across all four dimensions.",
      "terms": [
        "global adaptability",
        "cultural intelligence"
      ],
      "relatedPages": [
        "/research/superskill-global-adaptability"
      ]
    },
    {
      "id": "maddux-galinsky-2009",
      "citeAs": "https://thesuperskills.com/research/evidence#maddux-galinsky-2009",
      "section": "capability",
      "authors": "Maddux, W. W. and Galinsky, A. D.",
      "year": 2009,
      "title": "Cultural Borders and Mental Barriers: The Relationship Between Living Abroad and Creativity",
      "publication": "Journal of Personality and Social Psychology, 96(5)",
      "url": "https://psycnet.apa.org/doi/10.1037/a0014861",
      "grade": "peer-reviewed",
      "method": "Five studies with MBA and undergraduate participants, combining correlational designs with causal priming experiments.",
      "finding": "Time spent living abroad predicted success on creative-insight tasks and creative negotiation outcomes. Time spent travelling abroad did not.",
      "supports": "That adaptation to living in another culture, rather than exposure to it, is what relates to creativity.",
      "doesNotSupport": "A specific effect size for the general population. The samples are business and undergraduate students.",
      "terms": [
        "global adaptability"
      ],
      "relatedPages": [
        "/research/superskill-global-adaptability"
      ]
    },
    {
      "id": "stillman-2018",
      "citeAs": "https://thesuperskills.com/research/evidence#stillman-2018",
      "section": "capability",
      "authors": "Stillman, P. E., Fujita, K., Sheldon, O. and Trope, Y.",
      "year": 2018,
      "title": "From 'Me' to 'We': The Role of Construal Level in Promoting Maximized Joint Outcomes",
      "publication": "Organizational Behavior and Human Decision Processes, 147",
      "url": "https://www.sciencedirect.com/science/article/abs/pii/S0749597817303989",
      "grade": "peer-reviewed",
      "method": "Four laboratory and online experiments, pooled N approximately 691.",
      "finding": "Prompting a higher level of construal led participants to choose options that maximised joint outcomes, including where doing so reduced their own payoff.",
      "supports": "That the level at which a problem is framed changes whether people optimise for themselves or for the whole.",
      "doesNotSupport": "Field behaviour. These are economic games with student and online samples.",
      "terms": [
        "big picture thinking"
      ],
      "relatedPages": [
        "/research/superskill-big-picture-thinking"
      ]
    },
    {
      "id": "mehta-2014",
      "citeAs": "https://thesuperskills.com/research/evidence#mehta-2014",
      "section": "capability",
      "authors": "Mehta, R., Zhu, R. and Meyers-Levy, J.",
      "year": 2014,
      "title": "When Does a Higher Construal Level Increase or Decrease Indulgence? Resolving the Myopia versus Hyperopia Puzzle",
      "publication": "Journal of Consumer Research, 41(2)",
      "url": "https://academic.oup.com/jcr/article/41/2/475/2907518",
      "grade": "peer-reviewed",
      "method": "Multi-study laboratory experiments.",
      "finding": "Where the self is focal, a higher construal level increases indulgence rather than reducing it, reversing the effect the earlier literature predicted.",
      "supports": "That the benefit of stepping back is conditional, and the condition is identifiable.",
      "doesNotSupport": "That distant-future thinking is generally counterproductive. The effect is moderated, not reversed outright.",
      "terms": [
        "big picture thinking"
      ],
      "relatedPages": [
        "/research/superskill-big-picture-thinking"
      ]
    },
    {
      "id": "orlitzky-2003",
      "citeAs": "https://thesuperskills.com/research/evidence#orlitzky-2003",
      "section": "capability",
      "authors": "Orlitzky, M., Schmidt, F. L. and Rynes, S. L.",
      "year": 2003,
      "title": "Corporate Social and Financial Performance: A Meta-Analysis",
      "publication": "Organization Studies, 24(3)",
      "url": "https://journals.sagepub.com/doi/10.1177/0170840603024003910",
      "grade": "peer-reviewed",
      "method": "Meta-analysis of 52 studies, 33,878 observations.",
      "finding": "A positive association between corporate social performance and financial performance, which the authors describe as bidirectional.",
      "supports": "That principled conduct and financial results are not in general tension.",
      "doesNotSupport": "Causation in either direction, and the relationship is not uniform across how social performance is operationalised.",
      "terms": [
        "principled innovation"
      ],
      "relatedPages": [
        "/research/superskill-principled-innovation"
      ]
    },
    {
      "id": "bcg-diversity-2018",
      "citeAs": "https://thesuperskills.com/research/evidence#bcg-diversity-2018",
      "section": "capability",
      "authors": "Lorenzo, R., Voigt, N., Tsusaka, M., Krentz, M. and Abouzahr, K.",
      "year": 2018,
      "title": "How Diverse Leadership Teams Boost Innovation",
      "publication": "Boston Consulting Group",
      "url": "https://www.bcg.com/publications/2018/how-diverse-leadership-teams-boost-innovation",
      "grade": "institutional-survey",
      "method": "Survey of more than 1,700 companies across eight countries.",
      "finding": "Companies with above-average management diversity reported innovation revenue 19 percentage points higher than below-average companies, 45 per cent of total revenue against 26 per cent.",
      "supports": "That reported diversity and reported innovation revenue move together at scale.",
      "doesNotSupport": "Causation, and it is not audited financial data. Note that 19 percentage points is not the same as 19 per cent higher, a distinction frequently lost in citation.",
      "terms": [
        "principled innovation"
      ],
      "relatedPages": [
        "/research/superskill-principled-innovation"
      ]
    },
    {
      "id": "simons-2002",
      "citeAs": "https://thesuperskills.com/research/evidence#simons-2002",
      "section": "capability",
      "authors": "Simons, T.",
      "year": 2002,
      "title": "The High Cost of Lost Trust",
      "publication": "Harvard Business Review, September 2002",
      "url": "https://hbr.org/2002/09/the-high-cost-of-lost-trust",
      "grade": "compiled-review",
      "method": "Survey of more than 6,500 employees at 76 Holiday Inn hotels in the United States and Canada, matched to hotel financial records.",
      "finding": "A one-eighth point improvement in managers' behavioural integrity rating was associated with a 2.5 per cent increase in profitability, roughly 250,000 dollars a year for an average hotel. No other measured aspect of manager behaviour had as large an effect on profits.",
      "supports": "That whether managers are seen to mean what they say tracks measurable financial outcomes.",
      "doesNotSupport": "Causation. It is a correlational study within one hotel chain, published in a practitioner magazine rather than a peer-reviewed journal.",
      "terms": [
        "principled innovation"
      ],
      "relatedPages": [
        "/research/superskill-principled-innovation"
      ]
    },
    {
      "id": "mckinsey-shorttermism-2017",
      "citeAs": "https://thesuperskills.com/research/evidence#mckinsey-shorttermism-2017",
      "section": "capability",
      "authors": "Barton, D., Manyika, J., Koller, T., Palter, R., Godsall, J. and Zoffer, J.",
      "year": 2017,
      "title": "Measuring the Economic Impact of Short-Termism",
      "publication": "McKinsey Global Institute",
      "url": "https://www.mckinsey.com/featured-insights/long-term-capitalism/where-companies-with-a-long-term-view-outperform-their-peers",
      "grade": "institutional-modelling",
      "method": "Corporate Horizon Index applied to 615 large and mid-cap United States public companies, 2001 to 2014.",
      "finding": "Revenue of long-term firms grew cumulatively 47 per cent more than other firms, earnings 36 per cent more, and their share prices recovered faster after the financial crisis.",
      "supports": "That a measurable long-term orientation tracks with stronger cumulative growth in this sample.",
      "doesNotSupport": "Causation, or application outside large United States listed companies. The index is a constructed measure, not an observed policy.",
      "terms": [
        "principled innovation"
      ],
      "relatedPages": [
        "/research/superskill-principled-innovation"
      ]
    },
    {
      "id": "epa-vw-2015",
      "citeAs": "https://thesuperskills.com/research/evidence#epa-vw-2015",
      "section": "capability",
      "authors": "United States Environmental Protection Agency",
      "year": 2015,
      "title": "Notice of Violation, Volkswagen Group",
      "publication": "US EPA enforcement record",
      "url": "https://19january2021snapshot.epa.gov/enforcement/learn-about-volkswagen-violations_.html",
      "grade": "institutional-survey",
      "method": "Regulatory enforcement notice.",
      "finding": "Affected 2.0-litre vehicles emitted nitrogen oxides at up to 40 times the standard in normal driving while appearing compliant in laboratory testing.",
      "supports": "That the defeat device produced a measured gap between test and road conditions of that magnitude.",
      "doesNotSupport": "That the same multiple applies across the range. The separate 3.0-litre notice cited up to nine times.",
      "terms": [
        "principled innovation"
      ],
      "relatedPages": [
        "/research/superskill-principled-innovation"
      ]
    },
    {
      "id": "faa-parc-2013",
      "citeAs": "https://thesuperskills.com/research/evidence#faa-parc-2013",
      "section": "learning",
      "authors": "PARC/CAST Flight Deck Automation Working Group",
      "year": 2013,
      "title": "Operational Use of Flight Path Management Systems: Final Report",
      "publication": "Federal Aviation Administration",
      "url": "https://www.faa.gov/sites/faa.gov/files/aircraft/air_cert/design_approvals/human_factors/OUFPMS_Report.pdf",
      "grade": "institutional-modelling",
      "method": "Working-group synthesis of accident and incident data, operator surveys and prior research. 28 findings, 18 recommendations.",
      "finding": "Identified vulnerabilities in manual handling after transition from automated control, and in the definition, development and retention of those skills. Also found that pilots sometimes rely too much on automated systems and may be reluctant to intervene.",
      "supports": "That a regulator examined automation dependence in a whole industry and named skill retention as a finding.",
      "doesNotSupport": "A measured rate of skill decay. It is a synthesis of findings, not a controlled study.",
      "terms": [
        "automation complacency",
        "deskilling"
      ],
      "relatedPages": [
        "/research/what-professions-can-learn-from-aviation"
      ]
    },
    {
      "id": "bea-af447-2012",
      "citeAs": "https://thesuperskills.com/research/evidence#bea-af447-2012",
      "section": "learning",
      "authors": "Bureau d'Enquetes et d'Analyses",
      "year": 2012,
      "title": "Final Report on the accident on 1 June 2009 to the Airbus A330-203, flight AF 447",
      "publication": "BEA, France",
      "url": "https://bea.aero/docspa/2009/f-cp090601.en/pdf/f-cp090601.en.pdf",
      "grade": "institutional-survey",
      "method": "Statutory accident investigation. Flight recorder analysis and training record review. 25 new safety recommendations.",
      "finding": "The investigation cited the lack of practical training in high-altitude manual handling and in the procedure for speed anomalies among the contributing factors.",
      "supports": "That a formal investigation attributed part of an accident to manual handling practice that had not been maintained.",
      "doesNotSupport": "A general rate of skill loss across the pilot population. It is one accident.",
      "terms": [
        "deskilling"
      ],
      "relatedPages": [
        "/research/what-professions-can-learn-from-aviation"
      ]
    },
    {
      "id": "cfr-121-441",
      "citeAs": "https://thesuperskills.com/research/evidence#cfr-121-441",
      "section": "institutional",
      "authors": "United States Federal Aviation Regulations",
      "year": 1974,
      "title": "14 CFR 121.441, Proficiency checks",
      "publication": "Electronic Code of Federal Regulations",
      "url": "https://www.ecfr.gov/current/title-14/chapter-I/subchapter-G/part-121/subpart-O/section-121.441",
      "grade": "institutional-survey",
      "method": "Binding regulation.",
      "finding": "A pilot in command must pass a proficiency check every 12 calendar months, and within every 6 calendar months either a proficiency check or an approved simulator course.",
      "supports": "That one profession has made recurrent tested practice a legal condition of continuing to work.",
      "doesNotSupport": "That the intervals are calibrated to measured decay curves. The regulation sets a minimum, not an evidence-based optimum.",
      "terms": [],
      "relatedPages": [
        "/research/what-professions-can-learn-from-aviation"
      ]
    },
    {
      "id": "cfr-121-542",
      "citeAs": "https://thesuperskills.com/research/evidence#cfr-121-542",
      "section": "institutional",
      "authors": "United States Federal Aviation Regulations",
      "year": 1981,
      "title": "14 CFR 121.542, Flight crewmember duties, the sterile cockpit rule",
      "publication": "Electronic Code of Federal Regulations",
      "url": "https://www.ecfr.gov/current/title-14/chapter-I/subchapter-G/part-121/subpart-T/section-121.542",
      "grade": "institutional-survey",
      "method": "Binding regulation, Docket 20661, 46 FR 5502, 19 January 1981.",
      "finding": "No crew member may perform any duty during a critical phase of flight other than those required for safe operation. Critical phases include taxi, take-off, landing and all operations below 10,000 feet except cruise.",
      "supports": "That protected attention can be written into law as a condition rather than left to individual discipline.",
      "doesNotSupport": "Anything about automation or skill decay directly. It is a rule about distraction.",
      "terms": [],
      "relatedPages": [
        "/research/what-professions-can-learn-from-aviation"
      ]
    },
    {
      "id": "helmreich-1999",
      "citeAs": "https://thesuperskills.com/research/evidence#helmreich-1999",
      "section": "collaboration",
      "authors": "Helmreich, R. L., Merritt, A. C. and Wilhelm, J. A.",
      "year": 1999,
      "title": "The Evolution of Crew Resource Management Training in Commercial Aviation",
      "publication": "International Journal of Aviation Psychology, 9(1)",
      "url": "https://www.faa.gov/sites/faa.gov/files/2022-11/crmhistory.pdf",
      "grade": "peer-reviewed",
      "method": "Historical and evaluative review of CRM programmes from 1979 onwards.",
      "finding": "CRM began with a 1979 NASA workshop prompted by an NTSB finding that a captain had failed to accept input from junior crew. Line audits show CRM produces the intended behavioural change, though measured attitudes decay over time even with recurrent training.",
      "supports": "That an industry built a training response to a named human-factors failure and measured behaviour rather than opinion.",
      "doesNotSupport": "That CRM reduces accidents. The authors state directly that accident rates are too rare to serve as a validation criterion.",
      "terms": [],
      "relatedPages": [
        "/research/what-professions-can-learn-from-aviation"
      ]
    },
    {
      "id": "molloy-parasuraman-1996",
      "citeAs": "https://thesuperskills.com/research/evidence#molloy-parasuraman-1996",
      "section": "judgement",
      "authors": "Molloy, R. and Parasuraman, R.",
      "year": 1996,
      "title": "Monitoring an Automated System for a Single Failure: Vigilance and Task Complexity Effects",
      "publication": "Human Factors, 38(2)",
      "url": "https://journals.sagepub.com/doi/10.1177/001872089606380211",
      "grade": "peer-reviewed",
      "method": "Laboratory flight-simulation experiments varying task complexity and time on task.",
      "finding": "Detection of a single automation failure degrades with time on task, and the effect is strongest where the automation has been consistently reliable.",
      "supports": "That monitoring performance falls predictably rather than randomly, and that reliability itself is part of the cause.",
      "doesNotSupport": "Field incidence. It is simulation, not accident data.",
      "terms": [
        "automation complacency"
      ],
      "relatedPages": [
        "/research/what-professions-can-learn-from-aviation"
      ]
    },
    {
      "id": "arthur-1998",
      "citeAs": "https://thesuperskills.com/research/evidence#arthur-1998",
      "section": "learning",
      "authors": "Arthur, W., Bennett, W., Stanush, P. L. and McNelly, T. L.",
      "year": 1998,
      "title": "Factors that influence skill decay and retention: A quantitative review and analysis",
      "publication": "Human Performance, 11(1)",
      "url": "https://www.tandfonline.com/doi/abs/10.1207/s15327043hup1101_3",
      "grade": "peer-reviewed",
      "method": "Meta-analysis of 189 independent data points from 53 articles.",
      "finding": "Skill loss ran from d of -0.01 immediately after training to d of -1.4 after more than 365 days of non-use. Physical, natural and speed-based tasks decayed less than cognitive, artificial and accuracy-based tasks.",
      "supports": "That skill decay is measurable, that it is a function of the interval, and that cognitive skills go first.",
      "doesNotSupport": "How fast any particular professional skill decays, or how quickly it can be regained.",
      "terms": [
        "deskilling",
        "capability debt"
      ],
      "relatedPages": [
        "/research/can-you-regain-a-skill-you-have-lost"
      ]
    },
    {
      "id": "murre-dros-2015",
      "citeAs": "https://thesuperskills.com/research/evidence#murre-dros-2015",
      "section": "learning",
      "authors": "Murre, J. M. J. and Dros, J.",
      "year": 2015,
      "title": "Replication and Analysis of Ebbinghaus' Forgetting Curve",
      "publication": "PLoS ONE, 10(7)",
      "url": "https://journals.plos.org/plosone/article?id=10.1371/journal.pone.0120644",
      "grade": "peer-reviewed",
      "method": "Single-subject relearning experiment replicating Ebbinghaus across intervals from 20 minutes to 31 days, 10 replications per interval.",
      "finding": "Relearning to criterion took less time than original learning at every retention interval tested, confirming the savings effect Ebbinghaus reported in 1885.",
      "supports": "That something survives apparent forgetting, and that it shows up as faster relearning rather than as recall.",
      "doesNotSupport": "That professional skills behave like nonsense syllables, or that savings hold at the scale of a career. It is one subject and verbal material.",
      "terms": [],
      "relatedPages": [
        "/research/can-you-regain-a-skill-you-have-lost"
      ]
    },
    {
      "id": "casner-2014",
      "citeAs": "https://thesuperskills.com/research/evidence#casner-2014",
      "section": "learning",
      "authors": "Casner, S. M., Geven, R. W., Recker, M. P. and Schooler, J. W.",
      "year": 2014,
      "title": "The Retention of Manual Flying Skills in the Automated Cockpit",
      "publication": "Human Factors, 56(8)",
      "url": "https://journals.sagepub.com/doi/abs/10.1177/0018720814535628",
      "grade": "peer-reviewed",
      "method": "16 airline pilots flew routine and non-routine scenarios in a Boeing 747-400 simulator with automation level varied.",
      "finding": "Instrument scanning and manual control were mostly intact even where pilots reported little recent practice. The cognitive tasks, tracking position without a map, deciding the next navigational step and recognising instrument failures, showed frequent and significant problems.",
      "supports": "That the hands survive disuse better than the judgement does, in the profession with the most automation experience.",
      "doesNotSupport": "A general rule for knowledge work. The sample is 16 pilots in a simulator.",
      "terms": [
        "deskilling"
      ],
      "relatedPages": [
        "/research/can-you-regain-a-skill-you-have-lost",
        "/research/what-professions-can-learn-from-aviation"
      ]
    },
    {
      "id": "redcross-cpr-2009",
      "citeAs": "https://thesuperskills.com/research/evidence#redcross-cpr-2009",
      "section": "learning",
      "authors": "American Red Cross Advisory Council on First Aid, Aquatics, Safety and Preparedness",
      "year": 2009,
      "title": "Scientific Review: CPR Skill Retention",
      "publication": "American Red Cross",
      "url": "https://www.redcross.org/content/dam/redcross/Health-Safety-Services/scientific-advisory-council/Scientific%20Advisory%20Council%20SCIENTIFIC%20REVIEW%20-%20CPR%20Skill%20Retention.pdf",
      "grade": "compiled-review",
      "method": "Systematic review of 47 articles on CPR skill retention across healthcare and lay populations, with retest intervals from six weeks to 24 months.",
      "finding": "Substantial skill degradation occurs within the first year after training, with declining retention from six to twelve months unless there is refresher training.",
      "supports": "That a life-critical, heavily trained procedural skill decays on a timescale of months without practice.",
      "doesNotSupport": "Any link to patient outcomes. Decay was measured on manikins, not in resuscitations.",
      "terms": [
        "deskilling"
      ],
      "relatedPages": [
        "/research/can-you-regain-a-skill-you-have-lost"
      ]
    },
    {
      "id": "cepeda-2006",
      "citeAs": "https://thesuperskills.com/research/evidence#cepeda-2006",
      "section": "learning",
      "authors": "Cepeda, N. J., Pashler, H., Vul, E., Wixted, J. T. and Rohrer, D.",
      "year": 2006,
      "title": "Distributed Practice in Verbal Recall Tasks: A Review and Quantitative Synthesis",
      "publication": "Psychological Bulletin, 132(3)",
      "url": "https://pubmed.ncbi.nlm.nih.gov/16719566/",
      "grade": "peer-reviewed",
      "method": "Meta-analysis of 839 assessments across 317 experiments in 184 articles.",
      "finding": "Spacing and retention interval act jointly. The gap between practice sessions that produces best retention increases as the target retention interval increases.",
      "supports": "That when practice happens changes how much survives, independently of how much practice there is.",
      "doesNotSupport": "Application to procedural or professional skill. The synthesis covers verbal recall.",
      "terms": [
        "desirable difficulty"
      ],
      "relatedPages": [
        "/research/can-you-regain-a-skill-you-have-lost"
      ]
    },
    {
      "id": "sackett-2023",
      "citeAs": "https://thesuperskills.com/research/evidence#sackett-2023",
      "section": "learning",
      "authors": "Sackett, P. R., Zhang, C., Berry, C. M. and Lievens, F.",
      "year": 2023,
      "title": "Revisiting the design of selection systems in light of new findings regarding the validity of widely used predictors",
      "publication": "Industrial and Organizational Psychology, 16",
      "url": "https://doi.org/10.1017/iop.2023.24",
      "grade": "peer-reviewed",
      "method": "Meta-analytic re-correction of prior personnel-selection meta-analyses, addressing systematic overcorrection for range restriction.",
      "finding": "Work sample validity falls from the widely quoted .54 to .33. Structured interviews fall from .51 to .42 and become the strongest single predictor. Unstructured interviews fall from .38 to .19. General cognitive ability falls from .51 to .31.",
      "supports": "That the numbers most often cited for assessment methods were inflated, and by how much.",
      "doesNotSupport": "That these methods do not work. Work samples and structured interviews remain among the strongest predictors available. It also offers no validity data for AI-era assessment formats.",
      "terms": [],
      "relatedPages": [
        "/research/how-do-you-assess-capability-rather-than-output"
      ]
    },
    {
      "id": "mcdaniel-1994",
      "citeAs": "https://thesuperskills.com/research/evidence#mcdaniel-1994",
      "section": "learning",
      "authors": "McDaniel, M. A., Whetzel, D. L., Schmidt, F. L. and Maurer, S. D.",
      "year": 1994,
      "title": "The Validity of Employment Interviews: A Comprehensive Review and Meta-Analysis",
      "publication": "Journal of Applied Psychology, 79(4)",
      "url": "https://psycnet.apa.org/doi/10.1037/0021-9010.79.4.599",
      "grade": "peer-reviewed",
      "method": "Meta-analysis of 245 validity coefficients from 86,311 individuals.",
      "finding": "Structured interviews predict job performance substantially better than unstructured interviews.",
      "supports": "That structure, rather than the interview itself, is what carries the predictive weight.",
      "doesNotSupport": "A single stable figure for the gap. Estimates of the size of the structure effect vary considerably between meta-analyses.",
      "terms": [],
      "relatedPages": [
        "/research/how-do-you-assess-capability-rather-than-output"
      ]
    },
    {
      "id": "moonen-2013",
      "citeAs": "https://thesuperskills.com/research/evidence#moonen-2013",
      "section": "learning",
      "authors": "Moonen-van Loon, J. M. W., Overeem, K., Donkers, H. H. L. M., van der Vleuten, C. P. M. and Driessen, E. W.",
      "year": 2013,
      "title": "Composite reliability of a workplace-based assessment toolbox for postgraduate medical education",
      "publication": "Advances in Health Sciences Education, 18(5)",
      "url": "https://link.springer.com/article/10.1007/s10459-013-9450-z",
      "grade": "peer-reviewed",
      "method": "Generalisability study of 12,779 workplace-based assessments from 953 medical residents.",
      "finding": "A reliability coefficient of 0.80 required eight mini-CEX observations, nine DOPS or nine multi-source feedback rounds. Combined in a portfolio the requirement fell to seven, eight and one respectively.",
      "supports": "That observing someone at work can reach defensible reliability, and roughly how many observations that takes.",
      "doesNotSupport": "That these scores predict later performance or patient outcomes. It measures consistency, not criterion validity. A single observation is not reliable.",
      "terms": [],
      "relatedPages": [
        "/research/how-do-you-assess-capability-rather-than-output"
      ]
    },
    {
      "id": "prasad-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#prasad-2025",
      "section": "learning",
      "authors": "Prasad, S. et al.",
      "year": 2025,
      "title": "Enhancing medical assessment strategies: a comparative study between structured, traditional and hybrid viva-voce assessment",
      "publication": "BMC Medical Education, 25",
      "url": "https://link.springer.com/article/10.1186/s12909-025-07428-9",
      "grade": "peer-reviewed",
      "method": "151 medical students assessed by two examiners across three viva formats, compared on reliability and perceived fairness.",
      "finding": "Traditional unstructured viva showed significant inter-examiner variability. Structured formats improved fairness and coverage. The best format reached a reliability of 0.663, which is moderate rather than high.",
      "supports": "That oral examination can be made fairer by structuring it, and that unstructured viva has a measurable examiner problem.",
      "doesNotSupport": "That oral assessment is highly reliable. Even the best format tested was moderate, and it is one institution.",
      "terms": [],
      "relatedPages": [
        "/research/how-do-you-assess-capability-rather-than-output"
      ]
    },
    {
      "id": "euaiact-art12",
      "citeAs": "https://thesuperskills.com/research/evidence#euaiact-art12",
      "section": "institutional",
      "authors": "European Union",
      "year": 2024,
      "title": "Article 12, Record-keeping, Regulation (EU) 2024/1689",
      "publication": "Official Journal of the European Union",
      "url": "https://artificialintelligenceact.eu/article/12/",
      "grade": "institutional-survey",
      "method": "Binding regulation.",
      "finding": "High-risk AI systems must technically allow automatic recording of events over the system lifetime, to enable identification of risk situations, post-market monitoring and monitoring of operation. Article 19 requires providers to keep those logs for at least six months.",
      "supports": "That the capability to reconstruct what a system did is now a legal requirement rather than good practice.",
      "doesNotSupport": "That anyone must read the logs, or that a decision must be reviewable at the level of the individual case.",
      "terms": [],
      "relatedPages": [
        "/research/how-do-you-audit-an-ai-assisted-decision"
      ]
    },
    {
      "id": "cjeu-c203-22",
      "citeAs": "https://thesuperskills.com/research/evidence#cjeu-c203-22",
      "section": "institutional",
      "authors": "Court of Justice of the European Union",
      "year": 2025,
      "title": "Dun and Bradstreet Austria, Case C-203/22",
      "publication": "CJEU judgment of 27 February 2025",
      "url": "https://curia.europa.eu/site/upload/docs/application/pdf/2025-02/cp250022en.pdf",
      "grade": "institutional-survey",
      "method": "Preliminary ruling interpreting GDPR Articles 15(1)(h) and 22.",
      "finding": "A controller must describe the procedure and principles actually applied so the person can understand which of their data was used and how. Disclosing the algorithm is not a sufficient explanation, and a blanket trade-secret refusal is not permitted.",
      "supports": "That explanation now has a legal standard, and that the standard is comprehension rather than disclosure.",
      "doesNotSupport": "A right to source code or model weights. The balance with trade secrets is decided case by case.",
      "terms": [],
      "relatedPages": [
        "/research/how-do-you-audit-an-ai-assisted-decision"
      ]
    },
    {
      "id": "nist-airmf-2023",
      "citeAs": "https://thesuperskills.com/research/evidence#nist-airmf-2023",
      "section": "institutional",
      "authors": "National Institute of Standards and Technology",
      "year": 2023,
      "title": "AI Risk Management Framework 1.0",
      "publication": "NIST, 26 January 2023",
      "url": "https://www.nist.gov/itl/ai-risk-management-framework",
      "grade": "institutional-survey",
      "method": "Voluntary framework organised around four functions: Govern, Map, Measure and Manage.",
      "finding": "Provides a common structure and vocabulary for AI risk management, with a companion Playbook and a 2024 generative AI profile.",
      "supports": "That a shared vocabulary exists that a board is likely to recognise.",
      "doesNotSupport": "Compliance with anything. It is voluntary and confers no legal status.",
      "terms": [],
      "relatedPages": [
        "/research/how-do-you-audit-an-ai-assisted-decision"
      ]
    },
    {
      "id": "iso-42001-2023",
      "citeAs": "https://thesuperskills.com/research/evidence#iso-42001-2023",
      "section": "institutional",
      "authors": "International Organization for Standardization",
      "year": 2023,
      "title": "ISO/IEC 42001:2023, Artificial intelligence management system",
      "publication": "ISO, December 2023",
      "url": "https://www.iso.org/standard/42001",
      "grade": "institutional-survey",
      "method": "Certifiable management system standard developed by ISO/IEC JTC 1/SC 42.",
      "finding": "Specifies requirements for an AI management system on a Plan-Do-Check-Act structure, against which an organisation can be certified by a third party.",
      "supports": "That an auditable management standard for AI now exists and can be certified.",
      "doesNotSupport": "Legal compliance. Certification against ISO 42001 is not the same as meeting the EU AI Act, and the two are routinely conflated.",
      "terms": [],
      "relatedPages": [
        "/research/how-do-you-audit-an-ai-assisted-decision"
      ]
    },
    {
      "id": "sharma-2023",
      "citeAs": "https://thesuperskills.com/research/evidence#sharma-2023",
      "section": "judgement",
      "authors": "Sharma, M., Tong, M., Korbak, T. et al.",
      "year": 2023,
      "title": "Towards Understanding Sycophancy in Language Models",
      "publication": "ICLR 2024, arXiv 2310.13548",
      "url": "https://arxiv.org/abs/2310.13548",
      "grade": "peer-reviewed",
      "method": "Analysis of five production AI assistants across four free-form text generation tasks, plus analysis of the human preference datasets used to train them.",
      "finding": "All five assistants consistently exhibited sycophancy. Both humans and the preference models trained on their judgements prefer convincingly written sycophantic responses over correct ones a non-negligible share of the time, and optimising against those preference models sometimes sacrifices truthfulness.",
      "supports": "That sycophancy is a predictable consequence of training on human preference, rather than an incidental defect of one product.",
      "doesNotSupport": "Any effect on the quality of a user's decisions. It measures model behaviour, not user outcomes.",
      "terms": [],
      "relatedPages": [
        "/research/how-do-i-get-ai-to-challenge-me"
      ]
    },
    {
      "id": "cheng-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#cheng-2026",
      "section": "judgement",
      "authors": "Cheng, M., Lee, C., Khadpe, P., Yu, S., Han, D. and Jurafsky, D.",
      "year": 2026,
      "title": "Sycophantic AI Decreases Prosocial Intentions and Promotes Dependence",
      "publication": "Science",
      "url": "https://arxiv.org/abs/2510.01395",
      "grade": "peer-reviewed",
      "method": "Eleven models tested against human responses on interpersonal advice, plus two preregistered experiments with 1,604 participants, including a live-interaction study using a real personal conflict.",
      "finding": "Models affirmed users' actions about 50 per cent more often than humans did, including 47 per cent endorsement on prompts describing clearly harmful behaviour. Interacting with a sycophantic model reduced participants' willingness to repair an interpersonal conflict and increased their conviction that they were in the right. Participants rated the sycophantic model higher quality, trusted it more, and were more willing to use it again.",
      "supports": "That agreement changes what people subsequently do, and that preference runs in the opposite direction from benefit.",
      "doesNotSupport": "An effect on factual or analytical decisions. The scenarios are interpersonal advice, not technical judgement.",
      "terms": [],
      "relatedPages": [
        "/research/how-do-i-get-ai-to-challenge-me"
      ]
    },
    {
      "id": "openai-sycophancy-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#openai-sycophancy-2025",
      "section": "judgement",
      "authors": "OpenAI",
      "year": 2025,
      "title": "Sycophancy in GPT-4o: what happened and what we are doing about it",
      "publication": "OpenAI, 29 April and 2 May 2025",
      "url": "https://openai.com/index/sycophancy-in-gpt-4o/",
      "grade": "institutional-survey",
      "method": "First-party incident postmortem covering an update released on 25 April 2025 and rolled back from 28 April.",
      "finding": "OpenAI attributed the behaviour to weighting short-term user feedback too heavily, which weakened the reward signal that had been holding sycophancy in check. Offline evaluations and A/B tests looked positive. The problem was flagged only by informal qualitative checks, which were overridden.",
      "supports": "That the failure is a measurement failure as much as a training one, and that satisfaction metrics can rise while the product gets worse.",
      "doesNotSupport": "Anything independently verified. It is a company's account of its own incident.",
      "terms": [],
      "relatedPages": [
        "/research/how-do-i-get-ai-to-challenge-me"
      ]
    },
    {
      "id": "dubois-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#dubois-2026",
      "section": "judgement",
      "authors": "Dubois, M., Ududec, C., Summerfield, C. and Luettgau, L.",
      "year": 2026,
      "title": "Ask don't tell: Reducing sycophancy in large language models",
      "publication": "arXiv 2602.23971",
      "url": "https://arxiv.org/pdf/2602.23971",
      "grade": "working-paper",
      "method": "Factorial experiments on three frontier models using 40 debatable questions rendered in 11 framings, rated over ten epochs, with a follow-up test across 600 personas.",
      "finding": "Framing input as a statement rather than a question raised sycophancy by roughly 24 percentage points. Prompting the model to convert a user's statement into a question before answering reduced sycophancy more than instructing it not to be sycophantic.",
      "supports": "That how the user phrases the input matters more than telling the model to behave.",
      "doesNotSupport": "That adversarial or devil's advocate prompting works. That was not tested, and no controlled evidence for it was found.",
      "terms": [],
      "relatedPages": [
        "/research/how-do-i-get-ai-to-challenge-me"
      ]
    },
    {
      "id": "gaube-workflow-2022",
      "citeAs": "https://thesuperskills.com/research/evidence#gaube-workflow-2022",
      "section": "judgement",
      "authors": "Who Goes First? Influences of Human-AI Workflow on Decision Making in Clinical Imaging",
      "year": 2022,
      "title": "Who Goes First? Influences of Human-AI Workflow on Decision Making in Clinical Imaging",
      "publication": "arXiv 2205.09696",
      "url": "https://arxiv.org/abs/2205.09696",
      "grade": "working-paper",
      "method": "Between-subjects study, 19 veterinary radiologists reviewing 40 X-rays for 33 findings, comparing seeing the AI output alongside the image against committing to a provisional diagnosis first.",
      "finding": "Final diagnoses matched the AI 91 per cent of the time when the AI was seen first, against 89 per cent when the clinician committed first. Where the AI flagged a finding, agreement was 71 per cent against 65 per cent. The anchoring produced only marginal diagnostic gains because of over-reliance on erroneous advice.",
      "supports": "That forming a view before seeing the machine's answer measurably changes the final judgement.",
      "doesNotSupport": "Generalisation. Nineteen participants in one clinical speciality, and the effect sizes are small.",
      "terms": [
        "automation bias"
      ],
      "relatedPages": [
        "/research/how-do-i-get-ai-to-challenge-me",
        "/research/human-at-the-start"
      ]
    },
    {
      "id": "jakesch-2023",
      "citeAs": "https://thesuperskills.com/research/evidence#jakesch-2023",
      "section": "humanness",
      "authors": "Jakesch, M., Bhat, A., Buschek, D., Zalmanson, L. and Naaman, M.",
      "year": 2023,
      "title": "Co-Writing with Opinionated Language Models Affects Users' Views",
      "publication": "CHI 2023, ACM",
      "url": "https://arxiv.org/abs/2302.00560",
      "grade": "peer-reviewed",
      "method": "Online experiment with 1,506 participants writing a post on whether social media is good for society, assisted by a tool configured to argue for or against. Opinions rated by 500 independent judges, plus a post-task attitude survey.",
      "finding": "The opinionated model changed both the opinions expressed in participants' writing and their own opinions in the subsequent attitude survey. The effect held among participants who had ample time to write independently. The authors call it latent persuasion.",
      "supports": "That writing assistance moves what people think, not only what they type.",
      "doesNotSupport": "Generalisation across topics, or persistence after the task. One topic, one configuration, self-reported attitudes.",
      "terms": [],
      "relatedPages": [
        "/research/how-do-i-keep-my-own-voice-when-using-ai"
      ]
    },
    {
      "id": "joshi-vogel-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#joshi-vogel-2025",
      "section": "humanness",
      "authors": "Joshi, N. and Vogel, D.",
      "year": 2025,
      "title": "Writing with AI Lowers Psychological Ownership, but Longer Prompts Can Help",
      "publication": "ACM Conversational User Interfaces 2025",
      "url": "https://arxiv.org/pdf/2404.03108",
      "grade": "peer-reviewed",
      "method": "Two within-subjects experiments, 31 and 34 participants, writing short stories across conditions from a three-word prompt to writing unaided.",
      "finding": "Psychological ownership rose steadily with prompt length, from a mean of 1.80 with a three-word prompt to 6.29 writing alone. The benefit plateaued once the prompt reached roughly the length of the target text, and no AI-assisted condition reached the ownership of writing unaided.",
      "supports": "That how much of yourself you put in changes whether the output feels like yours, and by a large margin.",
      "doesNotSupport": "Anything about professional or long-form writing. Short fiction, small samples.",
      "terms": [],
      "relatedPages": [
        "/research/how-do-i-keep-my-own-voice-when-using-ai"
      ]
    },
    {
      "id": "sourati-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#sourati-2026",
      "section": "humanness",
      "authors": "Sourati, Z., Ziabari, A. S. and Dehghani, M.",
      "year": 2026,
      "title": "The Homogenizing Effect of Large Language Models on Human Expression and Thought",
      "publication": "Trends in Cognitive Sciences",
      "url": "https://arxiv.org/abs/2508.01491",
      "grade": "compiled-review",
      "method": "Synthesis across linguistics, psychology, cognitive science and computer science. Not an original experiment.",
      "finding": "Argues that models reflect and reinforce dominant styles while marginalising alternatives, and that reliance on a small number of systems amplifies convergence across users.",
      "supports": "That the concern is taken seriously across several fields rather than being a commentator's intuition.",
      "doesNotSupport": "Any specific stylistic feature converging. It presents no new measurement of its own.",
      "terms": [],
      "relatedPages": [
        "/research/how-do-i-keep-my-own-voice-when-using-ai"
      ]
    },
    {
      "id": "bls-projections-eval",
      "citeAs": "https://thesuperskills.com/research/evidence#bls-projections-eval",
      "section": "work",
      "authors": "United States Bureau of Labor Statistics",
      "year": 2018,
      "title": "Occupational Projections Evaluation, 2006 to 2016",
      "publication": "BLS Employment Projections programme",
      "url": "https://www.bls.gov/emp/evaluations/2006-2016-occupational.htm",
      "grade": "institutional-modelling",
      "method": "The agency's own retrospective scoring of its 2006 ten-year projections for 840 detailed occupations against actual 2016 outcomes.",
      "finding": "BLS correctly projected whether an occupation would grow or decline 78 per cent of the time, but correctly projected which occupations would grow faster than the economy as a whole only 57 per cent of the time. Projected average occupational growth was 10.4 per cent against an actual 3.6 per cent, the gap driven by a recession the projections could not foresee.",
      "supports": "That the only organisation which scores its own occupational forecasts gets the useful question, relative growth, barely better than a coin toss, and that its largest errors come from shocks.",
      "doesNotSupport": "That forecasting is worthless. Direction of change at aggregate level was reasonably good. It also predates generative AI, and no equivalent scored record exists for a technological discontinuity.",
      "terms": [],
      "relatedPages": [
        "/research/what-should-i-tell-my-children-to-study"
      ]
    },
    {
      "id": "ifs-lifetime-returns-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#ifs-lifetime-returns-2026",
      "section": "work",
      "authors": "Waltmann, B.",
      "year": 2026,
      "title": "New Estimates of the Impact of Undergraduate Degrees on Lifetime Earnings",
      "publication": "Institute for Fiscal Studies, commissioned by the Department for Education",
      "url": "https://ifs.org.uk/publications/new-estimates-impact-undergraduate-degrees-lifetime-earnings",
      "grade": "institutional-modelling",
      "method": "Administrative linkage of school, university and tax records for the whole 2002 English GCSE cohort, tracked to age 37 with earnings simulated to 67.",
      "finding": "Average net lifetime return to a degree is around 100,000 pounds, with very large variation by subject. Medicine and economics exceed 400,000 pounds on average. Creative arts, philosophy and languages show low or negative average returns. Around 20 per cent of women and 30 per cent of men are projected to see a negative net return.",
      "supports": "That subject choice carries a far larger financial spread than the decision to attend at all.",
      "doesNotSupport": "What any subject will return to someone choosing today. The authors explicitly decline to model structural change including AI.",
      "terms": [],
      "relatedPages": [
        "/research/what-should-i-tell-my-children-to-study"
      ]
    },
    {
      "id": "georgetown-major-payoff-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#georgetown-major-payoff-2025",
      "section": "work",
      "authors": "Georgetown University Center on Education and the Workforce",
      "year": 2025,
      "title": "The Major Payoff: Evaluating Earnings and Employment Outcomes Across Bachelor's Degrees",
      "publication": "Georgetown CEW",
      "url": "https://cew.georgetown.edu/cew-reports/major-payoff/",
      "grade": "institutional-survey",
      "method": "American Community Survey earnings and employment data across 152 majors for prime-age workers and 142 for early career.",
      "finding": "Median prime-age earnings run from 58,000 dollars in education and public service to 98,000 in STEM. Within STEM alone the range is 64,000 to 146,000, and several humanities majors beat the STEM 25th percentile.",
      "supports": "That the spread within a field is often wider than the gap between fields, which undercuts advice given at the level of STEM against humanities.",
      "doesNotSupport": "Anything about the future. It is a cross-sectional snapshot of people already employed.",
      "terms": [],
      "relatedPages": [
        "/research/what-should-i-tell-my-children-to-study"
      ]
    },
    {
      "id": "nyfed-recent-grads-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#nyfed-recent-grads-2026",
      "section": "work",
      "authors": "Federal Reserve Bank of New York",
      "year": 2026,
      "title": "The Labor Market for Recent College Graduates",
      "publication": "New York Fed, data through 2026 Q2",
      "url": "https://www.newyorkfed.org/research/college-labor-market",
      "grade": "institutional-survey",
      "method": "Current Population Survey and ACS tracking of unemployment and underemployment by major since 1990.",
      "finding": "As at the second quarter of 2026, unemployment among recent graduates runs around 5.6 per cent and underemployment around 42 per cent, the highest since 2020.",
      "supports": "That the graduate labour market has tightened measurably, and that underemployment is the larger number.",
      "doesNotSupport": "Attribution to AI. The series is descriptive and the Fed states it is not a forecast.",
      "terms": [],
      "relatedPages": [
        "/research/what-should-i-tell-my-children-to-study"
      ]
    },
    {
      "id": "gathmann-schonberg-2010",
      "citeAs": "https://thesuperskills.com/research/evidence#gathmann-schonberg-2010",
      "section": "work",
      "authors": "Gathmann, C. and Schoenberg, U.",
      "year": 2010,
      "title": "How General Is Human Capital? A Task-Based Approach",
      "publication": "Journal of Labor Economics, 28(1)",
      "url": "https://www.journals.uchicago.edu/doi/10.1086/649786",
      "grade": "peer-reviewed",
      "method": "German administrative employment panel, using task overlap between occupations to measure task-specific human capital.",
      "finding": "Task-specific human capital accounts for up to 52 per cent of overall wage growth. Workers move to occupations with similar task profiles, and the distance of those moves shrinks with experience.",
      "supports": "That skill is portable along task lines rather than being either fully general or locked to one job.",
      "doesNotSupport": "That broad general education transfers well. If anything it argues the opposite, which complicates the study-anything advice rather than supporting it.",
      "terms": [],
      "relatedPages": [
        "/research/what-should-i-tell-my-children-to-study"
      ]
    },
    {
      "id": "simons-2016",
      "citeAs": "https://thesuperskills.com/research/evidence#simons-2016",
      "section": "learning",
      "authors": "Simons, D. J., Boot, W. R., Charness, N., Gathercole, S. E., Chabris, C. F., Hambrick, D. Z. and Stine-Morrow, E. A. L.",
      "year": 2016,
      "title": "Do Brain-Training Programs Work?",
      "publication": "Psychological Science in the Public Interest, 17(3)",
      "url": "https://journals.sagepub.com/doi/10.1177/1529100616661983",
      "grade": "compiled-review",
      "method": "Systematic review applying pre-specified best-practice standards to every study cited by commercial brain-training companies as evidence of efficacy.",
      "finding": "Extensive evidence that training improves performance on the trained tasks, less evidence for closely related tasks, and little evidence that training improves distantly related tasks or everyday cognitive performance. No cited study met all best-practice standards.",
      "supports": "That near transfer is real and far transfer, which is what any claim to have trained attention requires, is essentially unsupported.",
      "doesNotSupport": "That practising a specific skill is useless. The failure is transfer, not learning.",
      "terms": [],
      "relatedPages": [
        "/research/is-attention-a-trainable-skill"
      ]
    },
    {
      "id": "verhaeghen-2021",
      "citeAs": "https://thesuperskills.com/research/evidence#verhaeghen-2021",
      "section": "learning",
      "authors": "Verhaeghen, P.",
      "year": 2021,
      "title": "Mindfulness as Attention Training: Meta-Analyses on the Links Between Attention Performance and Mindfulness Interventions, Long-Term Meditation Practice, and Trait Mindfulness",
      "publication": "Mindfulness, 12(3)",
      "url": "https://link.springer.com/article/10.1007/s12671-020-01532-1",
      "grade": "compiled-review",
      "method": "Three meta-analyses covering 109 effect sizes from 40 intervention studies, 59 effect sizes from 18 long-term meditator studies, and 197 effect sizes from 28 trait studies.",
      "finding": "Average effects were small to moderate, Hedges g of 0.29 for interventions and 0.32 for long-term practice, concentrated in inhibition and executive control rather than sustained attention.",
      "supports": "That something measurable happens, and that it is smaller and narrower than the popular claim.",
      "doesNotSupport": "That the effect survives comparison with an active control. The published breakdown does not separate active from passive controls, so demand effects cannot be excluded.",
      "terms": [],
      "relatedPages": [
        "/research/is-attention-a-trainable-skill"
      ]
    },
    {
      "id": "wiradhany-2017",
      "citeAs": "https://thesuperskills.com/research/evidence#wiradhany-2017",
      "section": "judgement",
      "authors": "Wiradhany, W. and Nieuwenstein, M. R.",
      "year": 2017,
      "title": "Cognitive Control in Media Multitaskers: Two Replication Studies and a Meta-Analysis",
      "publication": "Attention, Perception and Psychophysics, 79(8)",
      "url": "https://link.springer.com/article/10.3758/s13414-017-1408-4",
      "grade": "peer-reviewed",
      "method": "Two direct replications of Ophir, Nass and Wagner (2009), 14 tests at mean power 0.81, plus a meta-analysis of 39 effect sizes.",
      "finding": "Only five of 14 tests showed increased distractibility, and only two survived a Bayesian analysis. The meta-analytic association became non-significant after correcting for small-study effects. The authors question whether the association exists.",
      "supports": "That one of the most-cited findings about media multitasking and attention does not replicate.",
      "doesNotSupport": "That multitasking has no cost at all. This is one specific effect, distractor filtering, not the whole question.",
      "terms": [],
      "relatedPages": [
        "/research/is-attention-a-trainable-skill"
      ]
    },
    {
      "id": "leroy-2009",
      "citeAs": "https://thesuperskills.com/research/evidence#leroy-2009",
      "section": "judgement",
      "authors": "Leroy, S.",
      "year": 2009,
      "title": "Why Is It So Hard to Do My Work? The Challenge of Attention Residue When Switching Between Work Tasks",
      "publication": "Organizational Behavior and Human Decision Processes, 109(2)",
      "url": "https://www.sciencedirect.com/science/article/abs/pii/S0749597809000399",
      "grade": "peer-reviewed",
      "method": "Two laboratory experiments manipulating whether a task was completed or interrupted before switching.",
      "finding": "People struggle to move attention away from an unfinished task, and performance on the next task suffers. Time pressure on the first task helps disengagement.",
      "supports": "That the residue effect exists under controlled conditions, and that completion rather than willpower is what releases attention.",
      "doesNotSupport": "Real-world magnitude. Two laboratory experiments, not a field study.",
      "terms": [],
      "relatedPages": [
        "/research/is-attention-a-trainable-skill"
      ]
    },
    {
      "id": "heinz-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#heinz-2025",
      "section": "frontline",
      "authors": "Heinz, M. V., Mackin, D. M., Trudeau, B. M. et al.",
      "year": 2025,
      "title": "Randomized Trial of a Generative AI Chatbot for Mental Health Treatment",
      "publication": "NEJM AI, 2(4). DOI 10.1056/AIoa2400802",
      "url": "https://ai.nejm.org/doi/full/10.1056/AIoa2400802",
      "grade": "peer-reviewed",
      "method": "Randomised controlled trial, N=210 adults recruited by national advertising, with major depressive disorder, generalised anxiety disorder or clinically high risk for feeding and eating disorders. Four-week intervention with Therabot, an expert-fine-tuned generative chatbot built on a hand-curated CBT corpus, against a WAITLIST control, with four weeks of follow-up.",
      "finding": "Statistically significant symptom reductions across all three diagnostic groups at four and eight weeks. 95 per cent of participants engaged, averaging 260 messages and 6.18 hours over four weeks. Therapeutic alliance was comparable to outpatient psychotherapy (mean WAI 3.59). Expressions of suicidal ideation required staff intervention 15 times, and inappropriate responses required correction 13 times.",
      "supports": "That a purpose-built, expert-curated generative chatbot can produce measurable symptom improvement over four weeks, and that users form a working alliance with it.",
      "doesNotSupport": "That it is safe unsupervised, or that the effect is the chatbot rather than attention and expectation. The control was a waitlist rather than an active comparator, so the treatment effect cannot be separated from the effect of receiving something. The developer confirmed to the FDA advisory committee that humans reviewed all messages in near real time and conducted risk assessments where needed, so the safety record describes a supervised system rather than an autonomous one.",
      "terms": [],
      "relatedPages": [
        "/research/is-it-safe-to-use-ai-for-therapy"
      ]
    },
    {
      "id": "moore-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#moore-2025",
      "section": "frontline",
      "authors": "Moore, J., Grabb, D., Agnew, W., Klyman, K., Chancellor, S., Ong, D. C. and Haber, N.",
      "year": 2025,
      "title": "Expressing stigma and inappropriate responses prevents LLMs from safely replacing mental health providers",
      "publication": "Proceedings of the 2025 ACM Conference on Fairness, Accountability, and Transparency (FAccT). DOI 10.1145/3715275.3732039",
      "url": "https://arxiv.org/abs/2504.18412",
      "grade": "peer-reviewed",
      "method": "Mapping review of therapy guides used by major medical institutions to identify the requirements of a therapeutic relationship, followed by several experiments testing current models, including gpt-4o, against those requirements in naturalistic therapy settings.",
      "finding": "Models expressed stigma towards people with mental health conditions and responded inappropriately to common and critical presentations, including encouraging delusional thinking, which the authors attribute to sycophancy. The pattern persisted in larger and newer models.",
      "supports": "That the failures are structural rather than a matter of model size, and that current safety training does not remove them.",
      "doesNotSupport": "How often this occurs in real use, or what happens with purpose-built clinical systems rather than general-purpose models. These are constructed scenarios, not observed patient interactions.",
      "terms": [],
      "relatedPages": [
        "/research/is-it-safe-to-use-ai-for-therapy"
      ]
    },
    {
      "id": "fda-dhac-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#fda-dhac-2025",
      "section": "institutional",
      "authors": "United States Food and Drug Administration, Digital Health Advisory Committee",
      "year": 2025,
      "title": "Generative Artificial Intelligence-Enabled Digital Mental Health Medical Devices: meeting summary, 6 November 2025",
      "publication": "FDA Center for Devices and Radiological Health. Docket FDA-2025-N-2338",
      "url": "https://www.fda.gov/media/190450/download",
      "grade": "institutional-survey",
      "method": "Public advisory committee meeting with FDA presentations, 16 open public hearing speakers and structured committee deliberation on three scenarios: a prescription LLM therapy device for adults with major depressive disorder, over-the-counter and autonomous expansions, and use with under-21s.",
      "finding": "The director of CDRH stated that FDA has authorised more than 1,200 AI-enabled medical devices and none yet involve generative AI for mental health conditions. The committee asked for premarket comparators BEYOND waitlist controls, judged autonomous over-the-counter use for undiagnosed users substantially higher risk, described multi-condition autonomous use as the highest risk of all, and expressed strong discomfort with autonomous use in children and adolescents. One member noted that reminders that a system is not human cannot overcome automation bias.",
      "supports": "The regulatory position as at November 2025, in the regulator's own words, and that the committee independently identified the waitlist-control weakness in the existing trial evidence.",
      "doesNotSupport": "What FDA will decide. Advisory committee recommendations are non-binding, and no rule follows from this meeting.",
      "terms": [],
      "relatedPages": [
        "/research/is-it-safe-to-use-ai-for-therapy"
      ]
    },
    {
      "id": "illinois-wopr-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#illinois-wopr-2025",
      "section": "institutional",
      "authors": "State of Illinois",
      "year": 2025,
      "title": "Wellness and Oversight for Psychological Resources Act (HB1806), signed 1 August 2025",
      "publication": "Illinois Department of Financial and Professional Regulation",
      "url": "https://idfpr.illinois.gov/news/2025/gov-pritzker-signs-state-leg-prohibiting-ai-therapy-in-il.html",
      "grade": "institutional-survey",
      "method": "State legislation, passed almost unanimously and signed by the Governor.",
      "finding": "Prohibits the use of AI to provide therapy or perform therapeutic decision-making, including direct therapeutic communication with clients and detection of a client's emotional or mental state, while permitting administrative and supplementary use by licensed professionals. Penalties reach 10,000 dollars per violation.",
      "supports": "That at least one jurisdiction has moved from guidance to prohibition, and where it drew the line: the boundary is drawn at therapeutic decisions and direct therapeutic communication, not at the technology.",
      "doesNotSupport": "Anything about effectiveness, and nothing about other jurisdictions. One state, and its scope is contested.",
      "terms": [],
      "relatedPages": [
        "/research/is-it-safe-to-use-ai-for-therapy"
      ]
    },
    {
      "id": "frey-osborne-2013",
      "citeAs": "https://thesuperskills.com/research/evidence#frey-osborne-2013",
      "section": "work",
      "authors": "Frey, C. B. and Osborne, M. A.",
      "year": 2013,
      "title": "The Future of Employment: How Susceptible Are Jobs to Computerisation?",
      "publication": "Oxford Martin School working paper, 17 September 2013. Later published in Technological Forecasting and Social Change, 114 (2017)",
      "url": "https://oms-www.files.svdcdn.com/production/downloads/academic/The_Future_of_Employment.pdf",
      "grade": "working-paper",
      "method": "Probability of computerisation estimated for 702 detailed US occupations using a Gaussian process classifier, trained on 70 occupations hand-labelled by machine learning researchers at an Oxford workshop.",
      "finding": "In the authors' own words: 'about 47 percent of total US employment is at risk'. The paper's title asks how SUSCEPTIBLE jobs are, and the estimate is of technical susceptibility to computerisation, not a forecast of job losses.",
      "supports": "That a large share of US employment sits in occupations whose tasks were, in 2013, judged technically susceptible to computerisation.",
      "doesNotSupport": "That 47 per cent of jobs will be, or have been, lost. It is not a prediction, carries no date attached to any loss, and models whole occupations rather than tasks within them. The routine citation as '47 per cent of jobs will disappear' reverses what the paper claims. The working paper is 2013; the journal version is 2017, and the two dates are frequently confused.",
      "terms": [
        "labour market"
      ],
      "relatedPages": [
        "/research/the-most-quoted-ai-statistics-checked"
      ]
    },
    {
      "id": "arntz-2016",
      "citeAs": "https://thesuperskills.com/research/evidence#arntz-2016",
      "section": "institutional",
      "authors": "Arntz, M., Gregory, T. and Zierahn, U.",
      "year": 2016,
      "title": "The Risk of Automation for Jobs in OECD Countries: A Comparative Analysis",
      "publication": "OECD Social, Employment and Migration Working Papers No. 189",
      "url": "https://www.oecd.org/en/publications/the-risk-of-automation-for-jobs-in-oecd-countries_5jlz9h56dvq7-en.html",
      "grade": "institutional-modelling",
      "method": "Task-based re-estimation across 21 OECD countries using PIAAC survey data, accounting for the heterogeneity of tasks WITHIN occupations rather than treating whole occupations as automatable.",
      "finding": "9 per cent of jobs automatable on average across 21 countries, ranging from 6 per cent in Korea to 12 per cent in Austria. The authors state the occupation-based approach 'might lead to an overestimation of job automatibility, as occupations labelled as high-risk occupations often still contain a substantial share of tasks that are hard to automate'.",
      "supports": "That the headline automation figure is highly sensitive to whether you model occupations or tasks, and that the difference is roughly fivefold on the same question.",
      "doesNotSupport": "That 9 per cent is correct and 47 per cent wrong. Both are model outputs resting on assumptions, and neither has been scored against what happened.",
      "terms": [
        "labour market"
      ],
      "relatedPages": [
        "/research/the-most-quoted-ai-statistics-checked"
      ]
    },
    {
      "id": "hughes-2011",
      "citeAs": "https://thesuperskills.com/research/evidence#hughes-2011",
      "section": "institutional",
      "authors": "Hughes, M.",
      "year": 2011,
      "title": "Do 70 Per Cent of All Organizational Change Initiatives Really Fail?",
      "publication": "Journal of Change Management, 11(4), 451-464. DOI 10.1080/14697017.2011.630506",
      "url": "https://research.brighton.ac.uk/en/publications/do-70-per-cent-of-all-organizational-change-initiatives-really-fa/",
      "grade": "peer-reviewed",
      "method": "Critical review of five separate published instances of the 70 per cent organisational change failure rate, tracing each to its stated source.",
      "finding": "In the author's words: 'whilst the existence of a popular narrative of 70 percent organizational change failure is acknowledged, there is no valid and reliable empirical evidence to support such a narrative.'",
      "supports": "That one of the most repeated statistics in management has no traceable empirical basis, and that this was established in a peer-reviewed journal fifteen years ago and ignored.",
      "doesNotSupport": "That change programmes usually succeed. The finding is about the absence of evidence for a specific number, not about the true rate, which remains unmeasured.",
      "terms": [],
      "relatedPages": [
        "/research/the-most-quoted-ai-statistics-checked"
      ]
    },
    {
      "id": "acemoglu-restrepo-2019",
      "citeAs": "https://thesuperskills.com/research/evidence#acemoglu-restrepo-2019",
      "section": "work",
      "authors": "Acemoglu, D. and Restrepo, P.",
      "year": 2019,
      "title": "Automation and New Tasks: How Technology Displaces and Reinstates Labor",
      "publication": "Journal of Economic Perspectives, 33(2), 3-30. DOI 10.1257/jep.33.2.3",
      "url": "https://www.aeaweb.org/articles?id=10.1257/jep.33.2.3",
      "grade": "peer-reviewed",
      "method": "Task-based theoretical framework in which production is allocated between capital and labour, with an empirical decomposition of US industry-level data over recent decades.",
      "finding": "Automation shifts the task content of production against labour through a displacement effect, and therefore ALWAYS reduces the labour share in value added, and may reduce labour demand even while raising productivity. The counterweight is the creation of new tasks in which labour has a comparative advantage, which always raises the labour share. Their decomposition attributes slower employment growth over three decades to an accelerating displacement effect, a weaker reinstatement effect and slower productivity growth.",
      "supports": "That the distributional question is separate from the productivity question, and that a technology can raise output while reducing labour's share of it. This is the framework almost every serious argument in this area now runs through.",
      "doesNotSupport": "That AI specifically will behave this way. The empirical work predates generative AI and concerns industrial automation and robotics.",
      "terms": [
        "labour market"
      ],
      "relatedPages": [
        "/research/who-captures-the-productivity-gains-from-ai"
      ]
    },
    {
      "id": "lang-masai-2023",
      "citeAs": "https://thesuperskills.com/research/evidence#lang-masai-2023",
      "section": "frontline",
      "authors": "Lang, K., Josefsson, V., Larsson, A.-M. et al.",
      "year": 2023,
      "title": "Artificial intelligence-supported screen reading versus standard double reading in the Mammography Screening with Artificial Intelligence trial (MASAI): a clinical safety analysis",
      "publication": "The Lancet Oncology, 24(8), 936-944. DOI 10.1016/S1470-2045(23)00298-X",
      "url": "https://www.thelancet.com/journals/lanonc/article/PIIS1470-2045(23)00298-X/fulltext",
      "grade": "peer-reviewed",
      "method": "Planned interim safety analysis of a randomised, controlled, non-inferiority, single-blinded screening accuracy trial. 80,033 women aged 40-80 screened at four sites in southwest Sweden between April 2021 and July 2022, randomised 1:1 to AI-supported screen reading or standard double reading by two radiologists.",
      "finding": "Cancer detection was six per 1,000 screened women with AI support against five per 1,000 with standard double reading, 41 more cancers detected. The false-positive rate was 1.5 per cent in both arms. Screen readings fell from 83,231 in the control arm to 46,345 in the AI arm, a 44 per cent reduction in screen-reading workload.",
      "supports": "That a triage-plus-detection-support workflow with a radiologist retaining the recall decision can hold detection while roughly halving reading volume, in a randomised population-based programme.",
      "doesNotSupport": "Patient benefit, which the interim analysis was not designed to test. It is also one mammography device, one AI system, one country, and moderately to highly experienced readers, which the authors state as limits on generalisability.",
      "terms": [
        "human-AI collaboration",
        "clinical AI"
      ],
      "relatedPages": [
        "/research/how-will-ai-change-medicine"
      ]
    },
    {
      "id": "masai-final-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#masai-final-2026",
      "section": "frontline",
      "authors": "Gommers, J., Lang, K., Hofvind, S. et al.",
      "year": 2026,
      "title": "Interval cancer, sensitivity, and specificity comparing AI-supported mammography screening with standard double reading without AI in the MASAI study",
      "publication": "The Lancet. DOI 10.1016/S0140-6736(25)02464-X. Published 29 January 2026",
      "url": "https://www.thelancet.com/journals/lancet/article/PIIS0140-6736(25)02464-X/fulltext",
      "grade": "peer-reviewed",
      "method": "Full results of a randomised, controlled, non-inferiority, single-blinded, population-based screening-accuracy trial. Over 100,000 women screened at four Swedish sites between April 2021 and December 2022, with two years of follow-up.",
      "finding": "Interval cancers fell from 1.76 per 1,000 women (93/52,872) in the control arm to 1.55 per 1,000 (82/53,043) in the AI arm, a 12 per cent reduction. There were 16 per cent fewer invasive (75 v 89), 21 per cent fewer large (38 v 48) and 27 per cent fewer aggressive-subtype (43 v 59) interval cancers. Cancers detected at screening rose from 74 per cent (262/355) to 81 per cent (338/420) of all cases. False positives were 1.5 per cent in the intervention arm and 1.4 per cent in the control arm.",
      "supports": "That AI-supported screen reading, with a radiologist retaining the recall decision, reduced interval cancers over two years in a randomised population-based programme. The only outcome-level randomised evidence of this kind in clinical AI.",
      "doesNotSupport": "Generalisation beyond one country, one mammography device, one AI system and experienced readers, all stated as limitations by the authors. Mortality was not an endpoint, and cost-effectiveness was not assessed. The first author's own summary states the design does not support replacing radiologists, since at least one still reads every case.",
      "terms": [
        "human-AI collaboration",
        "clinical AI"
      ],
      "relatedPages": [
        "/research/how-will-ai-change-medicine"
      ]
    },
    {
      "id": "wong-2021",
      "citeAs": "https://thesuperskills.com/research/evidence#wong-2021",
      "section": "frontline",
      "authors": "Wong, A., Otles, E., Donnelly, J. P. et al.",
      "year": 2021,
      "title": "External Validation of a Widely Implemented Proprietary Sepsis Prediction Model in Hospitalized Patients",
      "publication": "JAMA Internal Medicine, 181(8), 1065-1070. DOI 10.1001/jamainternmed.2021.2626",
      "url": "https://jamanetwork.com/journals/jamainternalmedicine/fullarticle/2781307",
      "grade": "peer-reviewed",
      "method": "Retrospective external validation cohort study. 27,697 patients aged 18 or over across 38,455 hospitalisations at Michigan Medicine, 6 December 2018 to 20 October 2019. Sepsis occurred in 7 per cent of hospitalisations.",
      "finding": "The Epic Sepsis Model achieved a hospitalisation-level area under the curve of 0.63 (95% CI 0.62-0.64), against the 0.76-0.83 cited by its developer. At the alerting threshold in clinical use, sensitivity was 33 per cent, specificity 83 per cent, positive predictive value 12 per cent. It did not identify 1,709 of 2,552 septic hospitalisations (67 per cent), 60 per cent of whom received timely antibiotics anyway, while crossing the alert threshold in 18 per cent of all hospitalisations (6,971 of 38,455), requiring eight patients to be evaluated per case of sepsis found.",
      "supports": "That a proprietary clinical prediction model deployed at national scale can perform far below its developer's stated figures when validated independently, and that nobody had checked. The authors' own conclusion is that widespread adoption despite poor performance raises fundamental concerns about sepsis management nationally.",
      "doesNotSupport": "That all clinical prediction models fail, or that this model performs identically elsewhere. It is one model at one academic health system, and the authors note the theoretical alert burden does not account for real-world trigger criteria and lockouts.",
      "terms": [
        "clinical AI",
        "automation bias",
        "alert fatigue"
      ],
      "relatedPages": [
        "/research/how-will-ai-change-medicine"
      ]
    },
    {
      "id": "goh-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#goh-2024",
      "section": "frontline",
      "authors": "Goh, E., Gallo, R., Hom, J. et al.",
      "year": 2024,
      "title": "Large Language Model Influence on Diagnostic Reasoning: A Randomized Clinical Trial",
      "publication": "JAMA Network Open, 7(10), e2440969. DOI 10.1001/jamanetworkopen.2024.40969",
      "url": "https://jamanetwork.com/journals/jamanetworkopen/fullarticle/2825395",
      "grade": "peer-reviewed",
      "method": "Single-blind randomised clinical trial, 29 November to 29 December 2023. 50 US-licensed physicians (26 attendings, 24 residents) in family medicine, internal medicine or emergency medicine, randomised to GPT-4 or to conventional resources, working through up to six clinical vignettes, graded blind against a validated diagnostic reasoning rubric. 244 cases completed.",
      "finding": "Median diagnostic reasoning score per case was 76 per cent (IQR 66-87) with the LLM and 74 per cent (IQR 63-84) with conventional resources: an adjusted difference of 2 percentage points (95% CI -4 to 8, p=0.60). Median time per case was 519 seconds against 565, adjusted difference -82 seconds (95% CI -195 to 31, p=0.20). The LLM alone scored a median 92 per cent, 16 percentage points above the conventional-resources group (95% CI 2-30, p=0.03).",
      "supports": "That adding a capable model to a physician did not, in this trial, improve diagnostic reasoning or save time, while the same model working alone outperformed both groups of clinicians. The performance of a human-plus-model pairing cannot be inferred from the performance of either part.",
      "doesNotSupport": "That LLMs should diagnose autonomously, which the authors explicitly reject. Six curated vignettes exclude history-taking, examination, context and time, which is most of clinical reasoning. Participants received no prompt-engineering training, and the authors offer prompting and interaction design as explanations.",
      "terms": [
        "human-AI collaboration",
        "clinical AI",
        "verification"
      ],
      "relatedPages": [
        "/research/how-will-ai-change-medicine",
        "/research/which-professions-face-the-greatest-deskilling-risk"
      ]
    },
    {
      "id": "lukac-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#lukac-2025",
      "section": "frontline",
      "authors": "Lukac, P. J., Turner, W., Vangala, S., Chin, A. T., Khalili, J., Shih, Y.-C. T., Sarkisian, C., Cheng, E. M. and Mafi, J. N.",
      "year": 2025,
      "title": "Ambient AI Scribes in Clinical Practice: A Randomized Trial",
      "publication": "NEJM AI, 2(12). DOI 10.1056/AIoa2501000. Preprint: medRxiv 2025.07.10.25331333",
      "url": "https://ai.nejm.org/doi/abs/10.1056/AIoa2501000",
      "grade": "peer-reviewed",
      "method": "Parallel three-arm pragmatic randomised clinical trial at one US health system. 238 outpatient physicians across 14 specialties randomised 1:1:1 by covariate-constrained randomisation to Microsoft DAX, Nabla or usual care, 4 November 2024 to 3 January 2025, with the second intervention month compared to baseline.",
      "finding": "Time writing a note fell by an estimated 18 seconds in the control arm, 23 seconds in the DAX arm and 41 seconds in the Nabla arm. Only Nabla differed significantly from control (-9.5 per cent, 95% CI -17.2 to -1.8, p=0.02); DAX did not (-1.7 per cent, 95% CI -9.4 to +5.9, p=0.66). Scribe users improved on Mini-Z burnout (+2.76, p<0.001), task load (-35.8, p=0.01) and work exhaustion (-0.27, p=0.01), with no significant difference between the two products on any psychometric. Roughly 15 per cent of physicians given a tool never used it.",
      "supports": "That ambient documentation produces a small, product-dependent time saving and a more consistent wellbeing effect, measured against a randomised control rather than against a before-and-after.",
      "doesNotSupport": "A general time saving for ambient AI. One health system, majority female sample, a two-month contract-limited window, and the authors flag that the electronic record's own time metrics do not count editing done inside the scribe platform, so reported savings may be overstated. They note this limitation affects all studies using those metrics.",
      "terms": [
        "productivity",
        "clinical AI"
      ],
      "relatedPages": [
        "/research/how-will-ai-change-medicine"
      ]
    },
    {
      "id": "dellacqua-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#dellacqua-2025",
      "section": "collaboration",
      "authors": "Dell'Acqua, F., Ayoubi, C., Lifshitz, H., Sadun, R., Mollick, E., Mollick, L., Han, Y., Goldman, J., Nair, H., Taub, S. and Lakhani, K.",
      "year": 2025,
      "title": "The Cybernetic Teammate: A Field Experiment on Generative AI Reshaping Teamwork and Expertise",
      "publication": "NBER Working Paper 33641, April 2025. Published as 'The Cybernetic Teammate: A Field Experiment on Generative AI and Teamwork', Organization Science, June 2026. DOI 10.1287/orsc.2025.20702",
      "url": "https://www.nber.org/papers/w33641",
      "grade": "peer-reviewed",
      "method": "Pre-registered field experiment with 776 professionals at Procter and Gamble working on real product innovation challenges, randomised both on AI access and on working individually or in a two-person new product development team.",
      "finding": "Individuals working with AI matched the performance of two-person teams working without it. AI use removed the functional split in proposals: without AI, research and development professionals proposed more technical solutions and commercial professionals more commercially oriented ones, while professionals using AI produced balanced solutions regardless of background. Participants using AI reported more positive emotional responses.",
      "supports": "That a model can substitute for measurable parts of what a second human teammate contributes, including some of the social and motivational function, and that it flattens the differences in output that come from professional training.",
      "doesNotSupport": "Client or business outcomes, which were not measured, or any effect over time. One firm, one task type, a single session. Procter and Gamble provided financial support to the institute involved and one author had consulted for the firm, both disclosed in the paper. That the flattened output is better rather than merely more balanced is not established.",
      "terms": [
        "human-AI collaboration",
        "expertise",
        "productivity"
      ],
      "relatedPages": [
        "/research/how-will-ai-change-consulting"
      ]
    },
    {
      "id": "magesh-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#magesh-2025",
      "section": "professions",
      "authors": "Magesh, V., Surani, F., Dahl, M., Suzgun, M., Manning, C. D. and Ho, D. E.",
      "year": 2025,
      "title": "Hallucination-Free? Assessing the Reliability of Leading AI Legal Research Tools",
      "publication": "Journal of Empirical Legal Studies, 22(2), 216-242. DOI 10.1111/jels.12413",
      "url": "https://onlinelibrary.wiley.com/doi/full/10.1111/jels.12413",
      "grade": "peer-reviewed",
      "method": "First preregistered empirical evaluation of retrieval-augmented legal research tools. Over 200 handwritten legal queries across four categories, preregistered with the Open Science Foundation in March 2024, run against Lexis+ AI, Westlaw AI-Assisted Research, Ask Practical Law AI and GPT-4, graded on whether responses were correct and grounded in the sources cited.",
      "finding": "The three commercial legal tools each hallucinated between 17 and 33 per cent of the time. Lexis+ AI was accurate on 65 per cent of queries and incomplete on 18 per cent; Westlaw AI-Assisted Research was accurate 42 per cent of the time with a hallucination in one-third of responses; Ask Practical Law AI was incomplete on 62 per cent of queries. Westlaw produced the longest answers, averaging 350 words against 219 and 175, which the authors link to both its higher hallucination rate and the verification burden it imposes.",
      "supports": "That retrieval-augmented generation reduces hallucination relative to a general chatbot without eliminating it, and that provider claims of hallucination-free citation were overstated. Longer generated answers carry more falsifiable propositions and require more checking.",
      "doesNotSupport": "Current performance of any named product. These are specific versions tested in 2024 and providers update continuously. It also does not measure legal outcomes, only response accuracy against expert grading.",
      "terms": [
        "hallucination",
        "verification",
        "legal profession"
      ],
      "relatedPages": [
        "/research/how-will-ai-change-law"
      ]
    },
    {
      "id": "charlotin-hallucination-db",
      "citeAs": "https://thesuperskills.com/research/evidence#charlotin-hallucination-db",
      "section": "professions",
      "authors": "Charlotin, D.",
      "year": 2026,
      "title": "AI Hallucination Cases database",
      "publication": "damiencharlotin.com, updated daily. Figures read from the update of 27 August 2026",
      "url": "https://www.damiencharlotin.com/hallucinations/",
      "grade": "compiled-review",
      "method": "Curated database of legal decisions worldwide in which a court or tribunal has explicitly found or implied that a party relied on hallucinated material. Excludes mere allegations, with a small stated exception. Coverage begins in the second quarter of 2023.",
      "finding": "1,963 cases identified as at 27 August 2026. By jurisdiction: United States 1,345, Canada 214, Australia 98, United Kingdom 62, Israel 57, with more than thirty other countries represented. By party responsible: self-represented litigants 1,127, lawyers 784, judges 29, expert witnesses 15. By nature: fabricated material 1,634, misrepresented authority 816, false quotations 528, outdated advice 33.",
      "supports": "That fabricated legal authority reaching courts is a documented, dated, worldwide phenomenon rather than an anecdote, and that self-represented litigants account for more recorded instances than lawyers do.",
      "doesNotSupport": "The true rate. The database counts decisions where a court addressed the point, so instances nobody noticed, or resolved without a written decision, are invisible by construction; its author states this. It is a curated compilation rather than a sampled study, so it cannot support a denominator or a trend rate.",
      "terms": [
        "hallucination",
        "legal profession",
        "verification"
      ],
      "relatedPages": [
        "/research/how-will-ai-change-law"
      ]
    },
    {
      "id": "ayinde-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#ayinde-2025",
      "section": "professions",
      "authors": "Divisional Court of England and Wales (Dame Victoria Sharp P and Johnson J)",
      "year": 2025,
      "title": "Ayinde v London Borough of Haringey and Al-Haroun v Qatar National Bank",
      "publication": "[2025] EWHC 1383 (Admin), judgment of 6 June 2025",
      "url": "https://www.judiciary.uk/judgments/ayinde-v-london-borough-of-haringey-and-al-haroun-v-qatar-national-bank/",
      "grade": "compiled-review",
      "method": "Two referrals under the Hamid jurisdiction, heard together, arising from fabricated citations placed before the court. In Ayinde, five cited authorities did not exist. In Al-Haroun, a claim for £89.4 million, eighteen of forty-five cited authorities did not exist.",
      "finding": "At [6] the court held that freely available generative AI tools trained on a large language model are not capable of conducting reliable legal research. At [7] those using them have a professional duty to check accuracy against authoritative sources, which the court names. At [8] the duty extends to lawyers relying on others' AI-assisted work. At [81] a lawyer is not entitled to rely on their lay client for the accuracy of citations. At [23] the available powers run from public admonition and wasted costs to contempt and referral to the police, and at [31] admonishment alone is unlikely to suffice save in exceptional circumstances. At [9] leadership responsibility falls on heads of chambers and managing partners, and the court states it will inquire in future hearings whether that responsibility was fulfilled.",
      "supports": "That in England and Wales the verification duty is settled, non-delegable and extends upward to those who supervise. It also establishes that the court will treat a fabricated citation as a possible supervision failure rather than only as an individual one.",
      "doesNotSupport": "Anything about jurisdictions other than England and Wales, or about a lawyer who checked competently and was still misled, which no reported case has yet tested. It is a judgment rather than an empirical study, and it measures nothing.",
      "terms": [
        "verification",
        "legal profession",
        "accountability"
      ],
      "relatedPages": [
        "/research/how-will-ai-change-law"
      ]
    },
    {
      "id": "hohenstein-2023",
      "citeAs": "https://thesuperskills.com/research/evidence#hohenstein-2023",
      "section": "humanness",
      "authors": "Hohenstein, J., Kizilcec, R. F., DiFranzo, D., Aghajari, Z., Mieczkowski, H., Levy, K., Naaman, M., Hancock, J. and Jung, M. F.",
      "year": 2023,
      "title": "Artificial intelligence in communication impacts language and social relationships",
      "publication": "Scientific Reports, 13, 5487. DOI 10.1038/s41598-023-30938-9",
      "url": "https://www.nature.com/articles/s41598-023-30938-9",
      "grade": "peer-reviewed",
      "method": "Two randomised experiments on algorithmic response suggestions (smart replies) in live text chat. Study 1: 438 Mechanical Turk crowdworkers in 219 pairs discussing a policy question, smart-reply availability randomised separately for each partner, analysed with an instrumental-variable design. Study 2: 582 crowdworkers in 291 pairs, with the sentiment of the suggested replies manipulated. Preregistered (AsPredicted #40389).",
      "finding": "Smart replies accounted for 14.3 per cent of messages and produced 10.2 per cent more messages per minute. Greater ACTUAL use by a partner improved the other person's rating of their cooperation (b=15.66, p=0.018) and felt affiliation towards them (b=21.79, p=0.007), with no effect on dominance. Greater SUSPECTED use had the opposite sign: the more a participant believed their partner used smart replies, the less cooperative (p<0.0001) and less affiliative (p<0.0001) they rated them, after controlling for actual use. Suspicion tracked actual use only weakly (Pearson's r=0.22).",
      "supports": "That the interpersonal penalty for AI-assisted messaging attaches to being suspected rather than to using it, and that suspicion is a poor detector. Actual use made people seem warmer, not colder.",
      "doesNotSupport": "Anything longitudinal, which the authors state directly. The suspicion result is correlational and they say it does not show causally how attitudes shift in response to actual use. Participants were crowdworkers discussing policy with strangers, not intimates. A publisher correction (10.1038/s41598-023-43601-0, 3 October 2023) added an omitted funding statement and changed no result.",
      "terms": [
        "AI-mediated communication",
        "authenticity",
        "disclosure"
      ],
      "relatedPages": [
        "/research/should-i-use-ai-to-write-personal-messages",
        "/research/outsourced-recognition"
      ]
    },
    {
      "id": "melumad-yun-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#melumad-yun-2025",
      "section": "judgement",
      "authors": "Melumad, S. and Yun, J. H.",
      "year": 2025,
      "title": "Experimental evidence of the effects of large language models versus web search on depth of learning",
      "publication": "PNAS Nexus, 4(10), pgaf316. DOI 10.1093/pnasnexus/pgaf316",
      "url": "https://academic.oup.com/pnasnexus/article/4/10/pgaf316/8303888",
      "grade": "peer-reviewed",
      "method": "Seven online and laboratory experiments, four in the paper and three in the supplement. Participants learned a practical topic using either a large language model or web search links, then wrote advice for a friend. Experiment 1: 1,104 participants, real ChatGPT versus real Google. Experiment 2: 1,979 participants, simulated search holding the underlying FACTS identical across conditions. Experiment 3: 250 lab participants, standard Google versus Google with AI Overviews. Advice scored for length, named entities and pairwise similarity.",
      "finding": "Experiment 2, with facts held constant: time engaging with results 83.65 seconds with the summary against 124.32 with links; learned new things 3.71 against 3.96; ownership of knowledge 3.41 against 3.66; thought and effort in advice 3.85 against 4.11; advice 64.49 words against 74.22; references to facts 4.00 against 4.61; pairwise cosine similarity between participants' advice 0.224 against 0.072. Rated comprehensiveness did not differ (4.30 against 4.25, p=0.264). A supplementary condition adding real-time web links to the AI summary did not remove the effect, because only 26 per cent of participants clicked any link.",
      "supports": "That the format of a search result, independent of its content, changes how much effort people invest, how deeply they report learning, and the specificity and distinctiveness of what they can then produce.",
      "doesNotSupport": "That knowledge objectively declined. Depth of learning is self-reported throughout and no experiment included a recall or comprehension test; time on task is described by the authors as a proxy for effort. Topics were practical how-to tasks over short horizons. The paper states its own total sample twice and inconsistently, as 10,462 in the abstract and 10,426 in the introduction.",
      "terms": [
        "cognitive offloading",
        "desirable difficulty",
        "summarisation"
      ],
      "relatedPages": [
        "/research/should-i-let-ai-summarise-everything-i-read"
      ]
    },
    {
      "id": "fisher-2015",
      "citeAs": "https://thesuperskills.com/research/evidence#fisher-2015",
      "section": "judgement",
      "authors": "Fisher, M., Goddu, M. K. and Keil, F. C.",
      "year": 2015,
      "title": "Searching for explanations: How the Internet inflates estimates of internal knowledge",
      "publication": "Journal of Experimental Psychology: General, 144(3), 674-687. DOI 10.1037/xge0000070",
      "url": "https://bpb-us-w2.wpmucdn.com/campuspress.yale.edu/dist/c/259/files/2015/03/pdf-16ueczx.pdf",
      "grade": "peer-reviewed",
      "method": "Nine between-subjects experiments, 1,708 US participants via Amazon Mechanical Turk. An induction phase in which participants either searched the internet for explanations or were told not to, followed by self-ratings of their ability to explain questions in six domains unrelated to the induction material.",
      "finding": "Searching inflated self-rated explanatory ability, with Cohen's d from 0.35 to 0.63 across studies, and the effect appeared across all six unrelated domains. It persisted when the search returned no answer to the question asked (Experiment 4b: 4.11 against 4.00 for those who found an answer, both far above a 3.05 no-search baseline) and when it returned no results at all (Experiment 4c). It disappeared for autobiographical topics where the internet would not help (Experiment 3, p=0.30).",
      "supports": "That access to information is mistaken for knowledge held internally, that the illusion is specific to searchable domains rather than general overconfidence, and that it does not require the search to succeed.",
      "doesNotSupport": "That actual explanatory ability changes. Every dependent measure is a self-rating and no experiment tested real knowledge. The authors also note participants were presumably heavier internet users than average.",
      "terms": [
        "the Google effect",
        "cognitive offloading",
        "metacognition"
      ],
      "relatedPages": [
        "/research/should-i-let-ai-summarise-everything-i-read",
        "/research/do-i-still-need-to-remember-things"
      ]
    },
    {
      "id": "storm-stone-2015",
      "citeAs": "https://thesuperskills.com/research/evidence#storm-stone-2015",
      "section": "learning",
      "authors": "Storm, B. C. and Stone, S. M.",
      "year": 2015,
      "title": "Saving-Enhanced Memory: The Benefits of Saving on the Learning and Remembering of New Information",
      "publication": "Psychological Science, 26(2), 182-188. DOI 10.1177/0956797614559285",
      "url": "https://journals.sagepub.com/doi/10.1177/0956797614559285",
      "grade": "peer-reviewed",
      "method": "Three experiments on the effect of saving a digital file on memory for subsequently studied material. GRADED FROM THE PUBLISHED ABSTRACT ONLY: the full text is paywalled and could not be read at the primary source, so no sample sizes or statistics are recorded here.",
      "finding": "Saving one file before studying a new file significantly improved memory for the contents of the new file. The effect was not observed when the saving process was deemed unreliable, or when the contents of the to-be-saved file were not substantial enough to interfere with memory for the new file.",
      "supports": "That cognitive offloading can improve rather than degrade subsequent memory, and that the benefit depends on the external store being trusted.",
      "doesNotSupport": "Anything quantified. This entry rests on the abstract alone, which is a weaker basis than every other entry in this base and is stated as such. It also predates generative AI and concerns file saving rather than a system that can fabricate its own contents.",
      "terms": [
        "cognitive offloading",
        "memory"
      ],
      "relatedPages": [
        "/research/do-i-still-need-to-remember-things"
      ]
    },
    {
      "id": "commonsense-teens-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#commonsense-teens-2025",
      "section": "institutional",
      "authors": "Robb, M. B. and Mann, S. (Common Sense Media)",
      "year": 2025,
      "title": "Talk, Trust, and Trade-Offs: How and Why Teens Use AI Companions",
      "publication": "Common Sense Media, San Francisco, 16 July 2025. Fieldwork by NORC at the University of Chicago",
      "url": "https://www.commonsensemedia.org/research/talk-trust-and-trade-offs-how-and-why-teens-use-ai-companions",
      "grade": "institutional-survey",
      "method": "Survey of 1,060 US teens aged 13-17, interviewed 30 April to 14 May 2025, combining 719 probability interviews from NORC's AmeriSpeak Teen panel with 341 nonprobability interviews from Prodege, raked to February 2024 Current Population Survey totals. Margin of error plus or minus 4.2 percentage points. Cumulative response rate for the probability component 10.3 per cent.",
      "finding": "72 per cent have used an AI companion at least once and 52 per cent at least a few times a month. Against that: 80 per cent of users spend more time with real friends (68 per cent much more) and 6 per cent more time with AI; 67 per cent find AI conversations less satisfying than human ones; 50 per cent distrust the advice; 74 per cent have never shared personal information; 66 per cent have never felt uncomfortable; 9 per cent regard an AI as a friend or best friend; 46 per cent describe them as tools or programs.",
      "supports": "That conversational AI use is near-universal among US teenagers and that most of it is pragmatic rather than relational, on the report's own reading.",
      "doesNotSupport": "That 72 per cent use companion products. The definition given to respondents explicitly included using ChatGPT or Claude as companions, and the report's own limitations concede respondents may have conflated general AI use with companion use, potentially inflating usage statistics. It is cross-sectional and supports no causal claim, though the press release makes one. The published toplines and the report body also disagree on the social-skills transfer figure, giving 33 per cent and 39 per cent respectively.",
      "terms": [
        "teenagers",
        "AI companions",
        "adolescents"
      ],
      "relatedPages": [
        "/research/how-much-should-teenagers-use-ai"
      ]
    },
    {
      "id": "internetmatters-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#internetmatters-2025",
      "section": "institutional",
      "authors": "Internet Matters",
      "year": 2025,
      "title": "Me, myself and AI: Understanding and safeguarding children's use of AI chatbots",
      "publication": "Internet Matters, July 2025",
      "url": "https://www.internetmatters.org/wp-content/uploads/2025/07/Me-Myself-AI-Report.pdf",
      "grade": "institutional-survey",
      "method": "Mixed methods, March to July 2025. Survey of a representative sample of 1,000 UK children aged 9-17 and 2,000 parents of children aged 3-17, fielded April to May 2025; four focus groups with 27 children aged 13-17; 17 days of user testing across ChatGPT, Snapchat My AI and character.ai using two fictional child profiles; four expert interviews. Children are classified as vulnerable if they have an Education, Health and Care Plan, receive SEN support, or have a physical or mental health condition requiring professional help.",
      "finding": "64 per cent of children aged 9-17 have used an AI chatbot (ChatGPT 43 per cent, Google Gemini 32 per cent, Snapchat My AI 31 per cent). Among users the reasons are schoolwork 42 per cent, information 40 per cent, curiosity 40 per cent, chatting 24 per cent, advice 23 per cent, fun 18 per cent, wanting a friend 6 per cent, emotional help or therapy 3 per cent. 35 per cent say it feels like talking to a friend and 12 per cent say they use one because they have no one else to speak to, rising to 50 per cent and 23 per cent among vulnerable children, who are also nearly three times as likely to use companion-style products (17 against 6 per cent).",
      "supports": "That UK chatbot use among children is majority behaviour, dominated by schoolwork and information, and that companionship use concentrates in an identified vulnerable minority.",
      "doesNotSupport": "The precision of the vulnerable-group figures. Two charts are described as sharing the same base, children who have used at least one chatbot, but one uses 133 and 499 respondents and the other 188 and 802; the report does not flag this or publish significance testing, and its limitations section addresses the user testing only. The press release also substitutes a 42 per cent all-ages figure for the report's 47 per cent figure for 15-17 year olds. Nothing here is causal.",
      "terms": [
        "teenagers",
        "children",
        "AI companions"
      ],
      "relatedPages": [
        "/research/how-much-should-teenagers-use-ai"
      ]
    },
    {
      "id": "rohde-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#rohde-2026",
      "section": "work",
      "authors": "Rohde, W. (AiSuNe Foundation)",
      "year": 2026,
      "title": "Short-Term Gain, Long-Term Fragility: AI Labor Substitution and the Erosion of Sustainable Capability",
      "publication": "SSRN abstract 6577818, written 20 April 2026, revised 27 April 2026; also arXiv 2605.27399",
      "url": "https://papers.ssrn.com/sol3/papers.cfm?abstract_id=6577818",
      "grade": "working-paper",
      "method": "Sole-authored conceptual synthesis, 19 pages, no new empirical data. The author's own words: \"This paper is a conceptual synthesis rather than a new empirical study\" and \"the evidentiary strategy is selective and scoped\". Preprint, not peer reviewed, no journal reference. Dates verified at the SSRN record on 28 August 2026; the arXiv abstract page could not be fetched, so the arXiv submission date remains unconfirmed and is not asserted.",
      "finding": "Develops a mechanism of capability masking followed by capability erosion: AI output creates a persuasive appearance that organisational capability has been replaced while dependence on skilled human labour remains, supporting hiring restraint and deferred structural reform while costs accumulate. Frames the result as a stack of deferred obligations: technical debt in artifacts and systems, capability debt in the human layer that maintains them, and institutional debt in the wider structures that reproduce skill and resilience.",
      "supports": "That the capability-erosion argument has been reached independently, from software engineering and political economy rather than from organisational research, and that the masking-before-erosion sequence is not unique to this estate's account.",
      "doesNotSupport": "Anything measured. It is a preprint that argues from other people's empirical work rather than presenting its own, and its societal-scale claims about fragility and concentration of power are inference rather than finding. It is also the reason no claim of first use is made for the term capability debt on this site. Checked at source on 28 August 2026: Rohde does not claim to have coined capability debt either. The paper contains no claiming language for it, and what he does claim is a mechanism, \"to identify and formalize a mechanism of capability masking and capability erosion\".",
      "terms": [
        "capability debt",
        "deskilling",
        "labour substitution"
      ],
      "relatedPages": [
        "/research/capability-debt",
        "/research/cognitive-debt-and-capability-debt"
      ]
    },
    {
      "id": "metr-2026-update",
      "citeAs": "https://thesuperskills.com/research/evidence#metr-2026-update",
      "section": "collaboration",
      "authors": "Becker, J., Rush, N., Cunningham, T., Rein, D. and Mahamud, K. (METR)",
      "year": 2026,
      "title": "We are Changing our Developer Productivity Experiment Design",
      "publication": "METR, 24 February 2026",
      "url": "https://metr.org/blog/2026-02-24-uplift-update/",
      "grade": "working-paper",
      "method": "Second randomised task-level study, begun August 2025: 57 developers (10 returning from the original study, 47 newly recruited), 143 repositories, more than 800 tasks, paid 50 dollars an hour against 150 in the original. Accompanied by participant surveys and interviews.",
      "finding": "METR state the data gives an unreliable signal of the current productivity effect of AI tools, because 30 to 50 per cent of developers reported declining to submit tasks they did not want to do without AI, and an increased share declined to take part at all. Raw results now point the other way: an estimated speedup of -18 per cent for returning developers (CI -38 to +9) and -4 per cent for new recruits (CI -15 to +9), against the original +19 per cent slowdown (CI +2 to +39). They believe developers are likely more sped up in early 2026 than in early 2025, while stating their own data is only very weak evidence for the size of that change.",
      "supports": "That the 19 per cent slowdown belongs to early 2025 and should not be quoted as the current effect. It also demonstrates a measurement problem that will worsen: as adoption rises, the people most helped by AI are the ones most likely to select themselves out of any study that asks them to work without it.",
      "doesNotSupport": "That AI now speeds developers up by a specific amount. Every confidence interval here crosses zero, and METR say so. It does not retract the original study, whose perception-gap finding, that participants estimated a 20 per cent speed-up while measured slower, is untouched.",
      "terms": [
        "productivity",
        "self-report",
        "software development",
        "selection effects"
      ],
      "relatedPages": [
        "/research/usage-theatre",
        "/research/the-best-writing-on-ai",
        "/research/the-ai-reports-worth-reading"
      ]
    },
    {
      "id": "ft-junior-consultants-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#ft-junior-consultants-2026",
      "section": "professions",
      "authors": "Kissin, E. (Financial Times)",
      "year": 2026,
      "title": "Junior consultants called back to office as AI increases need for human skills",
      "publication": "Financial Times, 27 August 2026",
      "url": "https://www.ft.com/content/7fd9c234-a92b-4ab2-ba1f-969cf9a23f52",
      "grade": "compiled-review",
      "method": "News reporting. On-the-record interviews with named executives at EY, KPMG, Accenture and Azets, with reference to policies at Deloitte, PwC, BCG, Microsoft and JPMorgan. Not a study, and no measurement of skill or outcome.",
      "finding": "Consulting leaders report that AI has raised the value of interpersonal skills and are considering requiring junior staff in the office more often to develop them. EY's UK head of consulting, Sayeh Ghanbari, is quoted saying firms will \"have to reduce flexibility, but in order to help the human skills\", and that firms dropped training in empathy, storytelling and leadership during the remote-working period while prioritising AI and technical skills. KPMG describes reinventing in-person training; BCG is expanding office social activities; Azets has lowered its degree requirement and is encouraging four days a week. Deloitte and PwC began extra coaching for their youngest UK recruits in 2023 after finding weaker teamwork and communication than earlier cohorts.",
      "supports": "That the apprenticeship-erosion argument is now being acted on by the largest professional services firms, named and on the record. It is strong evidence of institutional belief and of policy change.",
      "doesNotSupport": "That AI caused the deficit, or that office attendance repairs it. No measurement appears anywhere in the reporting, the cohort effects described in 2023 are attributed to pandemic lockdowns rather than to AI, and EY as a firm restated its existing flexibility policy alongside its executive's comments. Testimony from interested parties is not evidence of a mechanism.",
      "terms": [
        "missing rungs",
        "apprenticeship",
        "early careers",
        "human skills"
      ],
      "relatedPages": [
        "/research/how-will-ai-change-consulting",
        "/research/missing-rungs"
      ]
    },
    {
      "id": "stankovic-2025-comment",
      "citeAs": "https://thesuperskills.com/research/evidence#stankovic-2025-comment",
      "section": "judgement",
      "authors": "Stanković, M., Hirche, E., Kollatzsch, S. and Doetsch, J.N.",
      "year": 2025,
      "title": "Comment on: Your Brain on ChatGPT: Accumulation of Cognitive Debt When Using an AI Assistant for Essay Writing Tasks",
      "publication": "arXiv 2601.00856, 29 December 2025",
      "url": "https://arxiv.org/abs/2601.00856",
      "grade": "working-paper",
      "method": "Methodological critique of Kosmyna et al. (arXiv 2506.08872) by researchers at the University of Vienna and TU Dresden. Includes an a priori power analysis using G*Power. Itself a preprint, not peer reviewed.",
      "finding": "Argues the MIT cognitive debt study is underpowered: a repeated-measures design at f=0.25, alpha=.05, power=.95 would require approximately 159 participants against the 54 used, with some figures interpreted from subsamples of two to four essays. The strongest objection concerns the construct itself: the search engine group relied on an external tool yet showed no impairment, with the search-engine to brain-only comparison returning p=1, which the authors say contrasts with the interpretation that task delegation increases cognitive debt. Also documents reporting inconsistencies, an unexplained 55 versus 54 participant discrepancy, and unclear FDR correction levels.",
      "supports": "That the most widely cited term in this area rests on a contested pilot, and that the contest is on the record and specific rather than rhetorical.",
      "doesNotSupport": "That the MIT findings are wrong. It is a critique offered to improve a manuscript for peer review, it is itself unreviewed, and it does not present competing data.",
      "terms": [
        "cognitive debt",
        "replication",
        "statistical power",
        "critique"
      ],
      "relatedPages": [
        "/research/cognitive-debt-and-capability-debt"
      ]
    },
    {
      "id": "huemmer-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#huemmer-2026",
      "section": "judgement",
      "authors": "Huemmer, M., Durner, F., Shyiramunda, T. and Cummings-Koether, M.J.",
      "year": 2026,
      "title": "AI, Metacognition, and the Verification Bottleneck: A Three-Wave Longitudinal Study of Human Problem-Solving",
      "publication": "arXiv 2601.17055, 21 January 2026",
      "url": "https://arxiv.org/abs/2601.17055",
      "grade": "working-paper",
      "method": "Three-wave longitudinal pilot over six months in an academic setting, convenience sample. Exact sample size is not stated in the abstract. Preprint, no journal reference, no control condition.",
      "finding": "Daily AI use rose from 52.4 to 95.7 per cent across the waves. Participants relied most heavily on AI for difficult tasks, 73.9 per cent, while showing declining verification confidence at 68.1 per cent and accuracy of 47.8 per cent on complex tasks. Objective performance fell across problem difficulty from 95.2 to 81.0 to 66.7 to 47.8 per cent, with belief-performance gaps widening to 34.6 percentage points. The authors describe verification rather than solution generation becoming the bottleneck.",
      "supports": "That the verification dimension is measurable and that confidence and accuracy can diverge sharply as task difficulty rises. Useful as a design precedent for measuring verification quality.",
      "doesNotSupport": "Any generalisable effect. The authors list their own limitations in the abstract: convenience sampling from a single academic cohort, self-report bias, no control condition, mathematical problems only, and a timeframe too short for skill trajectories. They state causal validation requires randomised trials.",
      "terms": [
        "verification",
        "metacognition",
        "overconfidence",
        "longitudinal"
      ],
      "relatedPages": [
        "/research/cognitive-debt-and-capability-debt",
        "/research/the-verifiers-discount"
      ]
    },
    {
      "id": "sankaranarayanan-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#sankaranarayanan-2026",
      "section": "learning",
      "authors": "Sankaranarayanan, S.",
      "year": 2026,
      "title": "Mitigating \"Epistemic Debt\" in Generative AI-Scaffolded Novice Programming using Metacognitive Scripts",
      "publication": "arXiv 2602.20206, 22 February 2026, revised 31 March 2026",
      "url": "https://arxiv.org/abs/2602.20206",
      "grade": "working-paper",
      "method": "Between-subjects experiment, 78 participants recruited via Prolific and UserInterviews.com, using a custom Cursor IDE plugin backed by Claude 3.5 Sonnet. Three conditions: manual control, unrestricted AI, and scaffolded AI. Followed by a 30-minute AI-blackout maintenance task. Preprint, not peer reviewed.",
      "finding": "Both AI groups outperformed the manual control on functional utility (p < .001) and did not differ from each other (p = .64). On the subsequent AI-blackout maintenance task, unrestricted AI users failed at 77 per cent against 39 per cent for the scaffolded group. The author describes \"fragile experts\": developers whose high functional utility masks critically low corrective competence.",
      "supports": "That the gap between producing output and being able to repair it can be created inside a single session, and that interface design changes the size of that gap. The scaffolded condition roughly halved the failure rate.",
      "doesNotSupport": "A durable effect on capability. It is a single session with a single blackout task, in novice programming, with no longitudinal follow-up. The author puts \"epistemic debt\" in quotation marks in his own title and builds explicitly on Kirschner, so it is not offered as a new construct.",
      "terms": [
        "epistemic debt",
        "scaffolding",
        "cognitive offloading",
        "novice programming"
      ],
      "relatedPages": [
        "/research/cognitive-debt-and-capability-debt"
      ]
    },
    {
      "id": "jadhav-danve-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#jadhav-danve-2026",
      "section": "work",
      "authors": "Jadhav, R. and Danve, J.",
      "year": 2026,
      "title": "The AI Skills Shift: Mapping Skill Obsolescence, Emergence, and Transition Pathways in the LLM Era",
      "publication": "arXiv 2604.06906, 8 April 2026",
      "url": "https://arxiv.org/abs/2604.06906",
      "grade": "working-paper",
      "method": "Benchmarking of four frontier models (LLaMA 3.3 70B, Mistral Large, Qwen 2.5 72B, Gemini 2.5 Flash) across 263 text-based tasks covering all 35 skills in the US Department of Labor O*NET taxonomy, 1,052 model calls. Cross-referenced against the Anthropic Economic Index. Measures models, not people. Preprint.",
      "finding": "Introduces a Skill Automation Feasibility Index. Mathematics scores 73.2 and programming 71.8 for automation feasibility; active listening scores 42.2 and reading comprehension 45.5. All four models converge to similar skill profiles within a 3.6-point spread. Reports that 78.7 per cent of observed AI interactions are augmentation rather than automation, and a \"capability-demand inversion\" in which the skills most demanded in AI-exposed jobs are those the models perform least well at.",
      "supports": "That measured model capability and stated employer demand point in different directions, which is a useful counterweight to displacement forecasts built on exposure scores alone.",
      "doesNotSupport": "What happens to human capability. The authors state their index \"measures LLM performance on text-based representations of skills, not full occupational execution\". No humans were studied.",
      "terms": [
        "skills taxonomy",
        "automation feasibility",
        "O*NET",
        "augmentation"
      ],
      "relatedPages": [
        "/research/cognitive-debt-and-capability-debt"
      ]
    },
    {
      "id": "sdaia-ethics-2023",
      "citeAs": "https://thesuperskills.com/research/evidence#sdaia-ethics-2023",
      "section": "international",
      "authors": "Saudi Data and AI Authority (SDAIA)",
      "year": 2023,
      "title": "AI Ethics Principles, Version 1.0",
      "publication": "SDAIA, September 2023",
      "url": "https://dgp.sdaia.gov.sa/wps/wcm/connect/4c56ed1c-1b82-447d-ac29-638f5f99c12e/ai-principles-EN.pdf",
      "grade": "institutional-survey",
      "method": "National AI ethics guidance, seven principles with an assessment checklist annexe covering the AI system lifecycle. Non-binding guidance rather than statute. Version 1.0 read in full at source; whether a later version exists could not be checked because SDAIA's main host rejects automated requests.",
      "finding": "The checklist annexe puts this question to designers at the plan and design stage: \"Does your AI system design prevent overconfidence in or overreliance on the AI system with necessary human intervention mechanisms?\" It also asks whether human oversight processes carry defined KPIs and assigned responsibility. The operative text states that decisions which are irreversible or life-and-death \"should trigger human oversight and final determination\", and rules out social scoring and mass surveillance.",
      "supports": "That over-reliance on AI has been named as a design defect to be engineered against in a national governance instrument. On the evidence gathered here it is the only Gulf instrument that does so.",
      "doesNotSupport": "Any obligation. It is guidance, not law, the over-reliance language sits in a checklist annexe rather than in the principle text, and Saudi Arabia had no binding AI statute as of August 2026, only a draft responsible AI policy out for consultation from 2 April 2026. Version 1.0 dates from September 2023 and may have been superseded.",
      "terms": [
        "human oversight",
        "over-reliance",
        "Saudi Arabia",
        "AI ethics"
      ],
      "relatedPages": [
        "/research/ai-and-work-in-the-gulf"
      ]
    },
    {
      "id": "gastat-ict-2025",
      "citeAs": "https://thesuperskills.com/research/evidence#gastat-ict-2025",
      "section": "international",
      "authors": "General Authority for Statistics (GASTAT), Saudi Arabia",
      "year": 2025,
      "title": "Establishments ICT Access and Usage Statistics 2025",
      "publication": "GASTAT, 2025",
      "url": "https://www.stats.gov.sa/documents/d/guest/establishments-ict-access-and-usage-statistics-2025-en-pdf",
      "grade": "institutional-survey",
      "method": "National statistical survey of establishments, methodology stated as aligned with UNCTAD international standards. Enterprise-side only.",
      "finding": "33.1 per cent of establishments use artificial intelligence technologies, a growth of 20.0 per cent against 2024. By sector: information and communication 61.1 per cent, financial and insurance 52.9 per cent, education 51.0 per cent, manufacturing 30.7 per cent, wholesale and retail 30.2 per cent. The companion household and individuals survey for the same year contains no AI indicator at all.",
      "supports": "That one Gulf state measures enterprise AI adoption with a published, standards-aligned method. It is the only official national AI adoption statistic found across Saudi Arabia, the UAE and Qatar.",
      "doesNotSupport": "Anything about citizens, or about employment. Saudi Arabia measures enterprise adoption but not individual use, and publishes no AI employment series. Adoption is also self-reported use of a technology category, not a measure of capability or of value obtained.",
      "terms": [
        "adoption",
        "official statistics",
        "Saudi Arabia",
        "international"
      ],
      "relatedPages": [
        "/research/ai-and-work-in-the-gulf",
        "/research/ai-and-work-by-country"
      ]
    },
    {
      "id": "uae-charter-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#uae-charter-2024",
      "section": "international",
      "authors": "Government of the United Arab Emirates",
      "year": 2024,
      "title": "The UAE Charter for the Development and Use of Artificial Intelligence",
      "publication": "UAE Legislation portal, issued 10 June 2024",
      "url": "https://uaelegislation.gov.ae/en/policy/details/the-uae-charter-for-the-development-and-use-of-artificial-intelligence",
      "grade": "institutional-survey",
      "method": "National charter of twelve principles. Statement of principle with no duty-holder, no enforcement mechanism and no competence requirement.",
      "finding": "Principle 6, Human Oversight, \"emphasizes the irreplaceable value of human judgment and human oversight over AI, aligning with ethical values and social standards to correct any errors or biases that may arise\". The charter is silent on deskilling, over-reliance and any obligation to train or assess the humans doing the overseeing.",
      "supports": "That human oversight is stated as a national principle in the UAE.",
      "doesNotSupport": "That it is operational. There is no named duty-holder, no enforcement, no competence standard and no test of whether oversight is real. The UAE had no federal AI statute as of August 2026.",
      "terms": [
        "human oversight",
        "United Arab Emirates",
        "governance"
      ],
      "relatedPages": [
        "/research/ai-and-work-in-the-gulf"
      ]
    },
    {
      "id": "difc-reg10-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#difc-reg10-2024",
      "section": "institutional",
      "authors": "DIFC Commissioner of Data Protection",
      "year": 2024,
      "title": "Regulation 10 on Personal Data Processed through Autonomous and Semi-Autonomous Systems",
      "publication": "Dubai International Financial Centre, DIFC-DP-GL-23 Rev.03, updated 27 August 2024; regulation enacted September 2023",
      "url": "https://www.difc.com/",
      "grade": "institutional-survey",
      "method": "Binding regulation within the DIFC free zone, with accompanying guidance. Applies to personal data processing by autonomous and semi-autonomous systems, not to AI generally.",
      "finding": "States that \"human-defined processing purposes must always prevail in Systems development and use\". Draws an explicit analogy between an autonomous system and an employee: where a system operates for the benefit of its deployer, \"its position is substantially similar to that of an employee within the Deployer organization, and the Deployer should be therefore liable for its actions in the same way it may be liable for an employee's actions\". Creates a named Autonomous Systems Officer performing a function similar to a data protection officer.",
      "supports": "That the employee analogy for accountability, and a named human role responsible for it, exist in a binding instrument somewhere in the Gulf.",
      "doesNotSupport": "Anything about capability or competence. It is a data protection regulation confined to one free zone, it addresses liability rather than skill, and it imposes no requirement that the responsible human be able to do the work being supervised.",
      "terms": [
        "accountability",
        "human oversight",
        "United Arab Emirates",
        "regulation"
      ],
      "relatedPages": [
        "/research/ai-and-work-in-the-gulf",
        "/research/who-owns-verification-when-ai-does-the-work"
      ]
    },
    {
      "id": "qatar-mcit-ai-principles",
      "citeAs": "https://thesuperskills.com/research/evidence#qatar-mcit-ai-principles",
      "section": "international",
      "authors": "Ministry of Communications and Information Technology, Qatar",
      "year": 2026,
      "title": "Artificial Intelligence in Qatar: Principles and Guidelines for Ethical Development and Deployment",
      "publication": "MCIT Qatar, undated; read at source 28 August 2026",
      "url": "https://www.mcit.gov.qa/en/",
      "grade": "compiled-review",
      "method": "National AI ethics guidance. Read in full at source, but with a provenance defect worth recording: the document carries no publication date, no version number and no reference number, and is not retrievable from the ministry's own website. It states that it \"is legally non-binding, and adherence to it is voluntary\".",
      "finding": "Principle 8, assign ultimate accountability to humans, states that \"AI systems should not be able to autonomously make decisions of significant consequence\" and should \"provide users with the ability to appeal or override decisions that have a substantial impact on individuals or society\". The phrase human in the loop appears nowhere in the document, and Principle 7, develop a human-centered approach, concerns cultural values, feedback, diverse teams and accessibility rather than oversight.",
      "supports": "That Qatar requires humans to retain control of consequential decisions, in guidance.",
      "doesNotSupport": "Any competence duty. Qatar imposes no obligation that the humans exercising control be trained, assessed or kept current, and the document contains no reference to deskilling, over-reliance or automation bias. Widely circulated dates of May 2024 or 2025 for this document are unsourced: it carries no date at all.",
      "terms": [
        "human oversight",
        "accountability",
        "Qatar",
        "governance"
      ],
      "relatedPages": [
        "/research/ai-and-work-in-the-gulf"
      ]
    },
    {
      "id": "imda-agentic-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#imda-agentic-2026",
      "section": "institutional",
      "authors": "Infocomm Media Development Authority (IMDA), Singapore",
      "year": 2026,
      "title": "Model AI Governance Framework for Agentic AI, Version 1.0",
      "publication": "IMDA, published 22 January 2026, launched at Davos",
      "url": "https://www.imda.gov.sg/-/media/imda/files/about/emerging-tech-and-research/artificial-intelligence/mgf-for-agentic-ai.pdf",
      "grade": "institutional-survey",
      "method": "National governance framework for agentic AI, read in full at source. Builds on IMDA's 2020 Model AI Governance Framework. Guidance rather than statute. Law firms report a version 1.5 of 20 May 2026; that could not be confirmed at an official page, so version 1.0 is cited.",
      "finding": "Names deskilling as a risk of agentic deployment, in the terms this research uses. Section 2.4.3: \"As agents take over entry level tasks, which typically serve as the training ground for new staff, this could lead to loss of basic operational knowledge for the users. Organisations should identify core capabilities of each job and provide sufficient training and work exposure so that users retain foundational skills.\" Section 2.4 warns of \"the potential loss of trade craft\" and requires \"sufficient training... to ensure that humans retain core skills\". It also names automation bias directly, requires that overseers be trained to identify common failure modes, and requires that the effectiveness of human oversight itself be audited. It concedes that \"continuous human oversight over all agent workflows becomes impractical at scale\".",
      "supports": "That a national government has written the removal of entry-level work, and the consequent loss of the training ground for junior staff, into an operative AI governance framework. It is the closest external corroboration of the missing rungs argument found in any policy document.",
      "doesNotSupport": "Anything measured. It is guidance, not law, and it states a risk and a duty to train rather than evidence that deskilling has occurred. It also sets no threshold for what counts as retaining core skills, and no test of whether the training works.",
      "terms": [
        "deskilling",
        "missing rungs",
        "human oversight",
        "automation bias",
        "Singapore",
        "agents"
      ],
      "relatedPages": [
        "/research/ai-and-work-in-asia",
        "/research/missing-rungs"
      ]
    },
    {
      "id": "imda-genai-2024",
      "citeAs": "https://thesuperskills.com/research/evidence#imda-genai-2024",
      "section": "institutional",
      "authors": "AI Verify Foundation and IMDA, Singapore",
      "year": 2024,
      "title": "Model AI Governance Framework for Generative AI",
      "publication": "IMDA and AI Verify Foundation, 30 May 2024",
      "url": "https://aiverifyfoundation.sg/resources/mgf-gen-ai/",
      "grade": "institutional-survey",
      "method": "National governance framework for generative AI, nine dimensions. Both published PDF versions read in full and searched at source.",
      "finding": "Contains zero occurrences of \"human oversight\", \"human-in-the-loop\", \"over-reliance\", \"automation bias\" or \"deskill\". Human oversight is not among its nine dimensions. The nearest it comes is a note that \"core skills such as creativity, critical thinking and complex problem-solving are important to helping people harness AI effectively\", and its only use of \"competency\" concerns third-party auditors rather than the human overseer.",
      "supports": "The value here is the confirmed absence, and the trajectory it establishes. In twenty months the same issuing body went from a framework with no oversight language at all to one built around it that also names deskilling. That shift is documented and quotable.",
      "doesNotSupport": "That Singapore was indifferent to oversight in 2024; the 2020 framework it builds on was not read at source and may carry such language. It establishes what the generative AI framework does not say, not what the whole regime did not say.",
      "terms": [
        "human oversight",
        "Singapore",
        "governance",
        "absence"
      ],
      "relatedPages": [
        "/research/ai-and-work-in-asia"
      ]
    },
    {
      "id": "mom-labour-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#mom-labour-2026",
      "section": "international",
      "authors": "Manpower Research and Statistics Department, Ministry of Manpower, Singapore",
      "year": 2026,
      "title": "Labour Market Report, First Quarter 2026",
      "publication": "Ministry of Manpower, Singapore, released 15 June 2026",
      "url": "https://stats.mom.gov.sg/Pages/Labour-Market-Report-1Q-2026.aspx",
      "grade": "institutional-survey",
      "method": "National firm survey by the labour ministry's statistics department. Note the companion press release omits the AI statistic entirely; it appears only in the full report.",
      "finding": "28.5 per cent of firms adopted AI in 2026, highest in information and communications at 74.1 per cent, professional services at 57.5 per cent and financial and insurance services at 56.4 per cent. Only 6.2 per cent reported AI-related reductions in headcount or hiring, against 18.9 per cent reporting redesign of job functions. The ministry's own reading: \"AI is currently having a greater impact on job redesign and work processes than on broad-based job displacement.\"",
      "supports": "That a government labour ministry measuring this finds job redesign running roughly three times ahead of headcount reduction. It is a useful corrective to displacement forecasting.",
      "doesNotSupport": "What happens to capability. Redesign is not neutral: the question this research asks is which tasks the redesign removes, and a firm survey of headcount cannot answer it.",
      "terms": [
        "adoption",
        "job redesign",
        "displacement",
        "Singapore",
        "official statistics"
      ],
      "relatedPages": [
        "/research/ai-and-work-in-asia"
      ]
    },
    {
      "id": "hk-censtatd-2026",
      "citeAs": "https://thesuperskills.com/research/evidence#hk-censtatd-2026",
      "section": "international",
      "authors": "Census and Statistics Department, Hong Kong SAR",
      "year": 2026,
      "title": "Report on the Survey on Information Technology Usage and Penetration in the Business Sector, 2025 Edition",
      "publication": "Census and Statistics Department, Hong Kong, released 27 February 2026",
      "url": "https://www.censtatd.gov.hk/",
      "grade": "institutional-survey",
      "method": "National business survey, fieldwork March to December 2025. Full report and all thirty tables read at source, together with three companion household and ICT publications.",
      "finding": "Artificial intelligence appears nowhere in the report: not in the tables, not in the explanatory notes, not in the definitions. The survey's ICT categories are cloud computing at 98.1 per cent, QR codes at 37.6 per cent, RFID at 20.3 per cent, internet of things at 7.4 per cent and augmented or virtual reality at 1.5 per cent. Hong Kong's statistical office measures AR and VR adoption and does not ask about AI at all. The 2023 edition also had no AI category, and no plan to add one has been announced.",
      "supports": "That Hong Kong publishes no official statistic on AI adoption or AI-related employment, which means every circulating Hong Kong AI adoption figure comes from a non-government survey with a self-selected sample.",
      "doesNotSupport": "That AI adoption in Hong Kong is low. It establishes that it is unmeasured by the government, which is a different and in some ways more useful fact.",
      "terms": [
        "official statistics",
        "Hong Kong",
        "measurement",
        "absence"
      ],
      "relatedPages": [
        "/research/ai-and-work-in-asia"
      ]
    }
  ]
}