{
  "slug": "question-and-exam-generation",
  "name": "Question & exam generation",
  "tier": "leading-edge",
  "trend": "steady",
  "blockerType": null,
  "tools": [
    {
      "name": "Conker.ai",
      "url": "https://conker.ai"
    },
    {
      "name": "QwizLab",
      "url": "https://qwizlab.com"
    },
    {
      "name": "Quizify",
      "url": "https://github.com/rotimi-best/quizify"
    },
    {
      "name": "QuizFlex",
      "url": "https://quizflex.ai"
    },
    {
      "name": "QuizGeniusAI",
      "url": "https://www.quizgeniusai.com"
    },
    {
      "name": "Tough Tongue AI",
      "url": null
    },
    {
      "name": "Wayground",
      "url": null
    },
    {
      "name": "QuestionWell",
      "url": null
    },
    {
      "name": "StudyFetch",
      "url": null
    },
    {
      "name": "Revisely",
      "url": null
    },
    {
      "name": "Quizlet",
      "url": null
    },
    {
      "name": "Kahoot",
      "url": null
    },
    {
      "name": "EduGenius",
      "url": null
    },
    {
      "name": "GenQue",
      "url": null
    },
    {
      "name": "Diffit",
      "url": null
    },
    {
      "name": "Brisk",
      "url": null
    }
  ],
  "evidence": [
    {
      "title": "'Cognitive Surrender': Brown University's Exam Integrity Collapse When AI-Generated Work Dominates",
      "url": "https://www.wgbh.org/news/education-news/2026-09-24/cognitive-surrender-are-college-students-who-use-ai-really-learning",
      "date": "2026-09-24",
      "type": "news-coverage",
      "added": "2026-09-25",
      "superseded_by": null,
      "window": null,
      "explanation": "Take-home exam average jumped to 96 (from 65–80 historical) with 40 perfect scores; in-person final fell to 48, demonstrating how AI-enabled cheating collapses traditional assessment validity."
    },
    {
      "title": "Kellogg's AI-Powered Oral Examination Deployment with Adaptive Questioning",
      "url": "https://www.toughtongueai.com/blog/kellogg-ai-oral-exam/",
      "date": "2026-09-21",
      "type": "case-study",
      "added": "2026-09-25",
      "superseded_by": null,
      "window": null,
      "explanation": "Kellogg professor deployed AI voice oral exams (Tough Tongue AI) for HKUST Executive MBA with adaptive follow-up questioning; institutional redesign response to integrity crisis."
    },
    {
      "title": "MIT's 26,000-Student Learning Outcome Study: AI Homework Gains, Exam Losses, and Institutional Assessment Redesign",
      "url": "https://www.inquirer.com/education/artificial-intelligence-college-students-universities-approaches-mit-harvard-ohio-chicago-20260918.html",
      "date": "2026-09-18",
      "type": "news-coverage",
      "added": "2026-09-25",
      "superseded_by": null,
      "window": null,
      "explanation": "MIT longitudinal study documents AI homework +18% but secure exams −20%, revealing metacognitive laziness; universities respond with bans and in-class assessment redesign."
    },
    {
      "title": "Ivory et al.: ChatGPT Passes 90% of BPS-Accredited Psychology Assessments—Item Design Flaws as the Vulnerability",
      "url": "https://edtechdev.github.io/aied/articles/ivory-psychology-assessment-integrity-2026/",
      "date": "2026-09-17",
      "type": "research-paper",
      "added": "2026-09-25",
      "superseded_by": null,
      "window": null,
      "explanation": "Instrumental case study: ChatGPT passed 36 of 40 (90%) BPS psychology assessments; weak marking criteria and item design flaws, not detection, define the vulnerability."
    },
    {
      "title": "A Scoping Review of Generative Artificial Intelligence Boundaries in Educational Assessment Systems",
      "url": "https://cjlt.ca/index.php/cjlt/en/article/view/29380",
      "date": "2026-09-14",
      "type": "research-paper",
      "added": "2026-09-25",
      "superseded_by": null,
      "window": null,
      "explanation": "Peer-reviewed synthesis of 43 studies (2023–2025) establishing that autonomous high-stakes AI generation is unsupported; hybrid human-AI configurations dominate, positioning governance as the maturity boundary."
    },
    {
      "title": "Teacher Adoption of AI Assessment Tools: RAND, Gallup, and Pew Survey Data on Scale, Satisfaction, and Skepticism",
      "url": "https://www.edugenius.app/blog/ai-education-tools-compared-the-2026-buyers-guide",
      "date": "2026-09-12",
      "type": "tutorial",
      "added": "2026-09-25",
      "superseded_by": null,
      "window": null,
      "explanation": "Independent surveys show ~25% of K-12 teachers use AI tools regularly with time-savings confirmed; one quarter believe AI does more harm than good, documenting adoption and institutional hesitation."
    },
    {
      "title": "Credentialing System Under Strain: AI's 52.6-Point Reality Gap from MedQA to Clinical Practice",
      "url": "https://www.university-365.com/post/ai-assessment-crisis-2026-report",
      "date": "2026-09-12",
      "type": "opinion",
      "added": "2026-09-25",
      "superseded_by": null,
      "window": null,
      "explanation": "Gemini scores 97.4% on MedQA (human pass ~65%) but 44.8% on real clinical cases—a 52.6-point gap; detector tools unreliable, credentialing system reliability under pressure."
    },
    {
      "title": "Governance Failure in AI Exam Generation: Traceability, Provenance, and the Binding Constraint on High-Stakes Deployment",
      "url": "https://news.theopeneyes.com/ai-in-credentialing-can-you-prove-its-still-trustworthy/",
      "date": "2026-09-11",
      "type": "opinion",
      "added": "2026-09-25",
      "superseded_by": null,
      "window": null,
      "explanation": "OpenEyes governance analysis: organizations cannot reconstruct which exam items were AI-generated or how scores were validated; four staff required two weeks to establish provenance for one batch."
    },
    {
      "title": "Independent Buyer's Guide for AI Quiz Generators: Teacher Review as Practitioner Norm",
      "url": "https://theaileaderboard.org/posts/best-ai-quiz-generators",
      "date": "2026-09-11",
      "type": "opinion",
      "added": "2026-09-25",
      "superseded_by": null,
      "window": null,
      "explanation": "Independent guide comparing seven generators emphasizes teacher review of generated items remains mandatory; reflects established practitioner workflow requiring human validation before deployment."
    },
    {
      "title": "MIT's AI-Resilient Curriculum Overhaul: What Every US Student Needs Now",
      "url": "https://eduleague.ng/2026/09/10/mits-ai-syllabus-overhaul-the-150k-course-redesign-every-us-student-needs/",
      "date": "2026-09-10",
      "type": "case-study",
      "added": "2026-09-11",
      "superseded_by": null,
      "window": null,
      "explanation": "MIT EECS redesigned course 6.036 after audit found 73-84% of submitted problem sets were AI-generated; course suspension triggered $150K institutional redesign replacing auto-graded problem sets with oral exams, studio sessions, and AI-augmented projects."
    },
    {
      "title": "AI In US Schools 2026: Policy, Privacy, And Parental Trust",
      "url": "https://eduleague.ng/2026/09/10/ai-in-us-schools-2026-policy-privacy-and-parental-trust/",
      "date": "2026-09-10",
      "type": "adoption-metric",
      "added": "2026-09-11",
      "superseded_by": null,
      "window": null,
      "explanation": "COSN/EdWeek survey: 68% of US public school districts (up from 42% two years prior) now have formally contracted generative AI platforms; Khanmigo 22% market share, MagicSchool 14%; demonstrates mainstream K-12 institutional adoption."
    },
    {
      "title": "Utah district reports critical thinking boost amid AI deployment",
      "url": "https://www.k12dive.com/news/utah-district-reports-critical-thinking-boost-amid-ai-deployment/829919/",
      "date": "2026-09-09",
      "type": "case-study",
      "added": "2026-09-11",
      "superseded_by": null,
      "window": null,
      "explanation": "Jordan School District (Utah) deployed AI conversational questioning tool across 82 teachers supporting 14,000 student interactions; documented 28% critical thinking skill increase and doubled higher-level reasoning; demonstrates production-scale learning gains."
    },
    {
      "title": "When AI Can Produce the Assignment, What Are We Actually Assessing?",
      "url": "https://www.riveraladvisory.com/ai-assessment-higher-education-evidence-of-learning/",
      "date": "2026-09-05",
      "type": "opinion",
      "added": "2026-09-11",
      "superseded_by": null,
      "window": null,
      "explanation": "Natural experiment across 1,066 undergraduates: failure rate shifted from 2-6% (AI-accessible take-home exam) to 18.4% (AI-restricted proctored exam); shows AI-assisted assessments mask competence gaps and misrepresent student achievement."
    },
    {
      "title": "ChatGPT for Teachers Expands to 55 More U.S. Districts",
      "url": "https://emergent.sh/news/chatgpt-teachers-expands-55-more-u-s",
      "date": "2026-09-03",
      "type": "product-ga",
      "added": "2026-09-11",
      "superseded_by": null,
      "window": null,
      "explanation": "OpenAI expanded ChatGPT for Teachers to 55 additional school districts reaching 100,000+ educators; explicitly lists quiz question generation as primary training use case; total deployment now 300,000+ educators across 30+ states."
    },
    {
      "title": "Faster homework, poor exam results: What AI is doing to students' learning",
      "url": "https://www.aljazeera.com/news/2026/9/2/faster-homework-poor-exam-results-what-ai-is-doing-to-students-learning",
      "date": "2026-09-02",
      "type": "news-coverage",
      "added": "2026-09-11",
      "superseded_by": null,
      "window": null,
      "explanation": "30-month longitudinal study of 27,000 students ages 12-18 showing AI homework assistance raised homework scores 18% but cut exam performance 20%; reveals 'metacognitive laziness' risk when students outsource thinking without learning retention."
    },
    {
      "title": "Artificial intelligence in radiology examinations: a psychometric comparison of question generation methods",
      "url": "https://dirjournal.org/articles/artificial-intelligence-in-radiology-examinations-a-psychometric-comparison-of-question-generation-methods/doi/dir.2025.253407",
      "date": "2026-09-01",
      "type": "research-paper",
      "added": "2026-09-11",
      "superseded_by": null,
      "window": null,
      "explanation": "Peer-reviewed study of 115 medical students comparing faculty-written MCQs vs. ChatGPT-4o vs. template-based automatic item generation; template-based AIG achieved acceptable discrimination on all items vs. ChatGPT-4o on 70%; validates AI methods viable for specialized domains."
    },
    {
      "title": "Disclose If Used: California's AI Bar Exam Legislation Offers No Human Review Exemption",
      "url": "https://yage.ai/share/ab1651-bar-exam-ai-disclosure-en-20260825.html",
      "date": "2026-08-25",
      "type": "case-study",
      "added": "2026-08-28",
      "superseded_by": null,
      "window": null,
      "explanation": "California Bar Exam deployed 23 AI-generated scored questions (13.5% of 171) without disclosure or attorney review; 85+ examinees required retake; triggered AB 1651 mandating AI disclosure in all state bar exams 60+ days pre-administration."
    },
    {
      "title": "AI in Indian Education: 2026 Outlook With Real Adoption Data",
      "url": "https://navneetedu.ai/blogs/ai-in-indian-education-2026-outlook/",
      "date": "2026-08-25",
      "type": "adoption-metric",
      "added": "2026-08-28",
      "superseded_by": null,
      "window": null,
      "explanation": "45M Indian students on AI platforms (14% of school population); 63% of metro CBSE schools using ≥1 AI tool; quiz/question generation most-adopted use case (51% of teachers); documented barriers: 30% rural schools lack broadband, 1 device per 13 students."
    },
    {
      "title": "Stanford AI Index 2026: 80% of US secondary students use AI for schoolwork; only 50% schools have AI policies",
      "url": "https://deti.znaj.ua/ru/560210-shi-vikonuye-shkilni-zavdannya-za-sekundi-80-uchniv-uzhe-koristuyutsya",
      "date": "2026-08-23",
      "type": "adoption-metric",
      "added": "2026-08-28",
      "superseded_by": null,
      "window": null,
      "explanation": "80%+ of US secondary students use AI for homework; only 50% of schools have policies and 6% of teachers understand them; reveals massive institutional-readiness gap and assessment-design pressure."
    },
    {
      "title": "教育特化型AIに関する最新研究 (Latest Research on Education-Specialized AI)",
      "url": "https://note.com/u17da/n/n5684ea379099",
      "date": "2026-08-23",
      "type": "opinion",
      "added": "2026-08-28",
      "superseded_by": null,
      "window": null,
      "explanation": "NTT Docomo/Digital Agency synthesis of peer RCTs: Khanmigo 17% usage, 14.5% with reasoning (no effect); Munich TU hint-only vs unrestricted AI (no difference in understanding); McGraw Hill ALEKS 31% time reduction but 25% test-score drop. Shows adoption barriers and learning penalties."
    },
    {
      "title": "The Generative AI Learning Penalty",
      "url": "https://eu.36kr.com/en/p/3950175530270084",
      "date": "2026-08-22",
      "type": "research-paper",
      "added": "2026-08-28",
      "superseded_by": null,
      "window": null,
      "explanation": "26,811-student, 30-month study (Stockholm University/Hong Kong) shows homework scores +18% but closed-book exams -20% after AI adoption; zhongkao/gaokao declines -24%/-18%; 80% of users exhibited homework-outsourcing behavior, revealing assessment validity collapse."
    },
    {
      "title": "What We Measure When We Measure the AI Learning Penalty",
      "url": "https://www.k-gsp.org/p/the-yardstick-problem-what-we-measure",
      "date": "2026-08-21",
      "type": "opinion",
      "added": "2026-08-28",
      "superseded_by": null,
      "window": null,
      "explanation": "Assessment design critique: exams designed for solo unassisted work measure different constructs when AI available; proposes redesigning questions to assess critique and error-detection in AI outputs rather than recall under AI-availability conditions."
    },
    {
      "title": "Google Launches Gemini Study Notebooks with Diagnostic Quizzes and Practice Questions",
      "url": "https://blog.google/products-and-platforms/products/education/back-to-school-2026/",
      "date": "2026-08-19",
      "type": "product-ga",
      "added": "2026-08-28",
      "superseded_by": null,
      "window": null,
      "explanation": "Google GA: diagnostic quizzes in Gemini study notebooks and customized practice quizzes in Search; represents platform-level normalization of AI question generation into mainstream student tools."
    },
    {
      "title": "UGC NET Under Scrutiny: Students & Teachers Question Answer Keys, Challenge Fees and NTA's Exam System",
      "url": "https://www.theunitedindian.com/news/ugc-net-2026-controversy",
      "date": "2026-08-19",
      "type": "news-coverage",
      "added": "2026-08-28",
      "superseded_by": null,
      "window": null,
      "explanation": "National Testing Agency cancelled UGC-NET papers for English, Commerce, Sociology after expert review found repeated questions, factual errors, misspelled scholar names, garbled titles, grammatical errors affecting 20,000 applicants; editorial speculates on AI-assisted generation."
    },
    {
      "title": "Can ChatGPT Write Good Practice Questions? The 7 Tells",
      "url": "https://freefellow.org/blog/can-chatgpt-write-good-practice-questions/",
      "date": "2026-08-17",
      "type": "opinion",
      "added": "2026-08-28",
      "superseded_by": null,
      "window": null,
      "explanation": "FSA/CFA audits AI-generated question banks; documents 7 reproducible defects: answer clustering (47% vs 25% expected), faulty math (16/39), length bias, 121 near-duplicates, position bias, timestamp errors. Evidence of systematic quality gaps."
    },
    {
      "title": "Kentucky Middle School Sends Students Home on First Day of Class With AI-Generated Educational Materials Full of Inexcusable Hallucinations",
      "url": "https://futurism.com/artificial-intelligence/kentucky-middle-school-ai",
      "date": "2026-08-17",
      "type": "news-coverage",
      "added": "2026-08-28",
      "superseded_by": null,
      "window": null,
      "explanation": "Farnsley Middle School distributed AI-generated materials with severe hallucinations (North Dahota, Olkchoma, impossible atomic masses, garbled text); required removing 17 pages before distribution; illustrates production-scale quality-assurance failure."
    },
    {
      "title": "Higher education needs better AI experiences, not more AI tools",
      "url": "https://www.ecampusnews.com/ai-in-education/2026/08/12/higher-education-needs-better-ai-experiences-not-more-ai-tools/",
      "date": "2026-08-12",
      "type": "opinion",
      "added": "2026-08-14",
      "superseded_by": null,
      "window": null,
      "explanation": "D2L platform analyst proposes design pattern where AI quiz generation provides faculty a strong starting point while preserving educator judgment, voice, and instructional intent as the centerpiece of question design workflows."
    },
    {
      "title": "Cinco alucinaciones de IA que pueden 'colarse' en el contenido educativo",
      "url": "https://www.infobae.com/educacion/2026/08/11/cinco-alucinaciones-de-ia-que-pueden-colarse-en-el-contenido-educativo-y-las-estrategias-para-detectarlas/",
      "date": "2026-08-11",
      "type": "news-coverage",
      "added": "2026-08-14",
      "superseded_by": null,
      "window": null,
      "explanation": "Spanish-language research synthesis on AI hallucinations in educational assessment content; cites specific failure rates—only 20% of students identified planted hallucinations; medical residents detected them only 55% of the time in complex scenarios."
    },
    {
      "title": "Ask-E: An Environment for Calibrated Question Generation",
      "url": "https://arxiv.org/abs/2608.06933",
      "date": "2026-08-07",
      "type": "research-paper",
      "added": "2026-08-14",
      "superseded_by": null,
      "window": null,
      "explanation": "Benchmark study finds frontier LLMs achieve below 50% calibration on generating difficulty-matched questions, revealing significant technical limitation in current models' ability to reliably match question difficulty to skill level."
    },
    {
      "title": "OpenAI Education Plugins Move AI From Answers To Workflows",
      "url": "https://www.forbes.com/sites/rayravaglia/2026/08/04/openai-education-plugins-move-ai-from-answers-to-workflows/",
      "date": "2026-08-04",
      "type": "product-ga",
      "added": "2026-08-14",
      "superseded_by": null,
      "window": null,
      "explanation": "OpenAI launched K-12 Educator, College Educator, and Student plugins with quiz and test generation capabilities, positioned to maintain student agency through structured workflows and critical inspection of AI output."
    },
    {
      "title": "A Method for Dialog-Based Knowledge Assessment Using a Two-Loop LLM Interaction",
      "url": "https://injoit.org/index.php/j1/article/view/2736",
      "date": "2026-08-04",
      "type": "research-paper",
      "added": "2026-08-14",
      "superseded_by": null,
      "window": null,
      "explanation": "Peer-reviewed validation of two-loop LLM method (expert-LLM and student-LLM loops) for generating and validating exam content with human oversight, demonstrating production-ready methodology for high-stakes assessment deployment."
    },
    {
      "title": "Best AI Medical Question Generators in 2026: Can Generated Questions Replace a Q-Bank?",
      "url": "https://www.iatrox.com/blog/best-ai-medical-question-generators-2026",
      "date": "2026-08-03",
      "type": "opinion",
      "added": "2026-08-14",
      "superseded_by": null,
      "window": null,
      "explanation": "Domain-expert analysis of medical question generators with five-axis quality framework (vignette realism, distractor plausibility, explanation quality, difficulty calibration, curriculum alignment); conclusion: AI generators are 'good enough to supplement but not foundation' of question banks."
    },
    {
      "title": "Harburg: Students shouldn't be AI's beta testers",
      "url": "https://www.bostonherald.com/2026/08/02/harburg-students-shouldnt-be-ais-beta-testers/",
      "date": "2026-08-02",
      "type": "opinion",
      "added": "2026-08-14",
      "superseded_by": null,
      "window": null,
      "explanation": null
    },
    {
      "title": "Ofqual 2026 AI Guidance for Awarding Bodies",
      "url": "https://eduface.me/resources/blog/ofqual-2026-ai-guidance-awarding-organisations",
      "date": "2026-07-31",
      "type": "industry-report",
      "added": "2026-08-14",
      "superseded_by": null,
      "window": null,
      "explanation": "UK regulator Ofqual identifies AI item and assessment generation as priority use case for awarding bodies with mandatory human-in-the-loop review, establishing governance framework for institutional question generation deployment."
    },
    {
      "title": "ChatGPT-Mediated Exam Preparation Among Pre-service English Teachers: A Qualitative Case Study",
      "url": "https://dergipark.org.tr/en/pub/jlr/article/1891228",
      "date": "2026-07-31",
      "type": "research-paper",
      "added": "2026-08-14",
      "superseded_by": null,
      "window": null,
      "explanation": "Peer-reviewed study of 64 pre-service teachers in Turkey using ChatGPT to generate practice exam questions; documents real adoption use case alongside balanced concerns about accuracy, hallucinations, and alignment with instructional intent."
    },
    {
      "title": "How to Build a Problem Solving Quiz in Minutes With AI",
      "url": "https://www.edugenius.app/blog/how-to-build-a-problem-solving-quiz-in-minutes-with-ai",
      "date": "2026-07-31",
      "type": "tutorial",
      "added": "2026-08-14",
      "superseded_by": null,
      "window": null,
      "explanation": "Practical five-step framework for generating problem-solving quizzes with scaffolded difficulty progression and mandatory human validation; demonstrates workflow time savings (8 min AI vs 60-90 min manual) documented by NCTM research on assessment creation burden."
    },
    {
      "title": "Mexico's UNAM weighs whether to rerun admissions exams amid AI cheating concerns",
      "url": "https://english.elpais.com/international/2026-07-30/mexicos-unam-weighs-whether-to-rerun-admissions-exams-amid-ai-cheating-concerns.html",
      "date": "2026-07-30",
      "type": "case-study",
      "added": "2026-07-31",
      "superseded_by": null,
      "window": null,
      "explanation": "Large-scale institutional deployment: National Autonomous University of Mexico (UNAM) conducted online remote admissions exams for 158,000 applicants (largest public university in Latin America); unusually strong results triggered AI cheating investigation; expert panel deciding between validation and rerun, demonstrating governance challenges at deployment scale."
    },
    {
      "title": "Study finds ChatGPT gets science wrong more often than you think",
      "url": "https://www.sciencedaily.com/releases/2026/03/260317064452.htm",
      "date": "2026-07-29",
      "type": "research-paper",
      "added": "2026-07-31",
      "superseded_by": null,
      "window": null,
      "explanation": "Washington State University empirical testing: ChatGPT showed 76–80% surface accuracy but only 60% above-chance when adjusted; failed to identify false statements 83.6% of time; consistency failure—same question 10 times produced answers flipping true/false multiple times; core reliability limitation for exam question generation."
    },
    {
      "title": "AI Tools & Content - zyBooks",
      "url": "https://www.zybooks.com/ai/",
      "date": "2026-07-28",
      "type": "product-ga",
      "added": "2026-07-31",
      "superseded_by": null,
      "window": null,
      "explanation": "Production deployment: zyBooks (Wiley-owned) released Multiple-Choice Question Generator (July 2026) enabling instructors to generate assessment questions grounded in course content, with teacher review gates and LMS integration; major platform embedding question generation as standard."
    },
    {
      "title": "How teachers are catching AI cheats and sending students back to the analogue age",
      "url": "https://www.independent.co.uk/tech/ai-exam-cheating-university-b3020308.html",
      "date": "2026-07-28",
      "type": "news-coverage",
      "added": "2026-07-31",
      "superseded_by": null,
      "window": null,
      "explanation": "Institutional policy response: multiple elite universities (Princeton, UChicago Law, UCLA, Waterloo) adopting in-person proctored assessment and abandoning take-home remote exams after documented AI cheating; demonstrates institutional loss of confidence in remote unproctored assessment validity."
    },
    {
      "title": "When AI Meets Institutional Reality: What Usage Data from 80+ Higher Education Institutions Actually Shows",
      "url": "https://oeb.global/oeb-insights/when-ai-meets-institutional-reality-what-usage-data-from-80-higher-education-institutions-actually-shows/",
      "date": "2026-07-23",
      "type": "adoption-metric",
      "added": "2026-07-31",
      "superseded_by": null,
      "window": null,
      "explanation": "Production-scale adoption metric: LearnWise AI's 2026 analysis of 191k+ student-AI tutor conversations across 80+ institutions shows 35% of interactions involve generating quizzes, flashcards, and revision questions; direct evidence of question generation as primary student use case."
    },
    {
      "title": "Best AI Quiz Makers for Teachers in 2026 (Tested and Compared)",
      "url": "https://blog.aieducator.tools/posts/best-ai-quiz-makers-teachers-2026",
      "date": "2026-07-23",
      "type": "tutorial",
      "added": "2026-07-31",
      "superseded_by": null,
      "window": null,
      "explanation": "Practitioner hands-on testing by educator (trained 150k+ teachers): evaluated Conker, Knowt, MagicSchool in real UK/US classrooms; finding—teacher judgment is essential; all generated quizzes require review; real value is workflow integration and response quality, not generation alone."
    },
    {
      "title": "Anthropic Launches Claude for Teachers. Why Some Critics Are Concerned",
      "url": "https://www.edweek.org/technology/anthropic-launches-claude-for-teachers-why-some-critics-are-concerned/2026/07",
      "date": "2026-07-17",
      "type": "news-coverage",
      "added": "2026-07-31",
      "superseded_by": null,
      "window": null,
      "explanation": "Market signal: major AI vendor (Anthropic) launched Claude for Teachers (July 2026), joining OpenAI, Google, Microsoft, Khan Academy; includes lesson planning and assessment generation features; critical expert opinion flags de-skilling risks and lack of differentiation from incumbents."
    },
    {
      "title": "AI homework tools cut exam scores by 20%, study of 26,000 Chinese students finds",
      "url": "https://www.thestar.com.my/tech/tech-news/2026/07/17/ai-homework-tools-cut-exam-scores-by-20-study-of-26000-chinese-students-finds",
      "date": "2026-07-17",
      "type": "research-paper",
      "added": "2026-07-31",
      "superseded_by": null,
      "window": null,
      "explanation": "Large-scale longitudinal negative outcome: 30-month study (26,000 students, Stockholm & Hong Kong universities) found AI homework tools boosted homework scores 18% but exam performance dropped 20% after 6 months, 24% on gaokao, 18% on zhongkao; behavioral evidence that outsourcing to AI undermines learning."
    },
    {
      "title": "More than 50% of Australian university assignments used AI. How should unis respond?",
      "url": "https://theconversation.com/more-than-50-of-australian-university-assignments-used-ai-how-should-unis-respond-287179",
      "date": "2026-07-13",
      "type": "news-coverage",
      "added": "2026-07-17",
      "superseded_by": null,
      "window": null,
      "explanation": "Turnitin data (Oct 2025–Apr 2026): 53.6% of Australian tertiary submissions used AI; educators demand education-specific AI tools rather than generic ChatGPT, indicating market shift toward purpose-built assessment and question-generation platforms."
    },
    {
      "title": "Comparative Evaluation of AI-Generated vs. Expert-written Answer Explanations for a Medical Education Self-Assessment",
      "url": "https://aclanthology.org/2026.bea-1.31/",
      "date": "2026-07-10",
      "type": "research-paper",
      "added": "2026-07-17",
      "superseded_by": null,
      "window": null,
      "explanation": "BEA 2026 peer-reviewed study: AI-generated MCQ explanations rated significantly higher on information amount (OR=1.99, p=0.001) vs. expert-written; 20% of AI explanations judged to need correction vs. 38% of expert; validates AI-assisted MCQ authoring in medical education."
    },
    {
      "title": "AI Writes Wrong SC-500 Exam Question, Developer Switches to Reading Microsoft Docs",
      "url": "https://www.linkedin.com/posts/mathewclarkau_sc500_microsoft_cybersecurity_activity-7480953972615168001-shsN",
      "date": "2026-07-09",
      "type": "case-study",
      "added": "2026-07-17",
      "superseded_by": null,
      "window": null,
      "explanation": "Production deployment: ReadRoost rebuilt 552-question SC-500 practice bank after AI hallucinated non-existent feature; implemented verification gate requiring grounding in live Microsoft Learn docs; demonstrates how production systems handle quality risks."
    },
    {
      "title": "I Tested Whether Its Quiz Maker Is Actually Classroom-Ready",
      "url": "https://www.smartpostly.com/blogs/conker-ai-review-whether-its-quiz-maker-actually-classroom-ready/",
      "date": "2026-07-09",
      "type": "opinion",
      "added": "2026-07-17",
      "superseded_by": null,
      "window": null,
      "explanation": "Critical product review of Conker AI (on tools list): questions are 'mostly recall-based,' distractors 'too easy to eliminate,' grade-level control limited; verdict: 'For simple comprehension, good. For deeper learning, needed improvement.' Teacher editing is 'not optional.'"
    },
    {
      "title": "Brown Professor Suspects Most of His Class Used AI to Cheat",
      "url": "https://www.insidehighered.com/news/faculty/learning-assessment/2026/07/08/brown-professor-suspects-most-his-class-used-ai-cheat",
      "date": "2026-07-08",
      "type": "case-study",
      "added": "2026-07-17",
      "superseded_by": null,
      "window": null,
      "explanation": "Critical negative signal: Brown University ECON 1170 take-home midterm showed 96% average (40 perfect scores); when ChatGPT-verified, only ~70% of problems returned correct answers; in-person final saw 48.6% average (historic low 65%+), 22 midterm perfect-scorers failed, documenting widespread AI cheating and learning collapse."
    },
    {
      "title": "Does This Open Questions or Close Them?",
      "url": "https://purposefulai.substack.com/p/does-this-open-questions-or-close?action=share",
      "date": "2026-07-07",
      "type": "opinion",
      "added": "2026-07-17",
      "superseded_by": null,
      "window": null,
      "explanation": "Field experiment (Bastani et al., ~1000 students): unrestricted ChatGPT scored 17% worse on final exams despite solving 48% more practice problems; guardrailed 'GPT Tutor' (hints only) posted 127% practice gains with zero exam penalty; negative signal on unstructured tool use."
    },
    {
      "title": "A Dartmouth Study Found 90.2% Engagement on 'Optional' Coursework — Here's the AI-Graded Assessment Service You Can Sell From It",
      "url": "https://www.figuringoutwithai.com/playbooks/ai-graded-embedded-quiz-service-playbook-2026",
      "date": "2026-07-05",
      "type": "case-study",
      "added": "2026-07-17",
      "superseded_by": null,
      "window": null,
      "explanation": "Dartmouth College deployment: Phosphor platform with LLM-graded generated questions achieved 90.2% adoption on optional coursework, +0.71–1.30 SD exam performance gains; constructed-response questions predict learning while MCQ-only formats do not."
    },
    {
      "title": "The Impact of Generative AI on Student Learning: Why the OECD Warns Against 'Fast AI'",
      "url": "https://www.thesify.ai/blog/impact-generative-ai-student-learning-oecd",
      "date": "2026-07-03",
      "type": "industry-report",
      "added": "2026-07-17",
      "superseded_by": null,
      "window": null,
      "explanation": "OECD Digital Education Outlook 2026: 'fast AI' (generic chatbots generating answers) causes 17% exam-score decline despite 127% practice gains, while 'slow AI' (pedagogically-designed tools) shows sustained learning; critical negative evidence on tool design mattering for outcomes."
    },
    {
      "title": "A Cross-National Analysis of Generative AI Guidance and Design Patterns in English-Medium Universities",
      "url": "https://dergipark.org.tr/en/pub/per/article/1885223",
      "date": "2026-07-03",
      "type": "research-paper",
      "added": "2026-07-17",
      "superseded_by": null,
      "window": null,
      "explanation": "Systematic analysis of 21 institutional GenAI-in-assessment policies across Europe, North America, Australasia, Asia: all allow student use under conditions with disclosure required; four recurrent design patterns emerged (process portfolios, AI+verification, critical engagement, secure exams + AI coursework)."
    },
    {
      "title": "Hallucination vs Confabulation: Why LLMs Invent Answers Instead of Saying 'I Don't Know'",
      "url": "https://cristobalsantana.substack.com/p/hallucination-vs-confabulation-why",
      "date": "2026-06-30",
      "type": "opinion",
      "added": "2026-07-03",
      "superseded_by": null,
      "window": null,
      "explanation": "Foundational explanation of LLM confabulation mechanisms: models optimize fluent continuation, not fact verification; cannot distinguish recalling facts from probabilistic guessing—explains why AI-generated exam questions lack intrinsic quality verification and require human review."
    },
    {
      "title": "Study: Using AI Chatbots Creates Significant Learning Loss",
      "url": "https://www.forbes.com/sites/dereknewton/2026/06/30/study-using-ai-chatbots-creates-significant-learning-loss/",
      "date": "2026-06-30",
      "type": "news-coverage",
      "added": "2026-07-03",
      "superseded_by": null,
      "window": null,
      "explanation": "Negative signal: students using AI on assessments experience 25% learning loss; reduce engagement on solvable problems 27%; signals why assessment design must evolve when AI access is present—context for reforming questions toward authentic tasks."
    },
    {
      "title": "K-12 Testing and Assessment Market Report",
      "url": "https://www.congruencemarketinsights.com/report/k-12-testing-and-assessment-market",
      "date": "2026-06-29",
      "type": "industry-report",
      "added": "2026-07-03",
      "superseded_by": null,
      "window": null,
      "explanation": "Market analysis: K-12 assessment market USD 1.1B→2.1B (2025–2033, 8.39% CAGR); 55% U.S. school districts deploy AI-powered assessment; 60%+ prefer adaptive platforms; by 2028, automated grading expected to reduce evaluation time 35%."
    },
    {
      "title": "Enhancing Assurance of Learning: A Human-AI Co-Assessment Model for Program Evaluation",
      "url": "https://aisel.aisnet.org/treos_amcis2026/132/",
      "date": "2026-06-25",
      "type": "research-paper",
      "added": "2026-07-03",
      "superseded_by": null,
      "window": null,
      "explanation": "Peer-reviewed AMCIS framework: AI performs rubric-based assessment at scale; human experts conduct higher-order judgment; disagreements surface design problems (ambiguous outcomes, rubric gaps)—governance model for quality assurance in AI-assisted assessment design."
    },
    {
      "title": "'A bit of chaos and madness': The AI Assessment Scale and the work of assessment reform",
      "url": "https://arxiv.org/abs/2606.26729",
      "date": "2026-06-25",
      "type": "research-paper",
      "added": "2026-07-03",
      "superseded_by": null,
      "window": null,
      "explanation": "Qualitative study of AIAS adoption at universities: shared language legitimizes GenAI use and clarifies boundaries; effectiveness depends on institutional governance, tool access, staff confidence, and disciplinary context—signals maturity barriers beyond technical capability."
    },
    {
      "title": "Using an AI practice bank to support formative assessment in economics",
      "url": "https://educational-innovation.sydney.edu.au/teaching@sydney/using-an-ai-practice-bank-to-support-formative-assessment-in-economics/",
      "date": "2026-06-24",
      "type": "case-study",
      "added": "2026-07-03",
      "superseded_by": null,
      "window": null,
      "explanation": "University of Sydney semester-long pilot: custom ChatGPT generated 100+ exam-style practice questions; 97/130 students (74%) used it; 65% rated useful; teacher reflection confirms AI scales practice but requires quality control and teacher judgment."
    },
    {
      "title": "For Better and for Worse? AI, New Technologies and the Future of Assessment (Part 1)",
      "url": "https://internationalednews.com/2026/06/24/for-better-and-for-worse-ai-new-technologies-and-the-future-of-assessment-part-1/",
      "date": "2026-06-24",
      "type": "news-coverage",
      "added": "2026-07-03",
      "superseded_by": null,
      "window": null,
      "explanation": "Editorial analysis documents AI production deployment: 'AI is already used to grade most writing on New Jersey's standardized tests'; names Classtime platform providing instant feedback; captures realized high-stakes capability amid legitimate concerns on validity."
    },
    {
      "title": "The Assessment Equity Problem: How AI-Powered Practice Tests Are Leveling the Playing Field for Underserved Students",
      "url": "https://www.evelynlearning.com/blog/the-assessment-equity-problem-how-ai-powered-practice-tests-are-leveling-the-playing-field-for-underserved-students",
      "date": "2026-06-24",
      "type": "case-study",
      "added": "2026-07-03",
      "superseded_by": null,
      "window": null,
      "explanation": "Operational deployment: Evelyn Learning AI Practice Test Generator serves standardized test prep (SAT/ACT/AP); addresses access barriers (cost, geographic clustering, personalized feedback); unit economics: AI-assisted generation at fraction of traditional MCQ bank cost."
    },
    {
      "title": "Why Machines Misread Pedagogical Quality: Human-Machine Alignment in LLM-Based Pretest Question Evaluation",
      "url": "https://arxiv.org/abs/2606.23629",
      "date": "2026-06-22",
      "type": "research-paper",
      "added": "2026-07-03",
      "superseded_by": null,
      "window": null,
      "explanation": "Empirical study of AI-assisted pretest question generation: human-machine disagreements on pedagogical quality are systematic, not random; rubric operationalization and rationale-first evaluation close alignment gaps—signals maturity requirement for production deployment."
    },
    {
      "title": "Corpus Prevalence of Multiple-Choice Question Options",
      "url": "https://deeplearn.org/arxiv/779065/corpus-prevalence-of-multiple-choice-question-options",
      "date": "2026-06-22",
      "type": "research-paper",
      "added": "2026-07-03",
      "superseded_by": null,
      "window": null,
      "explanation": "Computational analysis reveals LLM-generated MCQ distractors inherit corpus-bias: correct answers significantly more prevalent in text corpora than distractors; corpus prevalence unreliable signal for pedagogical plausibility—technical limitation in scaling quality questions."
    },
    {
      "title": "AI Practice Test Generator vs. Traditional Test Banks: The Data",
      "url": "https://www.evelynlearning.com/blog/the-science-of-standardized-test-prep-how-ai-generated-practice-questions-are-outperforming-traditional-test-banks",
      "date": "2026-06-21",
      "type": "adoption-metric",
      "added": "2026-07-03",
      "superseded_by": null,
      "window": null,
      "explanation": "Research synthesis: adaptive AI-generated practice questions yield 15–25 percentile point improvements over static test banks; personalized sequences 1.5× better than fixed; validates retrieval-practice design principles at scale in deployed platforms."
    },
    {
      "title": "Saudi Edtech M&A National AI Rollout Guide Outlook",
      "url": "https://saudimergersacquisitions.com/insights/article/saudi-edtech-ma-the-high-stakes-future-ready-play-to-buy-into-the-national-ai-curriculum-rollout",
      "date": "2026-06-20",
      "type": "adoption-metric",
      "added": "2026-07-03",
      "superseded_by": null,
      "window": null,
      "explanation": "National pilot: Saudi Arabia AI education rollout (2025–2026) reached 50,000+ students; Collage AI + StudyWise platform generates personalized exams, automates grading, detects knowledge gaps—signals scaled institutional deployment with specific capability set."
    },
    {
      "title": "Top 10 AI Quiz & Assessment Generation Tools: Features, Pros, Cons & Comparison",
      "url": "https://www.devopsschool.com/blog/top-10-ai-quiz-assessment-generation-tools-features-pros-cons-comparison/",
      "date": "2026-06-18",
      "type": "product-ga",
      "added": "2026-06-19",
      "superseded_by": null,
      "window": null,
      "explanation": "2026 market survey documenting 10 AI platforms: standards alignment, adaptive difficulty, multi-modal items, LMS integration, analytics; represents mature ecosystem across K-12, higher education, corporate training, test prep."
    },
    {
      "title": "Turn Entry Tickets Into Smart Item Analysis - TCEA Blog",
      "url": "https://blog.tcea.org/turn-entry-tickets-into-smart-item-analysis/",
      "date": "2026-06-17",
      "type": "tutorial",
      "added": "2026-06-19",
      "superseded_by": null,
      "window": null,
      "explanation": "K-12 teacher workflow using Gen AI to generate pre-assessment questions, design lessons via ALDO framework, and analyze item difficulty (Q1 46.67% correct, Q5 20%); demonstrates practical AI-assisted question generation in classroom practice."
    },
    {
      "title": "How to Use AI for Formative Assessment (Real Examples)",
      "url": "https://www.aimadefor.com/blog/ai-formative-assessment-teachers/",
      "date": "2026-06-16",
      "type": "tutorial",
      "added": "2026-06-19",
      "superseded_by": null,
      "window": null,
      "explanation": "Year-long classroom deployment guide with 4 tools (Conker, Quizizz, Formative, MagicSchool); documents limitation that AI-generated questions require review, establishing production workflow for formative assessment use."
    },
    {
      "title": "End-To-End Assessment Solutions for Schools — ManageBac Integration",
      "url": "https://www.managebac.com/assessprep",
      "date": "2026-06-04",
      "type": "adoption-metric",
      "added": "2026-06-05",
      "superseded_by": null,
      "window": null,
      "explanation": "AssessPrep deployment metrics: 800+ schools, 4M+ assessments delivered, 500K AI-generated questions, demonstrating production-scale institutional adoption across international curricula (IB DP/MYP, Cambridge IGCSE, A-Level, Edexcel)."
    },
    {
      "title": "AI chatbots fail medical misinformation test, returning inaccurate and fabricated advice",
      "url": "https://www.psypost.org/ai-chatbots-fail-medical-misinformation-test-returning-inaccurate-and-fabricated-advice/",
      "date": "2026-06-01",
      "type": "research-paper",
      "added": "2026-06-05",
      "superseded_by": null,
      "window": null,
      "explanation": "BMJ Open study (Tiller et al.): 49.6% of medical chatbot responses problematic (30% somewhat, 19.6% highly); hallucinated citations across all models; signals critical quality risk for AI-generated medical exam content without human review."
    },
    {
      "title": "Question Type, Cognitive Load, and CEFR Alignment: Evaluating LLM-Generated EFL Grammar Drill Exercises",
      "url": "https://arxiv.org/abs/2606.01592v2",
      "date": "2026-06-01",
      "type": "research-paper",
      "added": "2026-06-05",
      "superseded_by": null,
      "window": null,
      "explanation": "Peer-reviewed study with real student performance data from Japanese junior high EFL classroom: LLM-generated grammar exercises show pedagogically sound question design; cloze tasks showed highest cognitive load, demonstrating successful formative deployment with learning outcome analysis."
    },
    {
      "title": "AI Chatbots Fail Medical Questions One in Five Times, Study Reveals",
      "url": "https://www.gadgetreview.com/ai-chatbots-fail-medical-questions-one-in-five-times-study-reveals",
      "date": "2026-05-29",
      "type": "research-paper",
      "added": "2026-06-05",
      "superseded_by": null,
      "window": null,
      "explanation": "Penn State study of four chatbots on 212 medical questions: ChatGPT-4o 84.6%, Llama3-8b ~50%; domain-specific weakness in neurology/dermatology; physician review warns against overreliance, highlighting domain expertise requirement for assessment validity."
    },
    {
      "title": "生成AIパスポート AIクイズアプリ、利用回数1000万回を突破",
      "url": "https://www.managebac.com/assessprep",
      "date": "2026-05-27",
      "type": "adoption-metric",
      "added": "2026-06-05",
      "superseded_by": null,
      "window": null,
      "explanation": "Japanese AI Passport Quiz App reached 10 million uses in ~16 months (launched May 2024), generating true/false questions for AI literacy certification; demonstrates independent, grassroots adoption of AI-generated assessment at scale outside US/UK markets."
    },
    {
      "title": "Free AI Quiz Generator | Live Quizzes in 30s | QuizMaker",
      "url": "https://www.quiz-maker.com/AI-Quiz-Generator",
      "date": "2026-05-26",
      "type": "adoption-metric",
      "added": "2026-06-05",
      "superseded_by": null,
      "window": null,
      "explanation": "QuizMaker AI generator trusted by 10,000+ schools and organizations; generates finished quizzes in 30 seconds from topic/PDF/URL/image; 26 pre-built sample quizzes demonstrate ecosystem maturity and broad educator adoption."
    },
    {
      "title": "The Verification Gap: What Stanford's 2026 AI Index Reveals About Single-Model Reliability",
      "url": "https://dev.to/nick_18/the-verification-gap-what-stanfords-2026-ai-index-reveals-about-single-model-reliability-1gl7",
      "date": "2026-05-25",
      "type": "opinion",
      "added": "2026-06-05",
      "superseded_by": null,
      "window": null,
      "explanation": "Analysis of Stanford AI Index: hallucination rates 22–94% across 26 frontier models; models overconfident precisely where wrong (hard-easy effect); signals calibration failure critical for AI assessment oversight and aligns with EU AI Act transparency requirements."
    },
    {
      "title": "AI in the Assessor's Chair",
      "url": "https://andrewomalley.substack.com/p/ai-in-the-assessors-chair",
      "date": "2026-05-25",
      "type": "opinion",
      "added": "2026-06-05",
      "superseded_by": null,
      "window": null,
      "explanation": "Synthesis of four independent 2026 peer-reviewed studies on AI in medical education assessment: Claude 3.5 Sonnet 86% on expert evaluation, GPT-4 strong on relevance but weaker models 77%; consistent finding across studies: human-in-the-loop remains non-negotiable."
    },
    {
      "title": "AI chatbots right about daily news nearly all the time, but fail badly when users slip wrong details into chats",
      "url": "https://www.hackshackers.com/ai-chatbots-news-intermediaries-stanford-study/",
      "date": "2026-05-22",
      "type": "research-paper",
      "added": "2026-06-05",
      "superseded_by": null,
      "window": null,
      "explanation": "Stanford evaluation (Suzgun et al.): frontier models >95% on valid prompts, collapse to 19% (GPT-5) when false premises introduced; signals exam validity risk—AI cannot reliably detect or reject flawed question premises."
    },
    {
      "title": "Widespread AI misuse forces higher education to rethink assessment",
      "url": "https://phys.org/news/2026-05-widespread-ai-misuse-higher-rethink.html",
      "date": "2026-05-21",
      "type": "research-paper",
      "added": "2026-05-22",
      "superseded_by": null,
      "window": null,
      "explanation": "Large-scale study of 95,000+ students at 20 US universities: 37% use generative AI on assignments, 9% to cheat; motivates urgent assessment reform via three strategies including AI-generated questions that adapt to integrity challenges."
    },
    {
      "title": "Make any quiz with AI in 30 seconds — and the three attempts it took to ship it",
      "url": "https://dev.to/apostopher/make-any-quiz-with-ai-in-30-seconds-and-the-three-attempts-it-took-to-ship-it-5c2f",
      "date": "2026-05-20",
      "type": "opinion",
      "added": "2026-05-22",
      "superseded_by": null,
      "window": null,
      "explanation": "Engineering practitioner documents three implementation attempts and real-world constraints for deployed AI quiz generation, including graceful handling of imperfect AI output, illustrating practical deployment barriers beyond technical capability."
    },
    {
      "title": "School AI pilot programs: A K-12 roadmap",
      "url": "https://schoolai.com/blog/how-states-rolling-out-ai-public-education",
      "date": "2026-05-19",
      "type": "case-study",
      "added": "2026-05-22",
      "superseded_by": null,
      "window": null,
      "explanation": "Reports on K-12 district AI pilots in Connecticut, Utah, Michigan, NYC with specific examples of AI-generated exit questions and assessment feedback; shows real classroom deployment of question generation at scale."
    },
    {
      "title": "Universities After Generative AI: Assessment, Integrity and the Public Mission of Learning",
      "url": "https://gfoss.eu/universities-after-generative-ai-assessment-integrity-and-the-public-mission-of-learning/",
      "date": "2026-05-16",
      "type": "opinion",
      "added": "2026-05-22",
      "superseded_by": null,
      "window": null,
      "explanation": "Strategic analysis proposing mixed assessment ecology (extended essays, drafts, oral defenses, AI declarations) and process assessment over product assessment; directly informs question generation system design for resilient evaluation."
    },
    {
      "title": "PDF to Quiz Generator | Convert Documents to Tests - QuizMagic",
      "url": "https://quizmagic.io/pdf-to-quiz",
      "date": "2026-05-15",
      "type": "product-ga",
      "added": "2026-05-22",
      "superseded_by": null,
      "window": null,
      "explanation": "SaaS platform generating quizzes from PDFs, documents, videos, and topics with 30-45 second generation time; supports Bloom's Taxonomy alignment and multiple question types, signaling maturation of document-to-assessment generation at scale."
    },
    {
      "title": "Using randomization to compare AI and expert-generated formative assessment questions in medical education",
      "url": "https://2medical.news/2026/05/14/using-randomization-to-compare-ai-and-expert-generated-formative-assessment-questions-in-medical-education/",
      "date": "2026-05-14",
      "type": "research-paper",
      "added": "2026-05-22",
      "superseded_by": null,
      "window": null,
      "explanation": "Peer-reviewed randomized trial: AI-generated and expert questions showed equivalent performance but 53% of AI items rated 'very easy/easy' vs 31% of expert items, revealing quality perception gaps despite statistical equivalence."
    },
    {
      "title": "AI in Education 2026: What Schools Are Actually Deploying",
      "url": "https://openeducat.org/articles/ai-in-education-2026/",
      "date": "2026-05-13",
      "type": "adoption-metric",
      "added": "2026-05-22",
      "superseded_by": null,
      "window": null,
      "explanation": "Real-world K-12 deployment data shows AI quiz/question generation ranks among five highest-deployment AI use cases (3-5 month pilot-to-deployment timeline), with guidance grounded in UNESCO, OECD, and EU AI Act frameworks."
    },
    {
      "title": "Reimagining Assessment in the Age of Generative AI: Lessons from Open-Book Exams with ChatGPT",
      "url": "https://arxiv.org/abs/2605.12363",
      "date": "2026-05-12",
      "type": "research-paper",
      "added": "2026-05-22",
      "superseded_by": null,
      "window": null,
      "explanation": "Empirical study of ChatGPT use during exams reveals three usage patterns; argues assessment must shift from testing solution production to evaluating reasoning and verification skills, directly informing question design in AI-integrated contexts."
    },
    {
      "title": "Creating a quiz with AI: Benefits and Limitations",
      "url": "https://www.experquiz.com/en/articles/creating-a-quiz-with-ai-benefits-and-limitations",
      "date": "2026-05-12",
      "type": "opinion",
      "added": "2026-05-22",
      "superseded_by": null,
      "window": null,
      "explanation": "Practitioner perspective documents fundamental quality trade-offs: AI may generate overly simple questions with implausible distractors and inconsistent discrimination, signaling that quality variability remains a core adoption barrier."
    },
    {
      "title": "Mapping the Territory, Part 2: The Minefields of AI in Scientific Research and Teaching",
      "url": "https://grantwp.substack.com/p/mapping-the-territory-part-2-the",
      "date": "2026-05-11",
      "type": "opinion",
      "added": "2026-05-22",
      "superseded_by": null,
      "window": null,
      "explanation": "Academic critique identifies assessment validity collapse when AI can replicate student answers; documents pedagogical risk that traditional evaluation forms no longer measure what faculty assume, establishing critical boundary for question generation use cases."
    },
    {
      "title": "Quiz Creator: Create Custom Quizzes for Your Class - Edzo",
      "url": "https://www.edzo.com/tools/quiz-creator",
      "date": "2026-05-11",
      "type": "product-ga",
      "added": "2026-05-22",
      "superseded_by": null,
      "window": null,
      "explanation": "K-12 classroom tool with AI question generation from topic descriptions; includes five question types, curriculum alignment (Australia/US/UK), real-time marking, and read-aloud accessibility, demonstrating formative assessment adoption in schools."
    },
    {
      "title": "PressPrimer Quiz – AI Quiz Maker, Exam Builder & LMS Assessment Plugin - WordPress",
      "url": "https://ca.wordpress.org/plugins/pressprimer-quiz/",
      "date": "2026-05-11",
      "type": "product-ga",
      "added": "2026-05-22",
      "superseded_by": null,
      "window": null,
      "explanation": "WordPress plugin enabling unlimited AI question generation via OpenAI API with LMS integration (LearnDash, Tutor, LifterLMS) and server-side validation, demonstrating production-ready deployment in self-hosted/non-SaaS education ecosystems."
    },
    {
      "title": "Publisher Withdraws Study Claiming ChatGPT Boosts Learning",
      "url": "https://www.computing.co.uk/news/2026/ai/publisher-pulls-study-claiming-chatgpt-boosts-learning",
      "date": "2026-05-06",
      "type": "news-coverage",
      "added": "2026-05-08",
      "superseded_by": null,
      "window": null,
      "explanation": "Springer Nature retracted widely-cited meta-analysis (262 peer-reviewed citations) for discrepancies in analysis; signals research reliability concerns and premature claims in AI education field, critical context for tier classification uncertainty."
    },
    {
      "title": "Pearson Data Shows AI-Powered Practice Boosts Student Proficiency by 90%",
      "url": "https://www.prnewswire.com/news-releases/new-pearson-data-shows-students-build-proficiency-with-ai-powered-practice-302762025.html",
      "date": "2026-05-05",
      "type": "adoption-metric",
      "added": "2026-05-08",
      "superseded_by": null,
      "window": null,
      "explanation": "Large-scale production deployment: Pearson Study Prep's AI-adaptive practice questions with 62,000+ higher ed students achieved 90% higher mastery rates and 60% higher proficiency with goal-setting, demonstrating measurable learning gains at scale."
    },
    {
      "title": "Scenario-Based Assessment in the Age of Generative AI",
      "url": "https://ies.ed.gov/use-work/awards/scenario-based-assessment-age-generative-ai-making-space-education-market-alternative-assessment",
      "date": "2026-05-01",
      "type": "industry-report",
      "added": "2026-05-08",
      "superseded_by": null,
      "window": null,
      "explanation": "U.S. Department of Education (IES/NCER) awarded $3.6M, 3-year grant to develop AI-enhanced scenario-based assessment authoring tool; signals federal policy recognition of both opportunity and need for scaled AI assessment generation."
    },
    {
      "title": "AI Test Development Platform - PSI Exams",
      "url": "https://www.psiexams.com/test-owners/ai-test-development/",
      "date": "2026-04-30",
      "type": "product-ga",
      "added": "2026-05-08",
      "superseded_by": null,
      "window": null,
      "explanation": "Production-grade national licensure deployment: 77.4% of AI-generated items met psychometric thresholds vs. 75.5% human-authored; multi-agent AI validation with expert SME review ensures accuracy and accountability at scale."
    },
    {
      "title": "Podcast - Managing the Future of Work - Harvard Business School",
      "url": "https://www.hbs.edu/managing-the-future-of-work/podcast/Pages/podcast-details.aspx?episode=8097747294",
      "date": "2026-04-25",
      "type": "product-ga",
      "added": "2026-05-08",
      "superseded_by": null,
      "window": null,
      "explanation": "Khan Academy announces 'Assessments' product with psychometrics, norming, AI-powered question design including open-ended questions and narrative feedback; transition from tutoring focus to structured assessment capability."
    },
    {
      "title": "AI Impact on Decision-Making: Trade-offs Between Performance and Assessment Validity",
      "url": "https://jacehargis.substack.com/p/ai-impact-on-decision-making",
      "date": "2026-04-25",
      "type": "opinion",
      "added": "2026-05-08",
      "superseded_by": null,
      "window": null,
      "explanation": "Empirical research documents critical trade-off: AI-assisted assessments boost observable performance 30 points but collapse assessment reliability (Cronbach's α: 0.87→0.31), degrading diagnostic validity and ability discrimination."
    },
    {
      "title": "Outsmarting AI in the classroom - ASU News",
      "url": "https://news.asu.edu/b/20260421-outsmarting-ai-classroom",
      "date": "2026-04-21",
      "type": "case-study",
      "added": "2026-04-24",
      "superseded_by": null,
      "window": null,
      "explanation": "GAMED.AI interactive assessment generation system deployed in university NLP course: students showed stronger in-class exam performance and higher engagement; games generated in <1 minute at <$1 per instance, illustrating emerging interactive assessment capability."
    },
    {
      "title": "PrepAI | AI-Assisted Assessment Platform for Educators",
      "url": "https://www.prepai.io/us/",
      "date": "2026-04-20",
      "type": "case-study",
      "added": "2026-04-24",
      "superseded_by": null,
      "window": null,
      "explanation": "Named institutional deployment at NYC's leading STEM institute: professors saved 30 hours/week on question paper preparation with 20% operational cost reduction; demonstrates formative assessment productivity gains in higher education."
    },
    {
      "title": "CoSN 2026: How K–12 Districts Are Tackling Responsible AI Adoption",
      "url": "https://edtechmagazine.com/k12/article/2026/04/cosn-2026-how-k-12-districts-are-tackling-responsible-ai-adoption",
      "date": "2026-04-20",
      "type": "opinion",
      "added": "2026-04-24",
      "superseded_by": null,
      "window": null,
      "explanation": "K-12 districts establishing explicit governance frameworks for AI in assessment: Alexandria City Schools restricts AI role in high-stakes contexts; Niles Township implements red/yellow/green rubric visible in LMS; signals institutional assessment governance maturation."
    },
    {
      "title": "Generation of Kazakhstan's Unified National Testing Variants Using AI",
      "url": "https://www.frontiersin.org/journals/big-data/articles/10.3389/fdata.2026.1772101/full",
      "date": "2026-04-17",
      "type": "research-paper",
      "added": "2026-04-24",
      "superseded_by": null,
      "window": null,
      "explanation": "Peer-reviewed hybrid AI/expert validation study of automatic item generation in Kazakhstan's national testing system: 97.5% of 200 AI-generated mathematics items accepted after expert review, demonstrating practical production model for high-stakes deployment with governance."
    },
    {
      "title": "How testing programs use AI to scale test content responsibly",
      "url": "https://www.psiexams.com/knowledge-hub/how-testing-programs-are-using-ai-to-scale-test-content-responsibly/",
      "date": "2026-04-17",
      "type": "product-ga",
      "added": "2026-04-24",
      "superseded_by": null,
      "window": null,
      "explanation": "Major standardized testing vendor (PSI/ETS) offers structured AI test development product with 8-step playbook, SME review integration, and psychometric rigor guidance; signals ecosystem maturity and responsible scaling approach."
    },
    {
      "title": "Hidden risks in classroom AI: Bias, errors, and opaque systems",
      "url": "https://www.devdiscourse.com/article/technology/3871791-hidden-risks-in-classroom-ai-bias-errors-and-opaque-systems",
      "date": "2026-04-14",
      "type": "research-paper",
      "added": "2026-04-24",
      "superseded_by": null,
      "window": null,
      "explanation": "Academic evaluation of 20 educational AI tools including quiz generators: 80% failed to explain generative mechanisms, 0 disclosed training datasets, only 1 provided source attribution; documents systemic transparency gaps undermining informed deployment."
    },
    {
      "title": "AI That Does More Than Write: Instant Surveys, Quizzes, and Team Structures",
      "url": "https://www.mangoapps.com/articles/ai-instant-surveys-quizzes-and-team-structures",
      "date": "2026-04-09",
      "type": "product-ga",
      "added": "2026-04-10",
      "superseded_by": null,
      "window": null,
      "explanation": "MangoApps 2026 Winter Release added AI quiz generation with document-to-quiz feature, signaling enterprise platform maturity and shift from manual to automated assessment authoring workflows."
    },
    {
      "title": "AI in Education 2026: The $32 Billion Market and What Teachers Actually Think",
      "url": "https://www.aimagicx.com/blog/ai-education-teachers-perspective-classroom-2026",
      "date": "2026-04-08",
      "type": "adoption-metric",
      "added": "2026-04-10",
      "superseded_by": null,
      "window": null,
      "explanation": "RAND survey of 4,200 K-12 teachers finds only 38% rate AI assessment questions for higher-order thinking as good/excellent; 42% say they need significant editing and 20% rate them not useful, documenting quality limitations for complex assessment."
    },
    {
      "title": "AssessPrep: AI-Powered Assessment Platform for Schools",
      "url": "https://www.assessprep.com",
      "date": "2026-04-07",
      "type": "product-ga",
      "added": "2026-04-10",
      "superseded_by": null,
      "window": null,
      "explanation": "AssessPrep deployed across 800+ schools in 85+ countries, processing 5M+ student submissions with 500K AI-generated questions; reports 92% outcome improvement and 2-hour time savings per assessment, validating production-scale institutional adoption."
    },
    {
      "title": "Hidden pitfalls in AI-generated MCQs: A call for caution",
      "url": "https://medicine.nus.edu.sg/taps/issues/hidden-pitfalls-in-ai-generated-mcqs-a-call-for-caution/",
      "date": "2026-04-07",
      "type": "research-paper",
      "added": "2026-04-10",
      "superseded_by": null,
      "window": null,
      "explanation": "Peer-reviewed research documenting systematic classical MCQ design flaws in AI-generated items, including weak distractors and pedagogical validity concerns; validates quality assurance as a critical adoption barrier."
    },
    {
      "title": "Designing Assessments GenAI Cannot Ghost-Write",
      "url": "https://marvinstarominskiuehara.substack.com/p/designing-assessments-genai-cannot",
      "date": "2026-04-06",
      "type": "opinion",
      "added": "2026-04-10",
      "superseded_by": null,
      "window": null,
      "explanation": "Practitioner analysis grounded in Kofinas et al. (2025) research showing AI-generated assessments are indistinguishable from human work; reveals systemic integrity risk affecting fair evaluation and establishing need for performative assessment design."
    },
    {
      "title": "A teacher's guide to designing better quizzes with AI - SchoolAI",
      "url": "https://schoolai.com/blog/a-teacher-s-guide-to-designing-better-quizzes-with-ai",
      "date": "2026-04-02",
      "type": "adoption-metric",
      "added": "2026-04-10",
      "superseded_by": null,
      "window": null,
      "explanation": "SchoolAI deployment shows 500K personalized learning sessions in six months, indicating rapid adoption of AI question generation at scale with Bloom's Taxonomy-aligned difficulty calibration."
    },
    {
      "title": "The Strategic Move to New Quizzes: What We Covered in Session 1",
      "url": "https://community.instructure.com/en/discussion/665703/the-strategic-move-to-new-quizzes-what-we-covered-in-session-1",
      "date": "2026-04-01",
      "type": "adoption-metric",
      "added": "2026-04-10",
      "superseded_by": null,
      "window": null,
      "explanation": "Canvas New Quizzes shows 78% institutional penetration with 32M quizzes created in 2025 and 172M student submissions; IgniteAI integration signals platform-level normalization of AI question authoring in major LMS."
    },
    {
      "title": "From PDFs to Practice Questions: Are AI-Generated Q-Banks Actually Trustworthy?",
      "url": "https://www.iatrox.com/blog/pdfs-to-practice-questions-are-ai-generated-question-banks-trustworthy-2026",
      "date": "2026-03-28",
      "type": "opinion",
      "added": "2026-04-10",
      "superseded_by": null,
      "window": null,
      "explanation": "Expert critical assessment identifies architectural risks in medical education: fact extraction errors, distractor quality flaws, curriculum mismatches, and hallucination risks; establishes high-stakes assessment context where question generation remains dangerous without human review."
    },
    {
      "title": "How Khan Academy Optimizes AI Tutoring with Experimentation",
      "url": "https://blog.growthbook.io/how-khan-academy-optimizes-ai-tutoring-with-experimentation/",
      "date": "2026-03-22",
      "type": "case-study",
      "added": "2026-03-27",
      "superseded_by": null,
      "window": null,
      "explanation": "Khan Academy's production A/B testing framework for Khanmigo shows 64 completed experiments validating quiz generation features; demonstrates data-driven maturity and confidence in iterative quality improvement at scale."
    },
    {
      "title": "Survey: How Should Universities Prepare for the AI Era?",
      "url": "https://observatory.tec.mx/edu-news/study-how-should-universities-prepare-for-the-ai-era/",
      "date": "2026-03-20",
      "type": "adoption-metric",
      "added": "2026-03-27",
      "superseded_by": null,
      "window": null,
      "explanation": "Tecnológico de Monterrey survey of 29 Latin American universities shows 76% of teachers use AI to create teaching materials (highest use case); 92% student adoption; 48% believe task redesign necessary to preserve learning outcomes."
    },
    {
      "title": "AI-Assisted Model for Generating Multiple-Choice Questions",
      "url": "https://www.scribd.com/document/1005130481/%E8%AF%B4%E8%A6%81%E7%9C%8B%E7%9A%84",
      "date": "2026-03-17",
      "type": "research-paper",
      "added": "2026-03-27",
      "superseded_by": null,
      "window": null,
      "explanation": "University of Jyväskylä study on human-AI co-creation model finds 50% of MCQs acceptable without editing; emphasizes pedagogical expertise integration with AI tools and addresses governance through distributed prompts and human revision cycles."
    },
    {
      "title": "Uncovering the positive impact of practice study questions on how students learn and study",
      "url": "https://www.ecampusnews.com/teaching-learning/2026/03/16/uncovering-the-positive-impact-of-practice-study-questions-on-how-students-learn-and-study/",
      "date": "2026-03-16",
      "type": "case-study",
      "added": "2026-03-27",
      "superseded_by": null,
      "window": null,
      "explanation": "VitalSource deployment study (200+ undergraduates) shows AI-generated questions with distributed practice yield ~2% exam score gains; at 25th percentile, C− to C grade improvement, validating classroom learning impact."
    },
    {
      "title": "Study finds ChatGPT answers inaccurate and inconsistent — Washington State University",
      "url": "https://wcti12.com/news/nation-world/study-finds-chatgpt-answers-inaccurate-and-inconsistent-washington-state-university-says-ai-articicila-intelligence-work-automated-cheating-layoffs-openai-technology-tests-college-school-prompt-accuracy",
      "date": "2026-03-16",
      "type": "research-paper",
      "added": "2026-03-27",
      "superseded_by": null,
      "window": null,
      "explanation": "WSU study testing ChatGPT accuracy on true/false questions finds only 60% above-chance performance (2025); only 16.4% accuracy identifying false statements; demonstrates consistency and comprehension limitations limiting assessment reliability."
    },
    {
      "title": "Pushing the boundaries of generative AI: multiple-choice question generation and assessment performance within medical education",
      "url": "https://dergipark.org.tr/en/pub/jhsm/article/1842373",
      "date": "2026-03-12",
      "type": "research-paper",
      "added": "2026-03-27",
      "superseded_by": null,
      "window": null,
      "explanation": "Peer-reviewed study comparing Gemini and Copilot MCQ generation shows both tools achieving high inter-rater agreement on Bloom's taxonomy and learning outcome alignment; documents equivalent quality to human evaluation in specialized medical assessment."
    },
    {
      "title": "AI in Assessment: New Industry Survey Highlights Opportunities, Risks and the Road Ahead",
      "url": "https://www.e-assessment.com/news/press-release-news/ai-in-assessment-new-industry-survey-highlights-opportunities-risks-and-the-road-ahead",
      "date": "2026-03-12",
      "type": "industry-report",
      "added": "2026-03-27",
      "superseded_by": null,
      "window": null,
      "explanation": "e-Assessment Association survey identifies item generation as most frequently used AI application in assessment across organizations, educators, and vendors; signals leading adoption use case with persistent quality and integrity concerns."
    },
    {
      "title": "AI in HE: International study finds high use, low support",
      "url": "https://www.universityworldnews.com/post.php?story=20260306064259345",
      "date": "2026-03-06",
      "type": "adoption-metric",
      "added": "2026-03-27",
      "superseded_by": null,
      "window": null,
      "explanation": "Coursera survey of 4,200 educators across 5 countries shows 28% of faculty use AI to draft exams; 95%+ adopt AI tools generally; only 26% confident detecting AI-generated content, documenting widespread adoption with governance gaps."
    },
    {
      "title": "Assessing the Utility of AI Versus Human-Created MCQs in Pediatric...",
      "url": "https://pmc.ncbi.nlm.nih.gov/articles/PMC12957612/",
      "date": "2026-03-03",
      "type": "research-paper",
      "added": "2026-03-27",
      "superseded_by": null,
      "window": null,
      "explanation": "Peer-reviewed empirical study finds AI-generated pediatric MCQs show significantly lower discrimination indices (0.19 vs 0.29) and higher proportion outside acceptable difficulty range (56% vs 32%) compared to human-authored questions, documenting persistent quality limitations."
    },
    {
      "title": "The Last Decade of Automatic Question Generation",
      "url": "https://journals-sol.sbc.org.br/index.php/rbie/article/view/6093",
      "date": "2026-02-27",
      "type": "research-paper",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2026-02",
      "explanation": "Systematic literature review of 103 AQG studies identifying clear trend toward Transformer-based models; documents critical gap in studies on educator acceptance and lack of standardized evaluation metrics."
    },
    {
      "title": "TurinQ: AI Quiz Maker & AI Study Platform",
      "url": "https://turinq.com",
      "date": "2026-02-26",
      "type": "product-ga",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2026-02",
      "explanation": "Commercial AI quiz platform reports 50,000+ learners and educators with multi-format question generation (6+ types), AI grading for open-ended questions, and claimed time reduction in exam preparation."
    },
    {
      "title": "AI Quiz & Assessment Generator Tools",
      "url": "https://brittanywashburn.com/2026/02/ai-quiz-assessment-generator-tools/",
      "date": "2026-02-17",
      "type": "tutorial",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2026-02",
      "explanation": "Educator tutorial documenting 8+ AI quiz generator platforms (Quizizz, Kahoot, Conker, ProProfs, etc.); addresses benefits, features, and adoption barriers including data privacy and bias concerns."
    },
    {
      "title": "Khanmigo To Be Deployed in Classes in Spring Term",
      "url": "https://phillipian.net/2026/02/13/khanmigo-to-be-deployed-in-classes-in-spring-term/",
      "date": "2026-02-13",
      "type": "case-study",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2026-02",
      "explanation": "Phillips Academy Andover deploying Khanmigo for AI-powered quiz generation in spring term; article documents student skepticism about AI effectiveness, revealing adoption friction despite institutional commitment."
    },
    {
      "title": "92% of Students and 79% of Faculty Actively Engaging with AI",
      "url": "https://www.digitaleducationcouncil.com/post/92-of-students-and-79-of-faculty-actively-engaging-with-ai-findings-from-ai-in-higher-education-latam-survey-2026",
      "date": "2026-02-06",
      "type": "adoption-metric",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2026-02",
      "explanation": "AI in Higher Education LATAM Survey (30,000+ responses from 29 institutions) shows 92% student and 79% faculty AI engagement; 50% of students support AI-assisted feedback but only 19% of faculty use it; documents regional adoption momentum and integrity concerns."
    },
    {
      "title": "2026 Generative AI Pedagogies and Technologies Survey Data",
      "url": "https://researchdata.edu.au/2026-generative-ai-survey-data/3994019",
      "date": "2026-02-05",
      "type": "adoption-metric",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2026-02",
      "explanation": "Open dataset from Macquarie University surveying educator adoption of AI across teaching, learning, planning, and assessment; documents current usage patterns and adoption frequency of question generation tools."
    },
    {
      "title": "Instant AI Quiz: Generate MCQs From Your Notes | CogniGuide",
      "url": "https://www.cogniguide.app/quizzes/multiple-choice-questions-to-ask",
      "date": "2026-01-30",
      "type": "product-ga",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2026-01",
      "explanation": "Product launch enables instant MCQ generation from documents with option-by-option explanations and export to multiple formats (Moodle, QTI 2.1); signals continued ecosystem expansion."
    },
    {
      "title": "Eklavvya AI Question Paper Generation",
      "url": "https://www.eklavvya.com/blog/ai-in-education/",
      "date": "2026-01-29",
      "type": "product-ga",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2026-01",
      "explanation": "Vendor reports 99% reduction in question paper creation time (8-10 hours to 5 minutes), 95% reduction in paper leakage incidents, and 70% improvement in educator productivity; demonstrates institutional scaling."
    },
    {
      "title": "10 Best Practices for Creating Effective AI Quizzes - Estha",
      "url": "https://estha.ai/blog/10-best-practices-for-creating-effective-ai-quizzes-that-drive-engagement/",
      "date": "2026-01-28",
      "type": "tutorial",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2026-01",
      "explanation": "Adoption guidance documents 40-60% higher quiz completion rates with AI quizzes; organizations report 60%+ mobile quiz interactions; provides practitioner perspective on effective deployment patterns."
    },
    {
      "title": "OECD Digital Education Outlook 2026",
      "url": "https://www.policyedge.in/p/oecd-digital-education-outlook-2026",
      "date": "2026-01-19",
      "type": "industry-report",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2026-01",
      "explanation": "OECD analysis identifies AI 'item factories' generating exam questions at 10x speed and lower cost; documents 'crutch effect' risk where AI-assisted practice improves scores but reduces independent performance when removed."
    },
    {
      "title": "EduQuest: Hybrid AI-Powered Intelligent Quiz Generation",
      "url": "https://www.ijraset.com/research-paper/eduquest-a-hybrid-ai-powered-intelligent-quiz-generation",
      "date": "2026-01-17",
      "type": "research-paper",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2026-01",
      "explanation": "Research platform achieves 82% difficulty classification accuracy and 78% time savings for educators; 71% higher student engagement; demonstrates cost-efficient hybrid approach reducing reliance on commercial AI APIs."
    },
    {
      "title": "Quality of Human Expert vs Large Language Model-Generated Multiple-Choice Questions in the Field of Mechanical Ventilation",
      "url": "https://pubmed.ncbi.nlm.nih.gov/40684906/",
      "date": "2025-12-18",
      "type": "research-paper",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2025-Q4",
      "explanation": "Peer-reviewed Chest journal study with blinded expert evaluation finds AI-generated MCQs (ChatGPT-o1) statistically noninferior to human expert questions in medical education, with experts unable to differentiate provenance."
    },
    {
      "title": "AI Quiz Generator Traffic Rankings - June 2025",
      "url": "https://creati.ai/ranking/june-2025/categories/ai-tools/education-translation/ai-quiz-generator/",
      "date": "2025-12-11",
      "type": "adoption-metric",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2025-Q4",
      "explanation": "Traffic analytics show Quizgecko with 854,600 monthly visits and other AI quiz generators with significant user bases, documenting sustained market traction and consumer adoption of question generation tools."
    },
    {
      "title": "EXPLORING MULTI-LLM QUESTION GENERATION AND AI-BASED QUALITY CONTROL",
      "url": "https://library.iated.org/view/DANNECKER2025EXP",
      "date": "2025-11-12",
      "type": "research-paper",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2025-Q4",
      "explanation": "ICERI2025 conference paper on automated question bank creation for certification prep using multiple LLMs with AI quality control demonstrates time savings vs. manual creation and feasible scalable approach for specialized assessment domains."
    },
    {
      "title": "Accuracy of AI-Generated Multiple-Choice Questions in the Field of Mechanical Ventilation",
      "url": "https://discovery.researcher.life/article/assessment-in-the-advent-of-ai-examining-the-vulnerabilities-of-undergraduate-exams/a6b8c44c16b433eda92c6f864fcdcf76",
      "date": "2025-10-29",
      "type": "research-paper",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2025-Q4",
      "explanation": "Criminal justice education study evaluates ChatGPT 3.5 on 500 undergraduate exam questions, documenting 80% accuracy but with significant consistency limitations across test accounts, revealing both capability and integrity vulnerabilities."
    },
    {
      "title": "Assessing the Quality of AI-Generated Exams: A Large-Scale Field Study",
      "url": "https://chatpaper.com/paper/179908",
      "date": "2025-10-28",
      "type": "research-paper",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2025-Q4",
      "explanation": "Large-scale field study across 91 classes with 1,700 students shows AI-generated questions perform comparably to expert-created questions based on item response theory analysis, providing strong empirical validation of question quality."
    },
    {
      "title": "A Conversation with Sal Khan",
      "url": "https://www.aasa.org/resources/resource/a-conversation-with-sal-khan",
      "date": "2025-10-01",
      "type": "case-study",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2025-Q4",
      "explanation": "Khan Academy's Khanmigo AI tutor reached 1 million U.S. students in 2025 (up from 700K prior year), confirming rapid scaling of integrated question generation and teaching assistant capabilities in production educational environments."
    },
    {
      "title": "AI has turned college exams into a 'wicked problem' with no obvious fix, researchers warn",
      "url": "https://www.businessinsider.com/ai-college-exams-wicked-problem-no-clear-fix-researchers-warn-2025-9",
      "date": "2025-09-18",
      "type": "news-coverage",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2025-Q3",
      "explanation": "Academic research on AI's disruption to exam design; university unit chairs report impossible trade-offs between AI-proof and creative assessments, documenting institutional governance barriers and unresolved design challenges limiting AI-integrated assessment deployment."
    },
    {
      "title": "Pilot Recap: Khanmigo Teacher Tools",
      "url": "https://teach.its.uiowa.edu/news/2025/08/pilot-recap-khanmigo-teacher-tools",
      "date": "2025-08-20",
      "type": "case-study",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2025-Q3",
      "explanation": "University of Iowa pilot of Khanmigo found usage under once per week with no significant teaching impact; lack of Canvas LMS integration required manual copy-pasting, documenting deployment friction limiting institutional adoption."
    },
    {
      "title": "Have You Considered AI in Your Classroom? A Khanmigo Pilot Story",
      "url": "https://michiganvirtual.org/blog/have-you-considered-ai-in-your-classroom-a-khanmigo-pilot-story/",
      "date": "2025-08-14",
      "type": "case-study",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2025-Q3",
      "explanation": "Michigan Virtual two-phase Khanmigo pilot across K-12 (1,700+ participants) found teacher-facing tools 'surprisingly helpful' for brainstorming but highlighted need for intentional support; demonstrates real-world deployment with mixed signals on utility."
    },
    {
      "title": "AI Accuracy and Limitations - Duke Center for Teaching and Learning",
      "url": "https://ctl.duke.edu/caradite/ai-student-survey/ai-accuracy-and-limitations/",
      "date": "2025-08-14",
      "type": "adoption-metric",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2025-Q3",
      "explanation": "Duke survey shows 75% of students believe AI provides inaccurate answers and 90% expect AI to be transparent about limitations; documents widespread student skepticism about AI reliability, constraining educational adoption momentum."
    },
    {
      "title": "Can an AI-Powered Tutor Produce Meaningful Results?",
      "url": "https://www.edweek.org/technology/opinion-can-an-ai-powered-tutor-produce-meaningful-results/2025/07",
      "date": "2025-07-29",
      "type": "opinion",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2025-Q3",
      "explanation": "Khan Academy CLO reports Khanmigo user growth to 700K (2024-25) across 380+ districts, but expresses concern that teachers overuse AI for MCQ generation which 'rarely encourage' deep engagement; vendor perspective on adoption patterns and limitations."
    },
    {
      "title": "A psychometric comparison of question generation methods",
      "url": "https://www.dirjournal.org/articles/artificial-intelligence-in-radiology-examinations-a-psychometric-comparison-of-question-generation-methods/doi/dir.2025.253407",
      "date": "2025-07-21",
      "type": "research-paper",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2025-Q3",
      "explanation": "Peer-reviewed radiology study comparing AI-generated vs. faculty-written MCQs shows both ChatGPT-4o and template-based AIG produced questions with acceptable psychometric properties, validating AI question quality in specialized medical assessment."
    },
    {
      "title": "China Temporarily Disables AI Tools to Prevent Cheating During National College Entrance Exams",
      "url": "https://theoutpost.ai/news-story/chinese-tech-giants-disable-ai-tools-during-national-college-entrance-exam-to-prevent-cheating-16350/",
      "date": "2025-06-12",
      "type": "news-coverage",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2025-Q2",
      "explanation": "Chinese tech companies (Alibaba, ByteDance, Tencent) disabled chatbot features during national gaokao exams to prevent cheating, signaling integrity and security concerns limiting AI adoption in high-stakes assessment contexts."
    },
    {
      "title": "ConQuer: A Framework for Concept-Based Quiz Generation",
      "url": "https://aclanthology.org/2025.naacl-srw.9/",
      "date": "2025-04-04",
      "type": "research-paper",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2025-Q2",
      "explanation": "NAACL 2025 peer-reviewed research presents ConQuer framework for concept-based quiz generation showing 4.8% improvement in evaluation scores and 77.52% win rate over baselines, advancing technical quality of AI-generated questions."
    },
    {
      "title": "Why AI Fails: The Untold Truths Behind 2025's Biggest Tech Letdowns",
      "url": "https://www.techfunnel.com/information-technology/why-ai-fails-2025-lessons/",
      "date": "2025-03-30",
      "type": "opinion",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2025-Q1",
      "explanation": "Analysis showing 42% of businesses scrapped majority of AI initiatives (up from 17% six months prior); common failure modes include poor data quality, biased datasets, and low adoption due to change management barriers."
    },
    {
      "title": "AI search engines confidently wrong, citing sources - Columbia study",
      "url": "https://fortune.com/2025/03/18/ai-search-engines-confidently-wrong-citing-sources-columbia-study/",
      "date": "2025-03-18",
      "type": "news-coverage",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2025-Q1",
      "explanation": "Columbia University study tested eight AI systems and found >60% of answers to news questions were incorrect; error types included fabricated links and altered quotes, indicating hallucination risks in AI-generated content systems."
    },
    {
      "title": "Quality assurance and validity of AI-generated single best answer questions",
      "url": "https://pmc.ncbi.nlm.nih.gov/articles/PMC11854382/",
      "date": "2025-02-25",
      "type": "research-paper",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2025-Q1",
      "explanation": "Peer-reviewed BMC Medical Education study evaluates quality and validity of AI-generated single best answer (SBA) questions for medical education, providing empirical validation of question generation quality in clinical assessment."
    },
    {
      "title": "BBC research finds 'significant issues' over accuracy of AI responses to news questions",
      "url": "https://www.research-live.com/article/news/bbc-research-finds-significant-issues-over-accuracy-of-ai-responses-to-news-questions/id/5135763",
      "date": "2025-02-11",
      "type": "research-paper",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2025-Q1",
      "explanation": "BBC study found 51% of AI responses to news questions had significant factual inaccuracies, including incorrect dates and misrepresented information; highlights systemic accuracy limitations in AI-generated content relevant to question quality."
    },
    {
      "title": "Why Most AI Pilots Fail (And What Actually Works)",
      "url": "https://www.botdojo.com/blog/why-most-ai-pilots-fail-and-what-actually-works",
      "date": "2025-01-01",
      "type": "opinion",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2025-Q1",
      "explanation": "MIT study of 300+ AI initiatives found 95% of organizations got zero return; only 5% of custom AI pilots reached production despite $30-40B investment, documenting structural adoption barriers limiting question generation tool deployment."
    },
    {
      "title": "AI's Potential to Transform Assessments",
      "url": "https://www.abms.org/newsroom/ais-potential-to-transform-assessments/",
      "date": "2024-12-19",
      "type": "industry-report",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2024-Q4",
      "explanation": "American Board of Medical Specialties reports most ABMS boards focusing on AI for question development; American Board of Anesthesiology piloting AI question generation for longitudinal exams but paused due to copyright concerns, showing institutional adoption with governance barriers."
    },
    {
      "title": "Khan Academy Efficacy Results, November 2024",
      "url": "https://blog.khanacademy.org/khan-academy-efficacy-results-november-2024/",
      "date": "2024-12-06",
      "type": "case-study",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2024-Q4",
      "explanation": "Khan Academy efficacy study of ~350K students shows 30+ min/week usage associated with ~20% greater learning gains (effect size 0.36) on MAP Growth Assessment, demonstrating platform scale and measurable impact; broader context for integrated question generation capability."
    },
    {
      "title": "GenAI not production-ready",
      "url": "https://aitransform.net/blog/23549-genai-not-production-ready",
      "date": "2024-11-18",
      "type": "industry-report",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2024-Q4",
      "explanation": "Economist Impact survey of 1,100 executives: 85% of enterprises use/test GenAI but only 37% believe applications production-ready; quality (37%) and governance (33%) cited as top barriers, directly applicable to question generation deployment challenges."
    },
    {
      "title": "Survey shows skyrocketing AI use in education",
      "url": "https://www.eschoolnews.com/digital-learning/2024/11/07/survey-shows-skyrocketing-ai-use-in-education/",
      "date": "2024-11-07",
      "type": "adoption-metric",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2024-Q4",
      "explanation": "HMH 2024 Educator Confidence Report: 50% of educators use GenAI (5x increase YoY); 76% find it valuable; assessment creation ranks among top 5 use cases, signaling mainstream adoption momentum despite persisting plagiarism and accuracy concerns."
    },
    {
      "title": "AI-generated exam image draws student complaints",
      "url": "https://www.aiaaic.org/aiaaic-repository/ai-algorithmic-and-automation-incidents/ai-generated-exam-image-draws-student-complaints",
      "date": "2024-10-16",
      "type": "news-coverage",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2024-Q4",
      "explanation": "NSW Education Standards Authority used AI-generated image in October 2024 HSC English exam; student complaints about authenticity and suitability reveal real-world deployment of AI content in high-stakes assessment and quality acceptance barriers."
    },
    {
      "title": "Conker AI-Effortless AI-powered quiz creation for educators and classrooms",
      "url": "https://toolful.ai/t/conker-ai",
      "date": "2024-10-01",
      "type": "product-ga",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2024-Q4",
      "explanation": "Conker AI product directory updated metrics (Aug-Oct 2024): 104,721 monthly visits, top users from Spain, Malaysia, India, Vietnam, and U.S., confirming sustained adoption and geographic reach of AI quiz generation platform."
    },
    {
      "title": "AI-Teaching Assistant Khanmigo Now Available in Canvas LMS",
      "url": "https://blog.khanacademy.org/ai-teaching-assistant-khanmigo-now-available-in-canvas-trusted-ai-to-streamline-prep-right-inside-your-lms/",
      "date": "2024-09-30",
      "type": "product-ga",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2024-Q3",
      "explanation": "Khan Academy integrates Khanmigo Teacher Tools into Canvas LMS for U.S. educators with 20+ teaching activities; direct LMS integration confirms ecosystem adoption and removes friction for classroom deployment."
    },
    {
      "title": "The Future of Learning in the Age of Generative AI: Automated Question Generation and Assessment",
      "url": "https://github.com/cognitivetech/llm-research-summaries/blob/main/education/Automated-Question-Generation-and-Assessment_2410.09576.md",
      "date": "2024-09-13",
      "type": "research-paper",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2024-Q3",
      "explanation": "arXiv preprint review covering LLM methodologies for question generation and assessment; documents capabilities (contextual relevance, higher-order thinking) and persistent challenges (quality, accuracy, ethical implications)."
    },
    {
      "title": "Generative AI hype is ending – and now the technology might actually become useful",
      "url": "https://unisa.edu.au/connect/enterprise-magazine/articles/2024/generative-ai-hype-is-ending--and-now-the-technology-might-actually-become-useful/",
      "date": "2024-09-03",
      "type": "opinion",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2024-Q3",
      "explanation": "Analysis citing RAND study showing 80% of AI projects fail vs. non-AI baselines; references specific deployment cancellations and Khan Academy's Khanmigo revealing correct answers despite guardrails, documenting adoption barriers."
    },
    {
      "title": "Khanmigo for Teachers: Your free AI-powered teaching tool",
      "url": "https://www.microsoft.com/en-us/education/blog/2024/08/khanmigo-for-teachers-your-free-ai-powered-teaching-tool/",
      "date": "2024-08-13",
      "type": "product-ga",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2024-Q3",
      "explanation": "Microsoft/Khan Academy announces free Khanmigo for Teachers across 49 countries with 25+ educator tools including quiz generation, signaling major vendor scaling and broad geographic accessibility."
    },
    {
      "title": "Current Evaluation Methods are a Bottleneck in Automatic Question Generation",
      "url": "https://proceedings.mlr.press/v257/gorgun24a.html",
      "date": "2024-08-09",
      "type": "research-paper",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2024-Q3",
      "explanation": "AAAI 2024 peer-reviewed paper identifies evaluation methods as the bottleneck limiting reliable deployment of automatic question generation systems in educational settings, highlighting a critical maturity gap."
    },
    {
      "title": "Everyone Is Judging AI by These Tests. But Experts Say They're Close to Meaningless",
      "url": "https://themarkup.org/artificial-intelligence/2024/07/17/everyone-is-judging-ai-by-these-tests-but-experts-say-theyre-close-to-meaningless",
      "date": "2024-07-17",
      "type": "opinion",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2024-Q3",
      "explanation": "Investigative critique by Carnegie Mellon and U Washington experts argues AI benchmarks (MMLU, MMMU) lack construct validity for high-stakes applications, highlighting evaluation limitations that undermine deployment confidence."
    },
    {
      "title": "A real-world test of artificial intelligence infiltration of a university examinations system: A 'Turing Test' case study",
      "url": "https://journals.plos.org/plosone/article?id=10.1371%2Fjournal.pone.0305354&rut=8c6d3170d818348a8ffa25414fb3a7b8b9f2a61938f7b3c3272b89a8a1be5e3d",
      "date": "2024-06-26",
      "type": "research-paper",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2024-Q2",
      "explanation": "University of Reading blind study found 94% of AI-written exam submissions undetected and graded half a boundary higher than real students, documenting severe assessment integrity risks from AI-generated answers."
    },
    {
      "title": "Questgen - AI Powered Quiz Generator",
      "url": "https://aipure.ai/de/products/questgen-ai",
      "date": "2024-06-17",
      "type": "product-ga",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2024-Q2",
      "explanation": "Questgen product demonstrates multi-format quiz generation (MCQ, fill-in-blank, true/false) from diverse inputs (text, PDFs, URLs) with export to standard formats, showing ecosystem tool proliferation and maturity."
    },
    {
      "title": "AI Generators for Teachers - Khan Academy Blog",
      "url": "https://blog.khanacademy.org/ai-generators-for-teachers/",
      "date": "2024-06-10",
      "type": "product-ga",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2024-Q2",
      "explanation": "Khan Academy launches free AI question generators for teachers integrated into its platform, enabling quiz and exam generation from course resources within minutes, indicating vendor expansion into question generation."
    },
    {
      "title": "Top 12 AI quiz generators (2026): Tested for accuracy",
      "url": "https://forms.app/en/blog/best-quiz-maker-tools",
      "date": "2024-05-21",
      "type": "industry-report",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2024-Q2",
      "explanation": "Comparative analysis of 12 AI quiz generation platforms documenting ecosystem maturity, feature comparison, and market breadth across multiple vendors and use cases."
    },
    {
      "title": "ABR",
      "url": "https://www.theabr.org/beam/from-the-board-of-governors-april-2024",
      "date": "2024-04-09",
      "type": "news-coverage",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2024-Q2",
      "explanation": "American Board of Radiology formally bans use of generative AI in exam content development, citing copyright, authorship, and integrity concerns; signals institutional resistance despite technical maturity."
    },
    {
      "title": "Using generative AI in item development for the driving theory test",
      "url": "https://www.pearsonvue.com/us/en/about/news/highlights/using-generative-ai-in-item-development.html",
      "date": "2024-03-28",
      "type": "case-study",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2024-Q1",
      "explanation": "Pearson VUE reports on 2023 research evaluating AI-generated items against human-written ones for driving theory tests, finding comparable quality on key metrics but noting cognitive level mismatches and duplicate content issues."
    },
    {
      "title": "AI Question Generation: The Risks and Alternatives",
      "url": "https://www.grademaker.com/news/ai-question-generation-the-risks-and-alternatives/",
      "date": "2024-03-07",
      "type": "opinion",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2024-Q1",
      "explanation": "AQA research director (Cesare Aloisi) identifies critical adoption barriers: unresolved IP issues, bias risks, unreliability, and ethical concerns; argues high-stakes testing deployment remains premature."
    },
    {
      "title": "Conker for AI powered quizzes and more",
      "url": "https://www.conker.ai",
      "date": "2024-02-01",
      "type": "product-ga",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2024-Q1",
      "explanation": "Conker product page reports 600,000+ quizzes created on its platform, demonstrating substantial real-world adoption and continued scale growth through early 2024."
    },
    {
      "title": "Math Multiple Choice Question Generation via Human-Large Language Model Collaboration",
      "url": "https://educationaldatamining.org/edm2024/proceedings/2024.EDM-posters.113/index.html",
      "date": "2024-01-01",
      "type": "research-paper",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2024-Q1",
      "explanation": "EDM 2024 pilot study with four math educators found GPT-4 generated valid stems (70%) but only 37% valid distractors/misconceptions, indicating LLM gaps in capturing student errors and misconceptions."
    },
    {
      "title": "Free AI Quiz Generator | Create Quizzes from PDF & Video",
      "url": "https://quizflex.ai",
      "date": "2024-01-01",
      "type": "product-ga",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2024-Q1",
      "explanation": "QuizFlex product page reports 10,000+ educators worldwide trust the platform with 50,000+ quizzes created, indicating broad educator adoption and perceived utility for assessment preparation."
    },
    {
      "title": "Towards AI-Assisted Multiple Choice Question Generation and Quality Evaluation at Scale: Aligning with Bloom's Taxonomy",
      "url": "https://www.semanticscholar.org/paper/Towards-AI-Assisted-Multiple-Choice-Question-and-at-Hwang-Challagundla/8c20862d07859a7707bb2f6fa2d8dc9b3aba19b7",
      "date": "2024-01-01",
      "type": "research-paper",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2024-Q1",
      "explanation": "Chemistry/biology research demonstrates GPT-3.5 can generate higher-order thinking questions aligned with Bloom's Taxonomy, showing AI capability for cognitively complex assessments."
    },
    {
      "title": "Conker AI - AI Tools for Education - Research Guides",
      "url": "https://guides.libraries.uc.edu/ai-education/ca",
      "date": "2023-12-08",
      "type": "tutorial",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2023-H2",
      "explanation": "University library guide to Conker AI quiz generation tool documents practical deployment and limitations; emphasizes need for human supervision due to inaccuracy risks in AI-generated questions."
    },
    {
      "title": "U.S. Students Slower Than Others Globally to Adopt Generative AI, Survey Reveals",
      "url": "https://campustechnology.com/Articles/2023/11/30/US-Students-Slower-Than-Others-Globally-to-Adopt-Generative-AI-Survey-Reveals.aspx",
      "date": "2023-11-30",
      "type": "adoption-metric",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2023-H2",
      "explanation": "Anthology survey of 2,728 students across 11 countries shows only 10% of U.S. university students are frequent AI users vs. 23% globally; highlights adoption barriers for AI tools in education."
    },
    {
      "title": "Harnessing GPT-4 so that all students benefit. A nonprofit approach for equal access.",
      "url": "https://blog.khanacademy.org/harnessing-ai-so-that-all-students-benefit-a-nonprofit-approach-for-equal-access/",
      "date": "2023-11-17",
      "type": "case-study",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2023-H2",
      "explanation": "Khan Academy announces limited pilot of Khanmigo AI tutor with quiz/question generation features for thousands of users; acknowledges limitations like math errors but demonstrates real-world classroom integration."
    },
    {
      "title": "AI-BASED QUIZ SYSTEM FOR PERSONALISED LEARNING",
      "url": "https://library.iated.org/view/WANG2023AIB",
      "date": "2023-11-15",
      "type": "research-paper",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2023-H2",
      "explanation": "Conference paper presents iQS, an AI-assisted quiz system tested in Moodle with positive feedback from students and teachers; demonstrates institutional deployment and real LMS integration."
    },
    {
      "title": "Evaluating undergraduate mathematics examinations in...",
      "url": "https://arxiv.org/html/2509.13359v3",
      "date": "2023-11-13",
      "type": "research-paper",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2023-H2",
      "explanation": "Peer-reviewed study demonstrates GenAI achieving first-class degree performance on undergraduate mathematics exams, validating AI capability to understand and answer exam-level questions at high proficiency."
    },
    {
      "title": "Instructure Survey Shows Most Teachers, Students Are Optimistic...",
      "url": "https://www.instructure.com/press-release/instructure-survey-shows-most-teachers-students-are-optimistic-about-ai-classroom",
      "date": "2023-09-21",
      "type": "adoption-metric",
      "added": "2026-03-17",
      "superseded_by": null,
      "window": "2023-H2",
      "explanation": "Canvas LMS survey of 1,000+ respondents shows 28.8% of teachers cite AI as useful for question development; 54.5% express positive sentiment on AI in education overall."
    },
    {
      "title": "Quizify: AI Quiz generator powered by Vercel AI SDK",
      "url": "https://github.com/rotimi-best/quizify",
      "date": "2023-06-20",
      "type": "significant-repo",
      "added": null,
      "superseded_by": null,
      "window": "2023-H1",
      "explanation": null
    },
    {
      "title": "Conker.ai: AI quiz generator for educators",
      "url": "https://jitkapourova.cz/2023/06/19/conker/",
      "date": "2023-06-19",
      "type": "news-coverage",
      "added": null,
      "superseded_by": null,
      "window": "2023-H1",
      "explanation": null
    },
    {
      "title": "AI-Quiz-Generator: GPT-powered multiple choice quiz generator",
      "url": "https://github.com/quentin-mckay/AI-Quiz-Generator",
      "date": "2023-05-14",
      "type": "significant-repo",
      "added": null,
      "superseded_by": null,
      "window": "2023-H1",
      "explanation": null
    },
    {
      "title": "QwizLab: AI-powered quiz generator from notes",
      "url": "https://qwizlab.com",
      "date": "2023-01-01",
      "type": "product-ga",
      "added": null,
      "superseded_by": null,
      "window": "2023-H1",
      "explanation": null
    }
  ],
  "tierHistory": [
    {
      "tier": "research",
      "from": "2023-01-01",
      "to": "2023-01-01"
    },
    {
      "tier": "bleeding-edge",
      "from": "2023-01-01",
      "to": "2026-03-27"
    },
    {
      "tier": "leading-edge",
      "from": "2026-03-27",
      "to": null
    }
  ],
  "trendHistory": [
    {
      "trend": "steady",
      "blockerType": null,
      "from": "2026-09-26",
      "to": null
    }
  ],
  "description": "AI that generates assessment questions, quizzes, and examinations at specified difficulty levels and covering defined topics. Includes distractor generation and difficulty calibration; distinct from interview question generation in HR which targets hiring rather than education.",
  "overview": "AI-generated assessment questions have reached leading-edge maturity: the technical bar is cleared (peer-reviewed parity with expert items in specialized domains), vendor tooling is normalized into major platforms (78% of Canvas institutions use AI question authoring), and deployment at scale is underway (800+ schools, 500K+ questions in production). But formative adoption and high-stakes institutional assessment remain starkly bifurcated. Formative use (study aids, practice quizzes, low-stakes classroom review) has crossed into ubiquity—28-76% of educators adopt generative tools depending on context, with documented learning gains and established classroom workflows. High-stakes examination stays locked behind unresolved barriers: distractor quality flaws persist (weak distractors, design mismatches), consistency gaps limit reliability (lower discrimination indices on pediatric MCQs, inconsistency across LLM runs), domain-specific accuracy risks are documented (49.6% of medical chatbot responses problematic, neurology/dermatology gaps), and governance frameworks remain absent. Recent August 2026 evidence shows high-stakes deployment is accelerating despite unresolved governance: California's Bar Exam deployed 23 AI-generated scored questions (13.5% of exam) without disclosure or attorney review, affecting 85+ examinees and triggering AB 1651 (mandatory AI disclosure 60+ days pre-exam). Learning-outcome research (27,000 students, 30 months) reveals exam score collapse (-20%) despite homework improvement (+18%), signaling assessment validity risk when AI practice diverges from proctored assessment. September 2026 institutional response: MIT EECS redesigned course 6.036 after audit found 73-84% of submitted problem sets were AI-generated, replacing auto-graded problem sets with oral exams, studio sessions, and AI-augmented project work—$150K investment recognizing that traditional assessment became invalid in the AI era. Parallel validity research on 1,066 undergraduates shows failure rates jump from 2-6% to 18.4% when AI assistance is removed, confirming AI-assisted assessments mask competence gaps. The shift from bleeding-edge to leading-edge reflects proven capability and production scale; institutional integrity concerns and learning-outcome evidence are now forcing assessment redesign away from AI-generated items toward AI-resistant formats—the defining tension for high-stakes deployment.",
  "currentLandscape": "Platform-level normalization is now institutional-scale: 68% of U.S. public school districts (up from 42% two years prior) have formally adopted generative AI platforms; Khanmigo holds 22% K-12 market share, MagicSchool 14%, Google Gemini for Education 34% via Chromebook integration; OpenAI expanded ChatGPT for Teachers to 100,000+ educators across 55 new districts (September 2026), explicitly naming quiz question generation as primary training use case, bringing total educator deployment to 300,000+ across 30+ states. Canvas New Quizzes (78% of institutions, 32M quizzes created in 2025) integrates AI question authoring via IgniteAI; MangoApps released document-to-quiz generation in April 2026; specialized SaaS platforms (QuizMaker trusted by 10,000+ schools, QuizMagic, ConductExam, Edzo, PressPrimer, Quizify) address K-12, higher education, and self-hosted ecosystems; a June 2026 market survey documents 10 mature platforms supporting standards alignment, adaptive difficulty, multi-modal items, and LMS integration across K-12, higher education, corporate training, and professional certification contexts; enterprise platforms are embedding the capability as standard. Institutional deployments at scale are underway: AssessPrep operates across 800+ schools in 85+ countries with 4M+ assessments delivered and 500K AI-generated questions; SchoolAI shows 500K personalized learning sessions in six months with documented classroom gains (McNulty Academy students became 'more intentional with explanations'); Jordan School District (Utah) deployed AI conversational questioning across 82 teachers supporting 14,000 student interactions, documenting 28% critical thinking skill increase and doubled higher-level reasoning. Khan Academy's Khanmigo reached 1 million U.S. students and rolled out at Phillips Academy Andover; May 2026 announcement of new \"Assessments\" product signals transition to structured assessment design with psychometrics and norming. Pearson Study Prep deployed at scale (62,000+ students, Fall 2025) achieved 90% higher proficiency rates and 60% higher mastery with goal-setting, validating formative adoption with measurable learning gains. Independent adoption has reached beyond traditional education markets: Japan's AI Passport Quiz App generated 10 million uses in ~16 months (launched May 2024), demonstrating grassroots adoption of AI-generated assessment in professional certification contexts. Major testing vendor PSI (ETS subsidiary) demonstrates production-grade governance: 77.4% of AI-generated items meet psychometric thresholds at parity with human-authored items, using multi-agent AI validation with expert SME review. The U.S. Department of Education (IES) invested $3.6M over 3 years (2024-2027) in AI-enhanced scenario-based assessment authoring, signaling policy-level recognition of both opportunity and need. Practitioner workflows are now established: educators across K-12 use Gen AI tools (Conker, Quizizz, Formative, MagicSchool) in year-long classroom deployments generating pre-assessments and analyzing item performance, with the production workflow norm requiring teacher review of AI-generated questions before deployment. Speed and cost gains are proven: the OECD documents 10x reductions in question paper creation; educators report 60-80% time savings; teachers cite assessment creation as their top AI use case (76% in LATAM contexts); UK survey shows 76% of teachers use AI with quiz generation explicitly listed as common use case. But formative scaling masks persistent barriers to high-stakes deployment. A RAND survey of 4,200 K-12 teachers shows only 38% rate AI assessment questions for higher-order thinking as good/excellent; 42% need significant editing. Peer-reviewed research documents classical MCQ design flaws (weak distractors, lower discrimination indices on AI-generated items vs. human-authored); radiology education research (September 2026) compared faculty-written vs. ChatGPT-4o vs. template-based automatic item generation across 115 students, finding template-based methods achieved acceptable discrimination on all items versus ChatGPT-4o on 70%, confirming technical viability in specialized domains but template-based superiority. Medical accuracy research (June 2026) reveals systemic quality risks: BMJ Open study found 49.6% of medical chatbot responses problematic (19.6% highly problematic); Penn State study on medical questions shows ChatGPT-4o 84.6% accuracy but other models 50%, with domain-specific weakness in neurology and dermatology. Expert critical assessment identifies architectural risks specific to medical education: fact extraction errors, hallucination, curriculum mismatches that render AI-generated high-stakes content dangerous without human review. Fundamental reliability concerns emerged: Stanford AI Index (May 2026) documents 22-94% hallucination rates across 26 frontier models, with models overconfident precisely where wrong (hard-easy effect)—a critical failure mode for assessment where human supervisors need most reliable oversight. Learning-outcome validity crisis deepens: 30-month longitudinal research on 27,000 students (published September 2026) shows AI homework assistance raised homework scores 18% but cut exam performance 20%, revealing metacognitive laziness mechanism—students outsource thinking without developing retention. Validity analysis on 1,066 undergraduates documents failure rates jumping from 2-6% (AI-accessible take-home exam) to 18.4% (AI-restricted proctored exam), proving AI-assisted assessments mask competence gaps and misrepresent achievement. An integrity vulnerability is documented: large-scale study of 95,000+ students at 20 universities shows 37% use AI on assignments and 9% have used it to cheat, motivating urgent assessment redesign. Research reliability concerns were exposed when a widely-cited meta-analysis (262 peer-reviewed citations claiming \"large positive\" ChatGPT effects) was retracted for methodological discrepancies, revealing premature claims circulating in the field. Governance remains the binding constraint: copyright liability unresolved, evaluation standards unstandardized, integrity frameworks absent, and research validation gaps undermining institutional confidence. Assessment design experts argue the solution is not surveillance or detection but structural redesign toward process portfolios, in-class components, and oral defenses that make learning visible rather than attempting to lock questions against misuse.",
  "history": "- **2023-H1:** Early open-source projects (Quizify, AI-Quiz-Generator) and consumer tools (Conker.ai, QwizLab) demonstrated proof-of-concept that generative models could create quizzes and exams from text; institutional deployments remained minimal.\n- **2023-H2:** Khan Academy deployed Khanmigo with quiz generation at scale (1000s of users); research validated AI performance on degree-level mathematics exams; iQS deployed in university Moodle systems. Adoption surveys showed educators saw utility but lacked confidence; accuracy concerns and slow institutional trust remained primary barriers.\n- **2024-Q1:** Consumer adoption accelerated (Conker: 600k quizzes, QuizFlex: 10k+ educators, QuizGeniusAI: 300+ educators). Research identified specific gaps: EDM 2024 study found GPT-4 achieved 70% validity for question stems but only 37% for distractors/misconceptions. Pearson VUE's high-stakes testing research showed time-saving benefits but persistent issues with cognitive level calibration. Critical assessment from AQA highlighted unresolved IP, bias, and reliability concerns—signalling adoption barriers in formal assessment despite technical progress.\n- **2024-Q2:** Ecosystem maturation continued: Khan Academy expanded Khanmigo with free AI question generators for all teachers; forms.app's comparative analysis identified 12+ mature quiz generation platforms across the market. Questgen and other tools demonstrated multi-format generation capability. However, critical integrity research from University of Reading (June 2024) revealed AI-generated exam submissions achieved 94% undetection rate and grades 0.5 boundaries higher than real students—a stark demonstration of assessment integrity risk. Institutional resistance intensified: American Board of Radiology formally banned AI-generated exam content, citing copyright and integrity concerns. The tension sharpened: widespread consumer adoption and vendor expansion collided with accumulating evidence of both technical limitations (distractor quality, cognitive calibration) and integrity risks (undetectable submissions, grade inflation), creating a widening gap between bleeding-edge tooling and high-stakes institutional deployment readiness.\n- **2024-Q3:** Vendor consolidation and geographic scaling: Khan Academy made Khanmigo free to teachers in 49 countries via Microsoft partnership (Aug 2024), then integrated into Canvas LMS for U.S. educators (Sept 2024). Research highlighted evaluation as the key bottleneck: AAAI 2024 conference paper documented that question quality assessment methods limit full integration into education; arXiv review catalogued AI's capabilities (Bloom's-aligned questions) and gaps (ethical, accuracy, consistency). Project failure rates remained high (RAND: 80% of AI projects fail), and deployment risks intensified—documented guardrail failures in production systems, benchmark validity critiques, and evidence that 80%+ of enterprise AI projects abandoned. The practice reached a critical inflection: consumer adoption was evident and accelerating, but institutional deployment remained blocked by evaluation uncertainty, integrity concerns, and widespread deployment failures across AI systems generally.\n- **2024-Q4:** Adoption continued: HMH 2024 survey showed 50% of educators using GenAI tools (5x increase YoY) with assessment creation among top use cases; Khan Academy's efficacy study demonstrated ~350K student user base with measurable learning gains, confirming scale and real classroom impact. Vendor consolidation persisted: ABMS boards piloted AI question generation for medical certification (American Board of Anesthesiology), though paused due to copyright concerns. Real-world deployment risks materialized: NSW Education Authority's October 2024 HSC exam included AI-generated image, sparking student complaints about authenticity and suitability, illustrating quality acceptance barriers in high-stakes contexts. Critical finding emerged: only 37% of enterprises reported GenAI applications production-ready, with quality and governance as top barriers—a structural constraint directly limiting institutional adoption of question generation tools. The practice remained bifurcated: consumer and formative assessment adoption accelerated toward ubiquity (50%+ educator usage), yet high-stakes institutional assessment deployment remained blocked by unresolved evaluation methods, integrity concerns, copyright ambiguity, and enterprise-wide deployment readiness gaps.\n- **2025-Q1:** Accuracy research intensified confidence in barriers: BBC study found 51% of AI responses contained significant factual errors; Columbia study documented >60% incorrect answers to news questions, including fabricated citations. Medical education research (BMC Medical Education) provided validation of question quality measurement methods. Enterprise adoption data remained grim: 95% of AI pilots failed to deliver measurable value (MIT study); 42% of businesses abandoned AI initiatives in early 2025, up from 17% six months prior. Formative assessment adoption continued at scale, but institutional deployment remained constrained by accuracy risks, evaluation method immaturity, and enterprise-wide AI adoption failure rates exceeding 90%.\n- **2025-Q2:** Technical research advanced incrementally: NAACL 2025 introduced ConQuer framework with 4.8% quality improvements and 77% pairwise win rate over baselines. However, institutional adoption barriers deepened: Chinese exam authorities disabled AI tools during gaokao to prevent cheating, a concrete signal of assessment integrity concerns. Enterprise adoption constraints persisted: only 37% of enterprises believed GenAI applications production-ready, with quality and governance as top barriers. Formative assessment adoption remained strong (~50% educator usage), but high-stakes institutional deployment remained blocked by unresolved assessment integrity, governance, and production-readiness constraints.\n- **2025-Q3:** Real-world deployment pilots exposed persistent implementation friction: University of Iowa Khanmigo pilot (spring 2025) showed sub-weekly usage and zero teaching impact due to manual Canvas integration requirements; Michigan Virtual's large-scale K-12 pilot (1,700+ participants) documented value but emphasized need for intentional support. Technical quality validation advanced: peer-reviewed radiology education research confirmed acceptable psychometric properties of AI-generated MCQs. However, institutional barriers deepened: university unit chairs described exam design as a \"wicked problem\" with impossible trade-offs; student surveys showed 75% skepticism about AI accuracy. New vendor entries (Qzzr, others) signaled growing market, but formative assessment remained dominant use case; high-stakes institutional deployment remained blocked by governance ambiguity, implementation friction, and user acceptance barriers.\n\n- **2025-Q4:** Breakthrough technical validation confirmed question generation maturity in specialized domains: peer-reviewed Chest journal study found AI-generated MCQs (ChatGPT-o1) statistically noninferior to expert questions in mechanical ventilation education; large-scale field study across 1,700 students showed AI questions comparable to expert-created ones by psychometric analysis. Khan Academy's Khanmigo reached 1 million U.S. students, demonstrating rapid production scaling. Market adoption sustained (Quizgecko 854K+ monthly visits). However, integrity vulnerabilities persisted: criminal justice education study documented ChatGPT consistency issues (80% accuracy but unreliable across accounts), revealing assessment gaming risks. Governance, copyright, and user acceptance barriers remained binding constraints. The practice reached critical inflection: technical maturity validated, formative adoption accelerating toward ubiquity, yet institutional assessment deployment remained blocked by unresolved security and policy frameworks despite proven capability.\n\n- **2026-Jan:** OECD analysis documented AI \"item factories\" achieving 10x speed and cost reductions in exam question creation, while identifying \"crutch effect\" risk where AI-assisted practice improves scores but reduces independent performance when removed. Vendor ecosystem continued scaling: Eklavvya reported 99% time reduction in question paper creation and 95% reduction in paper leakage incidents; CogniGuide launched instant MCQ generation from documents. Research platforms advanced: EduQuest hybrid system achieved 82% difficulty classification accuracy and 71% higher student engagement with 78% time savings for educators. Engagement patterns documented: 40-60% higher quiz completion rates with AI-powered tools. The bifurcation persisted: formative assessment scaling with institutional adoption gains, yet high-stakes institutional deployment remained constrained by unresolved integrity, governance, and learning outcome trade-offs despite demonstrated technical capability and efficiency gains.\n\n- **2026-Feb:** Educator adoption surveys documented continued scaling across the globe. Macquarie University's 2026 open dataset surveyed educators on AI integration in assessment; LATAM higher education reported 92% student and 79% faculty AI engagement with 50% student support for AI-assisted feedback (though only 19% of faculty had deployed it). Systematic literature review of 103 AQG studies documented clear technical trend toward Transformer models but identified critical gap: educator acceptance remains largely understudied and evaluation methods lack standardization. Vendor ecosystem sustained momentum: TurinQ reported 50,000+ learners with multi-format generation and AI-based grading. Institutional deployment continued: Phillips Academy Andover committed to Khanmigo rollout in spring term for integrated tutoring and quiz generation, though student skepticism persisted about AI reliability. Practitioner guides enumerated 8+ mature platforms (Quizizz, Kahoot, Conker, ProProfs, QuestionWell, Twee, Gibbly, Formative) with documented adoption barriers (data privacy, bias risks). The practice remained bifurcated: technical tooling matured and vendor scaling accelerated, formative assessment adoption widened, yet institutional barriers (user skepticism, acceptance gaps, unresolved evaluation standards) continued constraining high-stakes deployment despite proven technical capability and efficiency gains.\n\n- **2026-Mar:** Production-scale evidence confirmed question generation maturity in enterprise contexts: Coursera's international survey (4,200 educators across 5 countries) reported 28% of faculty actively using AI to draft exams, up from isolated pockets a year prior; e-Assessment Association survey identified item generation as the most frequently used AI application in assessment across assessment organizations globally — a signal of mainstream adoption within the professional assessment sector. Peer-reviewed quality validation advanced: medical education comparative study (March 2026) found Gemini and Copilot MCQs achieved high inter-rater agreement on Bloom's taxonomy and learning outcome alignment, though pediatric study simultaneously documented AI-generated MCQs showing lower discrimination indices and higher proportion of difficulty mismatches vs. human questions. VitalSource field deployment (200+ undergraduates) confirmed classroom impact: distributed AI practice questions yielded 2% average exam score gains with letter-grade improvements at 25th percentile, validating formative assessment value in production. LATAM adoption data: 76% of teachers use AI tools for creating teaching materials (highest use case across surveyed practices); 92% student AI adoption in higher education. Production governance evidence: Khan Academy's Khanmigo optimization documented 64 completed A/B experiments (March 2026) testing iterative quiz generation improvements; University of Jyväskylä research on human-AI co-creation showed ~50% of AI-generated MCQs acceptable without editing through hybrid prompting and human revision. Persistent limitations remained: Washington State University study found ChatGPT only 60% above-chance on true/false accuracy (2025 iteration), 16.4% accuracy identifying false statements—demonstrating consistency gaps limiting high-stakes reliability. The practice showed clear bifurcation: formative assessment adoption accelerating toward ubiquity (28-76% faculty adoption depending on survey/context), with documented classroom learning gains; yet institutional assessment deployment remained constrained by unresolved consistency, governance, and integrity concerns despite production-scale optimization evidence and technical capability validation.\n- **2026-Apr:** Platform-level normalization advances with enterprise adoption milestones and persistent quality caveats. AssessPrep reports 800+ schools across 85+ countries with 5M+ student submissions and 500K AI-generated questions, providing hard deployment scale in production; MangoApps 2026 Winter Release adds document-to-quiz generation, signalling continued embedding of question generation into enterprise platforms as standard capability. A RAND survey of 4,200 K-12 teachers, however, finds only 38% rate AI assessment questions for higher-order thinking as good/excellent, with 42% requiring significant editing — confirming formative adoption breadth does not yet translate to quality confidence for complex assessment. Peer-reviewed caution on MCQ design flaws (hallucination, fact extraction errors, curriculum mismatches) reinforces that high-stakes deployment without systematic human review remains inadvisable. Governance maturation signals emerge: K-12 districts establish explicit AI assessment frameworks (Alexandria City Schools, Niles Township implement red/yellow/green policies); major testing vendor PSI/ETS releases structured AI test development product with SME review and psychometric rigor guidance. International high-stakes deployment models appear: Kazakhstan's national testing center deploys hybrid AI/expert item generation achieving 97.5% acceptance rates. Institutional productivity gains documented: NYC STEM institute reports 30 hours/week savings on question paper prep after adopting AI-assisted platform. Emerging capability: interactive assessment generation shows stronger exam performance signals and engagement gains while reducing per-instance cost to <$1. Transparency gaps persist: academic audit of 20 educational AI tools finds 80% fail to disclose generative mechanisms and 0 reveal training data sources, indicating maturation gap between capability and informed institutional decision-making.\n- **2026-May:** Production-scale deployment evidence strengthens: Pearson Study Prep (62,000+ students) documents 90% higher mastery rates; PSI Exams confirms 77.4% of AI-generated items meet psychometric thresholds at parity with human-authored content; Khan Academy announces a new \"Assessments\" product with psychometrics and norming, signaling transition from tutoring to structured high-stakes assessment. K-12 district pilots at scale confirm classroom deployment: Connecticut, Utah, Michigan, and NYC pilots document AI-generated exit questions and assessment feedback reaching students across named districts. A Cornell study of 95,000 students at 20 universities (published in Science) finds 37% use GenAI on assignments and 9% to cheat, motivating urgent structural redesign toward process portfolios and AI-adaptive question formats. A critical reliability finding emerged: empirical research documents AI-assisted assessments boosting observable performance 30 points but collapsing assessment reliability (Cronbach's α from 0.87 to 0.31), degrading diagnostic validity; simultaneously, a widely-cited ChatGPT learning meta-analysis (262 peer-reviewed citations) was retracted for methodological discrepancies, underscoring premature claims circulating in the field.\n\n- **2026-Jun:** Medical accuracy and reliability research deepens the case for mandatory human review: BMJ Open study (Tiller et al.) finds 49.6% of medical chatbot responses problematic or highly problematic with fabricated citations; Penn State study on 212 medical questions documents ChatGPT-4o at 84.6% but other models at ~50%, with domain-specific weakness in neurology and dermatology. Stanford AI Index analysis confirms a systemic reliability gap: hallucination rates 22-94% across 26 frontier models, with overconfidence on exactly the hardest items—a critical failure mode for oversight. Deployment at production scale continues to broaden: AssessPrep confirms 800+ schools and 500K AI-generated questions across international curricula; Japan's AI Passport Quiz App reached 10 million uses in 16 months; QuizMaker reports 10,000+ schools. A June 2026 market survey documents 10 mature platforms supporting standards alignment, adaptive difficulty, multi-modal items, and LMS integration across K-12, higher education, corporate training, and certification contexts; K-12 practitioner workflows using Gen AI for pre-assessment generation and item difficulty analysis (ALDO framework) are now documented as established classroom practice. The bifurcation holds: formative adoption and vendor scale are established, but medical and high-stakes domain deployment requires structural human-in-the-loop safeguards that governance frameworks have not yet standardized.\n\n- **2026-Jul:** Platform-level normalization accelerates: zyBooks (Wiley) launched a Multiple-Choice Question Generator (July 28) with teacher review gates and LMS integration, and Anthropic launched Claude for Teachers (July 17) with assessment-generation features, joining OpenAI, Google, and Microsoft in a competitive vendor field. Production-scale adoption evidence broadens: LearnWise's analysis of 191k+ student-AI interactions across 80+ institutions finds 35% involve generating quizzes, flashcards, and revision questions — confirming question generation as a primary student use case; Saudi Arabia's national AI education rollout engaged 50,000+ students with platform-level exam generation; and a University of Sydney pilot (100+ AI-generated practice questions) saw 74% student engagement with 65% rating them useful. Institutional governance strain surfaces at benchmark scale: UNAM (158,000 admissions applicants, Latin America's largest public university) paused validation of online exams after unusually strong scores triggered an AI-cheating investigation, with 22,000 places at stake, while Princeton, UChicago Law, UCLA, and Waterloo abandon take-home remote exams for in-person proctoring after documented AI cheating — both signaling eroding institutional confidence in unproctored, AI-vulnerable assessment. Reliability research hardens the case for pedagogical guardrails: an OECD \"fast AI vs. slow AI\" framework and a Wharton field experiment (~1,000 students, Bastani et al.) both find generic (\"fast AI\") tool use cuts exam scores 17% despite 48% more practice problems solved, while a guardrailed, hints-only tutor delivers 127% practice gains with no exam penalty; Brown University's ECON 1170 shows the same collapse in miniature (96% average on an AI-assisted take-home vs. 48.6%, with 22 perfect-scorers failing, once switched to a proctored final); and Dartmouth's Phosphor platform (90.2% engagement, +0.71–1.30 SD) shows the effect is format-dependent — gains hold only for constructed-response questions, not MCQ-only formats. Technical and quality limitations persist: Zotos et al. document LLM-generated MCQs inheriting training-corpus bias in distractor construction; Tseng, Akgun, and Liu show human-AI disagreement on question quality is systematic rather than random; a 552-question ReadRoost practice bank required rebuilding after an AI-hallucinated question, prompting a live-documentation verification gate; and hands-on practitioner testing of Conker, Knowt, and MagicSchool confirms teacher review remains non-optional, with output described as recall-heavy and editing-dependent. Governance and market signals continue maturing: a cross-national review of 21 universities finds explicit GenAI-assessment policies now standard, converging on four design patterns (process portfolios, AI+verification, critical engagement, secure exams with AI coursework); Turnitin data shows AI used in 53.6% of Australian university assignments; a peer-reviewed co-assessment governance model (Totty, Cunov, Erskine, AMCIS 2026) proposes AI-scored rubrics paired with faculty judgment; and the K-12 assessment market (55% of US districts now deploying AI-powered assessment) is projected at 8.39% CAGR through 2033. A 26,000-student, 30-month study reinforces the core tension: AI homework tools lift homework scores 18% but cut exam performance 20% — evidence that the bifurcation between ubiquitous formative use and blocked high-stakes deployment continues to deepen rather than resolve.\n- **2026-Aug:** Vendor platforms move toward workflow integration over standalone generation: OpenAI launched K-12 Educator, College Educator, and Student plugins with structured quiz/test-generation workflows designed to preserve student agency, while D2L's analyst commentary frames quiz generation as a starting point that must preserve educator judgment rather than replace it. UK regulator Ofqual formally designates AI item generation a priority use case for awarding bodies, mandating human-in-the-loop review — an early formal governance framework for high-stakes question generation. Technical limitations remain sharply documented: the Ask-E benchmark finds frontier LLMs achieve below 50% calibration on generating difficulty-matched questions, and Spanish-language reporting cites studies where only 20% of students and 55% of medical residents caught planted hallucinations in AI-generated assessment content. Peer-reviewed and applied evidence continues supporting narrower, human-reviewed deployment: a two-loop LLM validation method demonstrates production-ready methodology for high-stakes exam content, a domain-expert review of medical question generators concludes tools are \"good enough to supplement but not foundation\" for question banks, and a Turkish study of 64 pre-service teachers documents real classroom adoption alongside persistent accuracy concerns. High-stakes governance failures surface concretely: California's Bar Exam is found to have deployed 23 undisclosed AI-generated scored questions affecting 85+ examinees, triggering AB 1651's mandatory disclosure requirement, while India's UGC-NET cancels three subject papers after AI-linked errors affect 20,000 applicants. Adoption-scale data hardens the bifurcation: 80%+ of US secondary students use AI for schoolwork against only 50% school policy coverage, India reports 45M students on AI platforms with quiz generation the top teacher use case (51%), and a 26,811-student, 30-month study finds AI question practice lifts homework scores 18% while cutting exam scores 20%. Google's GA of diagnostic quizzes in Gemini study notebooks extends platform normalization, even as FSA/CFA audits document systematic distractor-quality defects and a Kentucky middle school distributes AI-generated materials with severe factual hallucinations.\n- **2026-Sep:** A high-profile integrity failure forces curriculum-level redesign: MIT's EECS course 6.036 found 73-84% of problem sets AI-generated, triggering a $150K overhaul replacing auto-graded sets with oral exams and studio sessions. Adoption keeps broadening at platform scale — 68% of US districts now formally contract genAI platforms (up from 42%), OpenAI's ChatGPT for Teachers expanded to 55 more districts (300,000+ educators, quiz generation a primary use case), and Jordan School District (Utah) reports a 28% critical-thinking gain across 14,000 AI-tutored interactions. The reliability bifurcation sharpens further: a 1,066-student natural experiment finds failure rates jump from 2-6% on AI-accessible take-home exams to 18.4% under AI-restricted proctored conditions, and a 27,000-student, 30-month study confirms AI homework help lifts scores 18% while cutting exam performance 20%, reinforcing \"metacognitive laziness\" concerns. In specialized domains, a peer-reviewed radiology study finds template-based automatic item generation outperforms ChatGPT-4o on item discrimination, validating narrower AI-assisted methods over general-purpose LLM generation. Late-month evidence adds that a 43-study scoping review finds autonomous high-stakes generation unsupported, with hybrid human-AI setups dominant; Brown's take-home average hit 96 against 48 in person, and Kellogg trialled AI oral exams.",
  "historyEntries": [
    {
      "period": "2023-H1",
      "text": "Early open-source projects (Quizify, AI-Quiz-Generator) and consumer tools (Conker.ai, QwizLab) demonstrated proof-of-concept that generative models could create quizzes and exams from text; institutional deployments remained minimal."
    },
    {
      "period": "2023-H2",
      "text": "Khan Academy deployed Khanmigo with quiz generation at scale (1000s of users); research validated AI performance on degree-level mathematics exams; iQS deployed in university Moodle systems. Adoption surveys showed educators saw utility but lacked confidence; accuracy concerns and slow institutional trust remained primary barriers."
    },
    {
      "period": "2024-Q1",
      "text": "Consumer adoption accelerated (Conker: 600k quizzes, QuizFlex: 10k+ educators, QuizGeniusAI: 300+ educators). Research identified specific gaps: EDM 2024 study found GPT-4 achieved 70% validity for question stems but only 37% for distractors/misconceptions. Pearson VUE's high-stakes testing research showed time-saving benefits but persistent issues with cognitive level calibration. Critical assessment from AQA highlighted unresolved IP, bias, and reliability concerns—signalling adoption barriers in formal assessment despite technical progress."
    },
    {
      "period": "2024-Q2",
      "text": "Ecosystem maturation continued: Khan Academy expanded Khanmigo with free AI question generators for all teachers; forms.app's comparative analysis identified 12+ mature quiz generation platforms across the market. Questgen and other tools demonstrated multi-format generation capability. However, critical integrity research from University of Reading (June 2024) revealed AI-generated exam submissions achieved 94% undetection rate and grades 0.5 boundaries higher than real students—a stark demonstration of assessment integrity risk. Institutional resistance intensified: American Board of Radiology formally banned AI-generated exam content, citing copyright and integrity concerns. The tension sharpened: widespread consumer adoption and vendor expansion collided with accumulating evidence of both technical limitations (distractor quality, cognitive calibration) and integrity risks (undetectable submissions, grade inflation), creating a widening gap between bleeding-edge tooling and high-stakes institutional deployment readiness."
    },
    {
      "period": "2024-Q3",
      "text": "Vendor consolidation and geographic scaling: Khan Academy made Khanmigo free to teachers in 49 countries via Microsoft partnership (Aug 2024), then integrated into Canvas LMS for U.S. educators (Sept 2024). Research highlighted evaluation as the key bottleneck: AAAI 2024 conference paper documented that question quality assessment methods limit full integration into education; arXiv review catalogued AI's capabilities (Bloom's-aligned questions) and gaps (ethical, accuracy, consistency). Project failure rates remained high (RAND: 80% of AI projects fail), and deployment risks intensified—documented guardrail failures in production systems, benchmark validity critiques, and evidence that 80%+ of enterprise AI projects abandoned. The practice reached a critical inflection: consumer adoption was evident and accelerating, but institutional deployment remained blocked by evaluation uncertainty, integrity concerns, and widespread deployment failures across AI systems generally."
    },
    {
      "period": "2024-Q4",
      "text": "Adoption continued: HMH 2024 survey showed 50% of educators using GenAI tools (5x increase YoY) with assessment creation among top use cases; Khan Academy's efficacy study demonstrated ~350K student user base with measurable learning gains, confirming scale and real classroom impact. Vendor consolidation persisted: ABMS boards piloted AI question generation for medical certification (American Board of Anesthesiology), though paused due to copyright concerns. Real-world deployment risks materialized: NSW Education Authority's October 2024 HSC exam included AI-generated image, sparking student complaints about authenticity and suitability, illustrating quality acceptance barriers in high-stakes contexts. Critical finding emerged: only 37% of enterprises reported GenAI applications production-ready, with quality and governance as top barriers—a structural constraint directly limiting institutional adoption of question generation tools. The practice remained bifurcated: consumer and formative assessment adoption accelerated toward ubiquity (50%+ educator usage), yet high-stakes institutional assessment deployment remained blocked by unresolved evaluation methods, integrity concerns, copyright ambiguity, and enterprise-wide deployment readiness gaps."
    },
    {
      "period": "2025-Q1",
      "text": "Accuracy research intensified confidence in barriers: BBC study found 51% of AI responses contained significant factual errors; Columbia study documented >60% incorrect answers to news questions, including fabricated citations. Medical education research (BMC Medical Education) provided validation of question quality measurement methods. Enterprise adoption data remained grim: 95% of AI pilots failed to deliver measurable value (MIT study); 42% of businesses abandoned AI initiatives in early 2025, up from 17% six months prior. Formative assessment adoption continued at scale, but institutional deployment remained constrained by accuracy risks, evaluation method immaturity, and enterprise-wide AI adoption failure rates exceeding 90%."
    },
    {
      "period": "2025-Q2",
      "text": "Technical research advanced incrementally: NAACL 2025 introduced ConQuer framework with 4.8% quality improvements and 77% pairwise win rate over baselines. However, institutional adoption barriers deepened: Chinese exam authorities disabled AI tools during gaokao to prevent cheating, a concrete signal of assessment integrity concerns. Enterprise adoption constraints persisted: only 37% of enterprises believed GenAI applications production-ready, with quality and governance as top barriers. Formative assessment adoption remained strong (~50% educator usage), but high-stakes institutional deployment remained blocked by unresolved assessment integrity, governance, and production-readiness constraints."
    },
    {
      "period": "2025-Q3",
      "text": "Real-world deployment pilots exposed persistent implementation friction: University of Iowa Khanmigo pilot (spring 2025) showed sub-weekly usage and zero teaching impact due to manual Canvas integration requirements; Michigan Virtual's large-scale K-12 pilot (1,700+ participants) documented value but emphasized need for intentional support. Technical quality validation advanced: peer-reviewed radiology education research confirmed acceptable psychometric properties of AI-generated MCQs. However, institutional barriers deepened: university unit chairs described exam design as a \"wicked problem\" with impossible trade-offs; student surveys showed 75% skepticism about AI accuracy. New vendor entries (Qzzr, others) signaled growing market, but formative assessment remained dominant use case; high-stakes institutional deployment remained blocked by governance ambiguity, implementation friction, and user acceptance barriers."
    },
    {
      "period": "2025-Q4",
      "text": "Breakthrough technical validation confirmed question generation maturity in specialized domains: peer-reviewed Chest journal study found AI-generated MCQs (ChatGPT-o1) statistically noninferior to expert questions in mechanical ventilation education; large-scale field study across 1,700 students showed AI questions comparable to expert-created ones by psychometric analysis. Khan Academy's Khanmigo reached 1 million U.S. students, demonstrating rapid production scaling. Market adoption sustained (Quizgecko 854K+ monthly visits). However, integrity vulnerabilities persisted: criminal justice education study documented ChatGPT consistency issues (80% accuracy but unreliable across accounts), revealing assessment gaming risks. Governance, copyright, and user acceptance barriers remained binding constraints. The practice reached critical inflection: technical maturity validated, formative adoption accelerating toward ubiquity, yet institutional assessment deployment remained blocked by unresolved security and policy frameworks despite proven capability."
    },
    {
      "period": "2026-Jan",
      "text": "OECD analysis documented AI \"item factories\" achieving 10x speed and cost reductions in exam question creation, while identifying \"crutch effect\" risk where AI-assisted practice improves scores but reduces independent performance when removed. Vendor ecosystem continued scaling: Eklavvya reported 99% time reduction in question paper creation and 95% reduction in paper leakage incidents; CogniGuide launched instant MCQ generation from documents. Research platforms advanced: EduQuest hybrid system achieved 82% difficulty classification accuracy and 71% higher student engagement with 78% time savings for educators. Engagement patterns documented: 40-60% higher quiz completion rates with AI-powered tools. The bifurcation persisted: formative assessment scaling with institutional adoption gains, yet high-stakes institutional deployment remained constrained by unresolved integrity, governance, and learning outcome trade-offs despite demonstrated technical capability and efficiency gains."
    },
    {
      "period": "2026-Feb",
      "text": "Educator adoption surveys documented continued scaling across the globe. Macquarie University's 2026 open dataset surveyed educators on AI integration in assessment; LATAM higher education reported 92% student and 79% faculty AI engagement with 50% student support for AI-assisted feedback (though only 19% of faculty had deployed it). Systematic literature review of 103 AQG studies documented clear technical trend toward Transformer models but identified critical gap: educator acceptance remains largely understudied and evaluation methods lack standardization. Vendor ecosystem sustained momentum: TurinQ reported 50,000+ learners with multi-format generation and AI-based grading. Institutional deployment continued: Phillips Academy Andover committed to Khanmigo rollout in spring term for integrated tutoring and quiz generation, though student skepticism persisted about AI reliability. Practitioner guides enumerated 8+ mature platforms (Quizizz, Kahoot, Conker, ProProfs, QuestionWell, Twee, Gibbly, Formative) with documented adoption barriers (data privacy, bias risks). The practice remained bifurcated: technical tooling matured and vendor scaling accelerated, formative assessment adoption widened, yet institutional barriers (user skepticism, acceptance gaps, unresolved evaluation standards) continued constraining high-stakes deployment despite proven technical capability and efficiency gains."
    },
    {
      "period": "2026-Mar",
      "text": "Production-scale evidence confirmed question generation maturity in enterprise contexts: Coursera's international survey (4,200 educators across 5 countries) reported 28% of faculty actively using AI to draft exams, up from isolated pockets a year prior; e-Assessment Association survey identified item generation as the most frequently used AI application in assessment across assessment organizations globally — a signal of mainstream adoption within the professional assessment sector. Peer-reviewed quality validation advanced: medical education comparative study (March 2026) found Gemini and Copilot MCQs achieved high inter-rater agreement on Bloom's taxonomy and learning outcome alignment, though pediatric study simultaneously documented AI-generated MCQs showing lower discrimination indices and higher proportion of difficulty mismatches vs. human questions. VitalSource field deployment (200+ undergraduates) confirmed classroom impact: distributed AI practice questions yielded 2% average exam score gains with letter-grade improvements at 25th percentile, validating formative assessment value in production. LATAM adoption data: 76% of teachers use AI tools for creating teaching materials (highest use case across surveyed practices); 92% student AI adoption in higher education. Production governance evidence: Khan Academy's Khanmigo optimization documented 64 completed A/B experiments (March 2026) testing iterative quiz generation improvements; University of Jyväskylä research on human-AI co-creation showed ~50% of AI-generated MCQs acceptable without editing through hybrid prompting and human revision. Persistent limitations remained: Washington State University study found ChatGPT only 60% above-chance on true/false accuracy (2025 iteration), 16.4% accuracy identifying false statements—demonstrating consistency gaps limiting high-stakes reliability. The practice showed clear bifurcation: formative assessment adoption accelerating toward ubiquity (28-76% faculty adoption depending on survey/context), with documented classroom learning gains; yet institutional assessment deployment remained constrained by unresolved consistency, governance, and integrity concerns despite production-scale optimization evidence and technical capability validation."
    },
    {
      "period": "2026-Apr",
      "text": "Platform-level normalization advances with enterprise adoption milestones and persistent quality caveats. AssessPrep reports 800+ schools across 85+ countries with 5M+ student submissions and 500K AI-generated questions, providing hard deployment scale in production; MangoApps 2026 Winter Release adds document-to-quiz generation, signalling continued embedding of question generation into enterprise platforms as standard capability. A RAND survey of 4,200 K-12 teachers, however, finds only 38% rate AI assessment questions for higher-order thinking as good/excellent, with 42% requiring significant editing — confirming formative adoption breadth does not yet translate to quality confidence for complex assessment. Peer-reviewed caution on MCQ design flaws (hallucination, fact extraction errors, curriculum mismatches) reinforces that high-stakes deployment without systematic human review remains inadvisable. Governance maturation signals emerge: K-12 districts establish explicit AI assessment frameworks (Alexandria City Schools, Niles Township implement red/yellow/green policies); major testing vendor PSI/ETS releases structured AI test development product with SME review and psychometric rigor guidance. International high-stakes deployment models appear: Kazakhstan's national testing center deploys hybrid AI/expert item generation achieving 97.5% acceptance rates. Institutional productivity gains documented: NYC STEM institute reports 30 hours/week savings on question paper prep after adopting AI-assisted platform. Emerging capability: interactive assessment generation shows stronger exam performance signals and engagement gains while reducing per-instance cost to <$1. Transparency gaps persist: academic audit of 20 educational AI tools finds 80% fail to disclose generative mechanisms and 0 reveal training data sources, indicating maturation gap between capability and informed institutional decision-making."
    },
    {
      "period": "2026-May",
      "text": "Production-scale deployment evidence strengthens: Pearson Study Prep (62,000+ students) documents 90% higher mastery rates; PSI Exams confirms 77.4% of AI-generated items meet psychometric thresholds at parity with human-authored content; Khan Academy announces a new \"Assessments\" product with psychometrics and norming, signaling transition from tutoring to structured high-stakes assessment. K-12 district pilots at scale confirm classroom deployment: Connecticut, Utah, Michigan, and NYC pilots document AI-generated exit questions and assessment feedback reaching students across named districts. A Cornell study of 95,000 students at 20 universities (published in Science) finds 37% use GenAI on assignments and 9% to cheat, motivating urgent structural redesign toward process portfolios and AI-adaptive question formats. A critical reliability finding emerged: empirical research documents AI-assisted assessments boosting observable performance 30 points but collapsing assessment reliability (Cronbach's α from 0.87 to 0.31), degrading diagnostic validity; simultaneously, a widely-cited ChatGPT learning meta-analysis (262 peer-reviewed citations) was retracted for methodological discrepancies, underscoring premature claims circulating in the field."
    },
    {
      "period": "2026-Jun",
      "text": "Medical accuracy and reliability research deepens the case for mandatory human review: BMJ Open study (Tiller et al.) finds 49.6% of medical chatbot responses problematic or highly problematic with fabricated citations; Penn State study on 212 medical questions documents ChatGPT-4o at 84.6% but other models at ~50%, with domain-specific weakness in neurology and dermatology. Stanford AI Index analysis confirms a systemic reliability gap: hallucination rates 22-94% across 26 frontier models, with overconfidence on exactly the hardest items—a critical failure mode for oversight. Deployment at production scale continues to broaden: AssessPrep confirms 800+ schools and 500K AI-generated questions across international curricula; Japan's AI Passport Quiz App reached 10 million uses in 16 months; QuizMaker reports 10,000+ schools. A June 2026 market survey documents 10 mature platforms supporting standards alignment, adaptive difficulty, multi-modal items, and LMS integration across K-12, higher education, corporate training, and certification contexts; K-12 practitioner workflows using Gen AI for pre-assessment generation and item difficulty analysis (ALDO framework) are now documented as established classroom practice. The bifurcation holds: formative adoption and vendor scale are established, but medical and high-stakes domain deployment requires structural human-in-the-loop safeguards that governance frameworks have not yet standardized."
    },
    {
      "period": "2026-Jul",
      "text": "Platform-level normalization accelerates: zyBooks (Wiley) launched a Multiple-Choice Question Generator (July 28) with teacher review gates and LMS integration, and Anthropic launched Claude for Teachers (July 17) with assessment-generation features, joining OpenAI, Google, and Microsoft in a competitive vendor field. Production-scale adoption evidence broadens: LearnWise's analysis of 191k+ student-AI interactions across 80+ institutions finds 35% involve generating quizzes, flashcards, and revision questions — confirming question generation as a primary student use case; Saudi Arabia's national AI education rollout engaged 50,000+ students with platform-level exam generation; and a University of Sydney pilot (100+ AI-generated practice questions) saw 74% student engagement with 65% rating them useful. Institutional governance strain surfaces at benchmark scale: UNAM (158,000 admissions applicants, Latin America's largest public university) paused validation of online exams after unusually strong scores triggered an AI-cheating investigation, with 22,000 places at stake, while Princeton, UChicago Law, UCLA, and Waterloo abandon take-home remote exams for in-person proctoring after documented AI cheating — both signaling eroding institutional confidence in unproctored, AI-vulnerable assessment. Reliability research hardens the case for pedagogical guardrails: an OECD \"fast AI vs. slow AI\" framework and a Wharton field experiment (~1,000 students, Bastani et al.) both find generic (\"fast AI\") tool use cuts exam scores 17% despite 48% more practice problems solved, while a guardrailed, hints-only tutor delivers 127% practice gains with no exam penalty; Brown University's ECON 1170 shows the same collapse in miniature (96% average on an AI-assisted take-home vs. 48.6%, with 22 perfect-scorers failing, once switched to a proctored final); and Dartmouth's Phosphor platform (90.2% engagement, +0.71–1.30 SD) shows the effect is format-dependent — gains hold only for constructed-response questions, not MCQ-only formats. Technical and quality limitations persist: Zotos et al. document LLM-generated MCQs inheriting training-corpus bias in distractor construction; Tseng, Akgun, and Liu show human-AI disagreement on question quality is systematic rather than random; a 552-question ReadRoost practice bank required rebuilding after an AI-hallucinated question, prompting a live-documentation verification gate; and hands-on practitioner testing of Conker, Knowt, and MagicSchool confirms teacher review remains non-optional, with output described as recall-heavy and editing-dependent. Governance and market signals continue maturing: a cross-national review of 21 universities finds explicit GenAI-assessment policies now standard, converging on four design patterns (process portfolios, AI+verification, critical engagement, secure exams with AI coursework); Turnitin data shows AI used in 53.6% of Australian university assignments; a peer-reviewed co-assessment governance model (Totty, Cunov, Erskine, AMCIS 2026) proposes AI-scored rubrics paired with faculty judgment; and the K-12 assessment market (55% of US districts now deploying AI-powered assessment) is projected at 8.39% CAGR through 2033. A 26,000-student, 30-month study reinforces the core tension: AI homework tools lift homework scores 18% but cut exam performance 20% — evidence that the bifurcation between ubiquitous formative use and blocked high-stakes deployment continues to deepen rather than resolve."
    },
    {
      "period": "2026-Aug",
      "text": "Vendor platforms move toward workflow integration over standalone generation: OpenAI launched K-12 Educator, College Educator, and Student plugins with structured quiz/test-generation workflows designed to preserve student agency, while D2L's analyst commentary frames quiz generation as a starting point that must preserve educator judgment rather than replace it. UK regulator Ofqual formally designates AI item generation a priority use case for awarding bodies, mandating human-in-the-loop review — an early formal governance framework for high-stakes question generation. Technical limitations remain sharply documented: the Ask-E benchmark finds frontier LLMs achieve below 50% calibration on generating difficulty-matched questions, and Spanish-language reporting cites studies where only 20% of students and 55% of medical residents caught planted hallucinations in AI-generated assessment content. Peer-reviewed and applied evidence continues supporting narrower, human-reviewed deployment: a two-loop LLM validation method demonstrates production-ready methodology for high-stakes exam content, a domain-expert review of medical question generators concludes tools are \"good enough to supplement but not foundation\" for question banks, and a Turkish study of 64 pre-service teachers documents real classroom adoption alongside persistent accuracy concerns. High-stakes governance failures surface concretely: California's Bar Exam is found to have deployed 23 undisclosed AI-generated scored questions affecting 85+ examinees, triggering AB 1651's mandatory disclosure requirement, while India's UGC-NET cancels three subject papers after AI-linked errors affect 20,000 applicants. Adoption-scale data hardens the bifurcation: 80%+ of US secondary students use AI for schoolwork against only 50% school policy coverage, India reports 45M students on AI platforms with quiz generation the top teacher use case (51%), and a 26,811-student, 30-month study finds AI question practice lifts homework scores 18% while cutting exam scores 20%. Google's GA of diagnostic quizzes in Gemini study notebooks extends platform normalization, even as FSA/CFA audits document systematic distractor-quality defects and a Kentucky middle school distributes AI-generated materials with severe factual hallucinations."
    },
    {
      "period": "2026-Sep",
      "text": "A high-profile integrity failure forces curriculum-level redesign: MIT's EECS course 6.036 found 73-84% of problem sets AI-generated, triggering a $150K overhaul replacing auto-graded sets with oral exams and studio sessions. Adoption keeps broadening at platform scale — 68% of US districts now formally contract genAI platforms (up from 42%), OpenAI's ChatGPT for Teachers expanded to 55 more districts (300,000+ educators, quiz generation a primary use case), and Jordan School District (Utah) reports a 28% critical-thinking gain across 14,000 AI-tutored interactions. The reliability bifurcation sharpens further: a 1,066-student natural experiment finds failure rates jump from 2-6% on AI-accessible take-home exams to 18.4% under AI-restricted proctored conditions, and a 27,000-student, 30-month study confirms AI homework help lifts scores 18% while cutting exam performance 20%, reinforcing \"metacognitive laziness\" concerns. In specialized domains, a peer-reviewed radiology study finds template-based automatic item generation outperforms ChatGPT-4o on item discrimination, validating narrower AI-assisted methods over general-purpose LLM generation. Late-month evidence adds that a 43-study scoping review finds autonomous high-stakes generation unsupported, with hybrid human-AI setups dominant; Brown's take-home average hit 96 against 48 in person, and Kellogg trialled AI oral exams."
    }
  ],
  "historyFallback": false,
  "lastUpdated": "2026-09-25",
  "domain": {
    "id": "education-learning",
    "label": "Education & Learning",
    "icon": "🎓"
  },
  "url": "https://www.thestateofplay.ai/practice/question-and-exam-generation",
  "license": "CC BY 4.0",
  "licenseUrl": "https://creativecommons.org/licenses/by/4.0/",
  "generatedAt": "2026-10-01"
}