@inproceedings{maheshwary-etal-2026-llm,
title = "{LLM}-Generated Stories for Students with Significant Cognitive Disabilities: Promise, Gaps, and Evaluation Framework",
author = "Maheshwary, Pragati and
Ganesh, Ananya and
Karumbaiah, Shamya",
editor = "Shardlow, Matthew and
Fran{\c{c}}ois, Thomas and
Amaro, Raquel and
Baptista, Jorge and
Cardon, R{\'e}mi and
Ribeiro, Eug{\'e}nio and
Saggion, Horacio and
Stodden, Regina and
Todirascu, Amalia and
Wilkens, Rodrigo",
booktitle = "Proceedings of the Joint Workshop on Readability and Text Simplification ({READI}x{TSAR}) @ {LREC} 2026",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://preview.aclanthology.org/ingest-lrec/2026.readi-1.10/",
doi = "10.63317/3b9f4txwp2n2",
pages = "130--141",
abstract = "Students with significant cognitive disabilities (SCD) require specially designed accessible stories for reading comprehension assessments, yet creating such content is labor-intensive and difficult to scale. This preliminary study investigates whether large language models (LLMs) can generate short accessible stories for alternate assessment system. Using an 8-fold cross-validation design, we generated 120 stories with GPT-4o via one-shot prompting with human-written exemplars and evaluated them against a test set comprising 7 expert-human written stories as baselines across three dimensions: simplicity, fluency {\&} coherence, and thematic adherence. Cross-validation results show that generated stories meet surface-level simplicity targets, with approximately two-thirds falling within the human baseline range for readability metrics. However, generated stories exhibited a systematic coherence gap where only 5{\%} fell within the human range for adjacent sentence similarity, a pattern consistent across all folds. Thematic adherence was moderate, with adequate diversity across stories. These findings suggest LLMs can serve as a drafting tool within accessible content generation pipelines, but human expert review remains essential to ensure coherence, testability, and alignment with quality standards required for high-stakes alternate assessments."
}Markdown (Informal)
[LLM-Generated Stories for Students with Significant Cognitive Disabilities: Promise, Gaps, and Evaluation Framework](https://preview.aclanthology.org/ingest-lrec/2026.readi-1.10/) (Maheshwary et al., READI-TSAR 2026)
ACL