@article{ART003335483},
author={Jo Meounggun and Lee, Ju Hyun},
title={Effectiveness of a GPT-OSS-Based Automated Essay Scoring System and the Educational Value of LLM-Generated Feedback: A Case Study of Descriptive Writing Tasks},
journal={The Journal of General Education},
issn={2465-7581},
year={2026},
number={35},
pages={127-157},
doi={10.24173/jge.2026.04.30.4}
TY - JOUR
AU - Jo Meounggun
AU - Lee, Ju Hyun
TI - Effectiveness of a GPT-OSS-Based Automated Essay Scoring System and the Educational Value of LLM-Generated Feedback: A Case Study of Descriptive Writing Tasks
JO - The Journal of General Education
PY - 2026
VL - null
IS - 35
PB - Da Vinci Mirae Institute of General Education
SP - 127
EP - 157
SN - 2465-7581
AB - This study aims to examine whether an automated essay scoring system based on a large language model (LLM) can function as a practical assessment tool in university writing education, and to investigate the educational value of the feedback it generates. To this end, an automated scoring system was implemented using the GPT-OSS 20B model in a local environment, and three scoring strategies—persona-based, chain-of-thought, and pairwise comparison—were applied to the same dataset and comparatively analyzed. In addition to measuring score agreement, the quality of the generated feedback was also evaluated by experts to examine its qualitative validity.
The results show that the chain-of-thought approach demonstrated the highest agreement with human scoring, while the persona-based approach showed a conservative scoring tendency and the pairwise comparison approach showed a lenient tendency. In addition, the LLM showed high consistency in identifying low-quality texts but had limitations in distinguishing subtle qualitative differences among high-performing texts. The generated feedback focused on content organization and expression improvement, indicating its educational applicability. Meanwhile, differences in computational cost were observed depending on the scoring strategy, and in particular, the pairwise comparison approach showed high computational burden, suggesting limitations for practical application. These findings suggest that LLM-based automated scoring systems are more effective when used as collaborative tools supporting feedback provision rather than as replacements for human evaluators.
KW - LLM Feedback;Comparison of LLM Scoring Strategies;College Writing Feedback;Large Language Models;Automated Essay Scoring;Writing Education
DO - 10.24173/jge.2026.04.30.4
ER -
Jo Meounggun and Lee, Ju Hyun. (2026). Effectiveness of a GPT-OSS-Based Automated Essay Scoring System and the Educational Value of LLM-Generated Feedback: A Case Study of Descriptive Writing Tasks. The Journal of General Education, 35, 127-157.
Jo Meounggun and Lee, Ju Hyun. 2026, "Effectiveness of a GPT-OSS-Based Automated Essay Scoring System and the Educational Value of LLM-Generated Feedback: A Case Study of Descriptive Writing Tasks", The Journal of General Education, no.35, pp.127-157. Available from: doi:10.24173/jge.2026.04.30.4
Jo Meounggun, Lee, Ju Hyun "Effectiveness of a GPT-OSS-Based Automated Essay Scoring System and the Educational Value of LLM-Generated Feedback: A Case Study of Descriptive Writing Tasks" The Journal of General Education 35 pp.127-157 (2026) : 127.
Jo Meounggun, Lee, Ju Hyun. Effectiveness of a GPT-OSS-Based Automated Essay Scoring System and the Educational Value of LLM-Generated Feedback: A Case Study of Descriptive Writing Tasks. 2026; 35 : 127-157. Available from: doi:10.24173/jge.2026.04.30.4
Jo Meounggun and Lee, Ju Hyun. "Effectiveness of a GPT-OSS-Based Automated Essay Scoring System and the Educational Value of LLM-Generated Feedback: A Case Study of Descriptive Writing Tasks" The Journal of General Education no.35(2026) : 127-157.doi: 10.24173/jge.2026.04.30.4
Jo Meounggun; Lee, Ju Hyun. Effectiveness of a GPT-OSS-Based Automated Essay Scoring System and the Educational Value of LLM-Generated Feedback: A Case Study of Descriptive Writing Tasks. The Journal of General Education, 35, 127-157. doi: 10.24173/jge.2026.04.30.4
Jo Meounggun; Lee, Ju Hyun. Effectiveness of a GPT-OSS-Based Automated Essay Scoring System and the Educational Value of LLM-Generated Feedback: A Case Study of Descriptive Writing Tasks. The Journal of General Education. 2026; 35 127-157. doi: 10.24173/jge.2026.04.30.4
Jo Meounggun, Lee, Ju Hyun. Effectiveness of a GPT-OSS-Based Automated Essay Scoring System and the Educational Value of LLM-Generated Feedback: A Case Study of Descriptive Writing Tasks. 2026; 35 : 127-157. Available from: doi:10.24173/jge.2026.04.30.4
Jo Meounggun and Lee, Ju Hyun. "Effectiveness of a GPT-OSS-Based Automated Essay Scoring System and the Educational Value of LLM-Generated Feedback: A Case Study of Descriptive Writing Tasks" The Journal of General Education no.35(2026) : 127-157.doi: 10.24173/jge.2026.04.30.4