@article{4826, author = {Pit Pichappan}, title = {Structural and Linguistic Markers for Distinguishing Human Manuscripts from AI-Generated Summaries}, journal = {Journal of Digital Information Management}, year = {2026}, volume = {24}, number = {3}, doi = {https://doi.org/10.6025/jdim/2026/24/3/127-148}, url = {https://www.dline.info/fpaper/jdim/v24i3/jdimv24i3_1.pdf}, abstract = {The rapid development of Large Language Models (LLMs) has made it increasingly hard to distinguish human written from AI-generated academic papers, raising questions about authorship, originality, and academic integrity. This research presents a systematic approach to identifying and measuring multidimensional markers that differentiate human written academic manuscripts from AI-generated and AI-paraphrased summaries. After analysing 52 human written academic papers and their AI-generated versions (created with GPT-4o, DeepSeek V4, and Qwen 3.7plus), we developed an automated, multi-layered text-annotation tool. It incorporates rule based matching, zero shot classification, and structured LLMaided annotation, and we validated it using human inter annotator agreement. We found that no single language marker is enough for classification. However, the accurate classification (85-90%) requires a set of markers which include the completeness of the scholarly apparatus (the most reliable one), significant information loss (40-50% of granular details such as exact statistics or software name), decrease of syntactic burstiness (by 40-50%), formulaic transition substitutions, and the drop in epistemic hedging by 60%. Although the absence of structure and metadata yields perfect classification, the stylometric approach alone produced an AUC of 0.78, underscoring the complexity of linguistic signatures. In summary, successfully implementing AI for identifying text in academic environments requires a multi marker approach that favours completeness, informativeness, and text depth over fluency and vocabulary sophistication, which are highly prone to false positive results. Previous research has addressed AI academic detection; however, we shift our emphasis from surface fluency (which LLMs excel at mimicking) to epistemic rigour and granular information loss.}, }