@inproceedings{84b5f839203d4ae4acfff000355075ff,
title = "GPT-4 is Judged More Human than Humans in Displaced and Inverted Turing Tests",
abstract = "Everyday AI detection requires differentiating between humans and AI in informal, online conversations. At present, human users most often do not interact directly with bots but instead read their conversations with other humans. We measured how well humans and large language models can discriminate using two modified versions of the Turing test: inverted and displaced. GPT-3.5, GPT-4, and displaced human adjudicators judged whether an agent was human or AI on the basis of a Turing test transcript. We found that both AI and displaced human judges were less accurate than interactive interrogators, with below chance accuracy overall. Moreover, all three judged the best-performing GPT-4 witness to be human more often than human witnesses. This suggests that both humans and current LLMs struggle to distinguish between the two when they are not actively interrogating the person, underscoring an urgent need for more accurate tools to detect AI in conversations.",
author = "Ishika Rathi and Sydney Taylor and Ben Bergen and Cameron Jones",
note = "Publisher Copyright: {\textcopyright} 2025 International Conference on Computational Linguistics.; 1st Workshop on GenAI Content Detection, GenAIDetect 2025 ; Conference date: 19-01-2025",
year = "2025",
language = "English",
series = "Proceedings - International Conference on Computational Linguistics, COLING",
publisher = "Association for Computational Linguistics (ACL)",
pages = "96--110",
editor = "Firoj Alam and Preslav Nakov and Nizar Habash and Iryna Gurevych and Iryna Gurevych and Shammur Chowdhury and Artem Shelmanov and Yuxia Wang and Ekaterina Artemova and Mucahid Kutlu and George Mikros",
booktitle = "GenAIDetect 2025 - Proceedings of the 1st Workshop on GenAI Content Detection, Proceedings of the Workshop - 31st International Conference on Computational Linguistics, COLING 2025",
}