<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "http://jats.nlm.nih.gov/publishing/1.3/JATS-journalpublishing1-3.dtd">
<article article-type="research-article" dtd-version="1.3" xml:lang="en" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<processing-meta>
<custom-meta-group content-type="composition">
<custom-meta specific-use="newgen" xlink:href="https://www.newgen.co/">
<meta-name>Composition Vendor</meta-name>
<meta-value>Newgen KnowledgeWorks (P) Ltd.</meta-value>
</custom-meta>
</custom-meta-group>
</processing-meta>
<front>
<journal-meta>
<journal-id journal-id-type="nlm-ta">PLoS One</journal-id>
<journal-id journal-id-type="publisher-id">plos</journal-id>
<journal-id journal-id-type="pmc">plosone</journal-id>
<journal-title-group>
<journal-title>PLOS One</journal-title>
</journal-title-group>
<issn pub-type="epub">1932-6203</issn>
<publisher>
<publisher-name>Public Library of Science</publisher-name>
<publisher-loc>San Francisco, CA USA</publisher-loc>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.1371/journal.pone.0347757</article-id>
<article-id pub-id-type="publisher-id">PONE-D-25-15257</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Research Article</subject>
</subj-group>
<subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Neuroscience</subject><subj-group><subject>Cognitive science</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Research and analysis methods</subject><subj-group><subject>Research design</subject><subj-group><subject>Survey research</subject><subj-group><subject>Surveys</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Physical sciences</subject><subj-group><subject>Mathematics</subject><subj-group><subject>Statistics</subject><subj-group><subject>Statistical data</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Psychology</subject><subj-group><subject>Behavior</subject><subj-group><subject>Imitation</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Social sciences</subject><subj-group><subject>Psychology</subject><subj-group><subject>Behavior</subject><subj-group><subject>Imitation</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Physical sciences</subject><subj-group><subject>Mathematics</subject><subj-group><subject>Probability theory</subject><subj-group><subject>Probability distribution</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Neuroscience</subject><subj-group><subject>Cognitive science</subject><subj-group><subject>Cognitive psychology</subject><subj-group><subject>Language</subject></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Psychology</subject><subj-group><subject>Cognitive psychology</subject><subj-group><subject>Language</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Social sciences</subject><subj-group><subject>Psychology</subject><subj-group><subject>Cognitive psychology</subject><subj-group><subject>Language</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Social sciences</subject><subj-group><subject>Linguistics</subject><subj-group><subject>Sociolinguistics</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Physical sciences</subject><subj-group><subject>Mathematics</subject><subj-group><subject>Probability theory</subject><subj-group><subject>Probability density</subject></subj-group></subj-group></subj-group></subj-group></article-categories>
<title-group>
<article-title>LLM-impersonated debate contributions are more authentic, relevant and coherent than their original: A representative study using BBC1’s Question Time</article-title>
<alt-title alt-title-type="running-head">LLM-impersonated debate contributions are more authentic, relevant and coherent than their original</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes" xlink:type="simple">
<contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0001-9765-2803</contrib-id>
<name name-style="western">
<surname>Herbold</surname>
<given-names>Steffen</given-names>
</name>
<role content-type="http://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role content-type="http://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role content-type="http://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role content-type="http://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role content-type="http://credit.niso.org/contributor-roles/project-administration/">Project administration</role>
<role content-type="http://credit.niso.org/contributor-roles/resources/">Resources</role>
<role content-type="http://credit.niso.org/contributor-roles/visualization/">Visualization</role>
<role content-type="http://credit.niso.org/contributor-roles/writing-original-draft/">Writing – original draft</role>
<role content-type="http://credit.niso.org/contributor-roles/writing-review-editing/">Writing – review &amp; editing</role>
<xref ref-type="corresp" rid="cor001">*</xref>
<xref ref-type="aff" rid="aff001"/>
</contrib>
<contrib contrib-type="author" xlink:type="simple">
<name name-style="western">
<surname>Trautsch</surname>
<given-names>Alexander</given-names>
</name>
<role content-type="http://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role content-type="http://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role content-type="http://credit.niso.org/contributor-roles/software/">Software</role>
<role content-type="http://credit.niso.org/contributor-roles/writing-review-editing/">Writing – review &amp; editing</role>
<xref ref-type="aff" rid="aff001"/>
</contrib>
<contrib contrib-type="author" xlink:type="simple">
<name name-style="western">
<surname>Kikteva</surname>
<given-names>Zlata</given-names>
</name>
<role content-type="http://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role content-type="http://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role content-type="http://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role content-type="http://credit.niso.org/contributor-roles/writing-original-draft/">Writing – original draft</role>
<role content-type="http://credit.niso.org/contributor-roles/writing-review-editing/">Writing – review &amp; editing</role>
<xref ref-type="aff" rid="aff001"/>
</contrib>
<contrib contrib-type="author" xlink:type="simple">
<contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-5901-9633</contrib-id>
<name name-style="western">
<surname>Hautli-Janisz</surname>
<given-names>Annette</given-names>
</name>
<role content-type="http://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role content-type="http://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role content-type="http://credit.niso.org/contributor-roles/project-administration/">Project administration</role>
<role content-type="http://credit.niso.org/contributor-roles/resources/">Resources</role>
<role content-type="http://credit.niso.org/contributor-roles/writing-review-editing/">Writing – review &amp; editing</role>
<xref ref-type="aff" rid="aff001"/>
</contrib>
</contrib-group>
<aff id="aff001"><addr-line>Department of Computer Science and Mathematics, University of Passau, Passau, Germany</addr-line></aff>
<contrib-group>
<contrib contrib-type="editor" xlink:type="simple">
<name name-style="western">
<surname>Hassan</surname>
<given-names>Mohammad Salah</given-names>
</name>
<role>Editor</role>
<xref ref-type="aff" rid="edit1"/></contrib>
</contrib-group>
<aff id="edit1"><addr-line>A Sharqiyah University, OMAN</addr-line></aff>
<author-notes>
<corresp id="cor001">* E-mail: <email xlink:type="simple">steffen.herbold@uni-passau.de</email></corresp>
<fn fn-type="conflict" id="coi001">
<p>The authors have declared that no competing interests exist.</p>
</fn>
</author-notes>
<pub-date pub-type="epub"><day>1</day><month>7</month><year>2026</year></pub-date>
<pub-date pub-type="collection"><year>2026</year></pub-date>
<volume>21</volume>
<issue>7</issue>
<elocation-id>e0347757</elocation-id>
<history>
<date date-type="received"><day>24</day><month>3</month><year>2025</year></date>
<date date-type="accepted"><day>1</day><month>4</month><year>2026</year></date>
</history>
<permissions>
<copyright-year>2026</copyright-year>
<copyright-holder>Herbold et al</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">
<license-p>This is an open access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">Creative Commons Attribution License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited.</license-p></license>
</permissions>
<self-uri content-type="pdf" xlink:href="info:doi/10.1371/journal.pone.0347757"/>
<abstract>
<p>Generative AI has the potential to pollute the public information sphere with made-up content, posing a significant threat to the cohesion of societies at large. This paper offers the first large-scale and systematic study of how authentic, relevant and coherent impersonated content from Large Language Models (LLMs) is perceived by the general public. Based on a cross-section of British society, we show that LLM-generated responses to questions drawn from a broadcast political debate programme in the UK are judged to be more authentic and relevant than the original responses given by the panel members who were impersonated. We also show that stylistic differences do not influence these judgments, meaning that the distinction of original and generated content is challenging for the general public. Taken together, this means that LLMs can be made to deceive the public regarding the nature of statements in the political domain, with the consequence that there is a dire need to inform the general public of the potential harm this can have on society.</p>
</abstract>
<funding-group>
<award-group id="award001">
<funding-source>
<institution-wrap>
<institution-id institution-id-type="funder-id">http://dx.doi.org/10.13039/501100001663</institution-id>
<institution>Volkswagen Foundation</institution>
</institution-wrap>
</funding-source><award-id>98544</award-id>
<principal-award-recipient><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-5901-9633</contrib-id><name name-style="western">
<surname>Hautli-Janisz</surname><given-names>Annette</given-names></name></principal-award-recipient></award-group>
<funding-statement>A.H. work was partially funded by the VolkswagenStiftung under grant Az. 98544 ‘Deliberation Laboratory’ URL: <ext-link ext-link-type="uri" xlink:href="https://www.volkswagenstiftung.de/" xlink:type="simple">https://www.volkswagenstiftung.de/</ext-link> The funders had no role in study design, data collection and analysis, decision to publish, or preparation of the manuscript.</funding-statement>
</funding-group>
<counts>
<fig-count count="6"/>
<table-count count="1"/>
<page-count count="16"/>
</counts>
<custom-meta-group>
<custom-meta id="data-availability">
<meta-name>Data Availability</meta-name>
<meta-value>All materials are available online in the form of a replication package that contains the data and the analysis code at <ext-link ext-link-type="uri" xlink:href="https://github.com/aieng-lab/replication-kit-qtgpt-study" xlink:type="simple">https://github.com/aieng-lab/replication-kit-qtgpt-study</ext-link> and <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.5281/zenodo.12698363" xlink:type="simple">https://doi.org/10.5281/zenodo.12698363</ext-link>.</meta-value>
</custom-meta>
</custom-meta-group>
</article-meta>
</front>
<body>
<sec id="sec001" sec-type="intro">
<title>Introduction</title>
<p>Modern Generative Artificial Intelligence models (GenAI) like GPT [<xref ref-type="bibr" rid="pone.0347757.ref001">1</xref>], Claude [<xref ref-type="bibr" rid="pone.0347757.ref002">2</xref>], and Gemini [<xref ref-type="bibr" rid="pone.0347757.ref003">3</xref>] have been shown to generate high-quality textual content including source code [<xref ref-type="bibr" rid="pone.0347757.ref004">4</xref>], persuasive student essays [<xref ref-type="bibr" rid="pone.0347757.ref005">5</xref>], and legal analysis [<xref ref-type="bibr" rid="pone.0347757.ref006">6</xref>]. They also seem (at least partly) to be able to role-play [<xref ref-type="bibr" rid="pone.0347757.ref007">7</xref>], to mimic the linguistic pattern of authors [<xref ref-type="bibr" rid="pone.0347757.ref008">8</xref>], to generate content that reflects general political identity [<xref ref-type="bibr" rid="pone.0347757.ref009">9</xref>,<xref ref-type="bibr" rid="pone.0347757.ref010">10</xref>] and to assume the role of a domain expert who responds accordingly [<xref ref-type="bibr" rid="pone.0347757.ref011">11</xref>]. Whereas developments like these are already questionable in terms of moral and ethical responsibility, Large Language Models (LLMs) influencing the political opinion of humans [<xref ref-type="bibr" rid="pone.0347757.ref012">12</xref>], purporting political bias [<xref ref-type="bibr" rid="pone.0347757.ref013">13</xref>] and successfully generating targeted persuasive communication [<xref ref-type="bibr" rid="pone.0347757.ref014">14</xref>,<xref ref-type="bibr" rid="pone.0347757.ref015">15</xref>] clearly signals that the technological advancement is reaching a critical stage where harm to societies is only a prompt away. This is solidly supported by the findings of this paper.</p>
<p>In our study we condition an LLM in such a way that it impersonates well-known political and societal personalities in a political talk show on UK national television. Specifically, the LLM is tasked to respond to the audience questions in the show as an impersonation of the original speakers on the show. Based on a representative cross-section of British society (n = 948), we study how UK citizens rate the actual response of the person in comparison to the impersonated response along three axes: authenticity (the likelihood that the impersonated response comes from the actual person), coherence (the logical flow of the response), and relevance (the extent to which the response is relevant to the question). We also compare how content and linguistic properties differ between original and impersonated statements and elicit the openness of citizens to using AI technology for generating contribution to public debates. These three lines of research answer the following research questions:</p>
<list list-type="simple">
<list-item>
<p><bold>RQ1:</bold> To what extent do UK citizens rate the authenticity, coherence, and relevance of impersonated debate responses differently from actual debate responses by that person?</p>
</list-item>
<list-item>
<p><bold>RQ2:</bold> To what extent does a difference in content and linguistic style between actual and impersonated debate responses impact their authenticity ratings?</p>
</list-item>
<list-item>
<p><bold>RQ3:</bold> What is the general public’s view on using AI in public debates and is this view affected by exposure to technology?</p>
</list-item>
</list>
<p>The data underlying our study originates from 30 episodes of BBC1’s Question Time, one of the most viewed political debate programmes in the UK, which were broadcast from 2020 to 2022 [<xref ref-type="bibr" rid="pone.0347757.ref016">16</xref>]. The survey participants are asked to (i) attribute both actual and impersonated responses to public persons; (ii) evaluate how coherent and relevant both actual and impersonated responses are; and (iii) express their opinion regarding the use of AI in public debates. The last task is split into two sub-tasks: First, the participants give their opinions on AI unaware of the source of the material they just rated (actual vs. impersonated). In the next step, they are shown the source of their rated content, and they express their opinion on the technology again.</p>
</sec>
<sec id="sec002" sec-type="materials|methods">
<title>Materials and methods</title>
<sec id="sec003">
<title>Data</title>
<sec id="sec004">
<title>Original debate content.</title>
<p>The original debate data is from ‘Question Time’ (QT), one of the most-viewed political talk shows on UK television. QT30 [<xref ref-type="bibr" rid="pone.0347757.ref016">16</xref>] is a collection of 30 episodes of QT aired on BBC1 between June 2020 and November 2021, currently the largest dataset of analysed broadcast political debate. QT features a moderated panel format, where well-known members of politics and society sit on a panel and respond to questions from the audience on the current topics of the week. The panellists are directed by the moderator and are asked to respond to the questions independent of a prior conversation on the topic and the initial statements by other panellists. The panel members featured in the dataset belong to one of six categories: politicians (50%), business people (16.67%), journalists (14.17%), medical experts (6.67%), writers (5.83%), and other well-known members of UK society (activists, actors, political experts and sports personalities – 6.67%). The use of this data for research is covered by Section 29 (1) of the British Copyright, Designs and Patents Act (CDPA). The data cannot be shared publicly and will be shared with other researchers for the sole use of research upon request (see data availability statement).</p>
<p>From QT30 we manually extract a total of 119 unique questions with 555 responses from 119 different speakers. We discard the responses of seven speakers who do not have a Wikipedia page, a requirement for generating the impersonated debate responses. At the same time, this criterion serves as a filter to determine whether the panel members are well-known personalities in the public sphere. We also discard one response where the corpus data does not contain information about the speaker. This yields a set of 527 valid question/response pairs from 112 different speakers. We randomly drop seven responses to achieve a final count of 520 question/response pairs to facilitate easier sampling. A manual check of the resulting data ensures that the responses are understandable without context and do not refer to other panel members’ previous contributions, as this could affect judgments of authenticity and relevance.</p>
</sec>
<sec id="sec005">
<title>Impersonated debate content.</title>
<p>To generate the impersonated responses, we use GPT-4 Turbo [<xref ref-type="bibr" rid="pone.0347757.ref001">1</xref>]. While more recent models, e.g., Opus Claude [<xref ref-type="bibr" rid="pone.0347757.ref002">2</xref>], seem to be slightly better at logical tasks like mathematics, we are not aware of any benchmark where GPT was significantly outscored in tasks that involve common knowledge (as shown, for instance, in [<xref ref-type="bibr" rid="pone.0347757.ref017">17</xref>]). For prompting, we use a complex emulation protocol similar to Bhandarkar et al. [<xref ref-type="bibr" rid="pone.0347757.ref008">8</xref>] with the following prompts:</p>
<list list-type="bullet">
<list-item>
<p>System prompt: <monospace>You are an expert at mimicking different persons in debates. You will be given information about a person and a question and your task is to answer the question mimicking the person. You only answer as the person you are asked to mimic. Do not say the name of the person you are mimicking. Do not introduce yourself. Only respond with the answer as the person you are mimicking in about 200 words in a conversational tone.</monospace></p>
</list-item>
<list-item>
<p>User prompt: <monospace>Please only answer this question: [QUESTION] as this person: [SPEAKER_WIKIPEDIA]. Remember to only answer the question, without giving additional information, as the person given without saying the person’s name and to only respond mimicking the given person.</monospace></p>
</list-item>
</list>
<p>The system prompt defines the behaviour we expect from the model, i.e., it is tasked to mimic a well-known person, to be brief in the response to the question and to use a conversational tone. The user prompt defines the task, provides the question and adds the short biography of the speaker that we obtain from the first paragraph of their Wikipedia article (this paragraph provides a summary of the information on their origin, career, party affiliation, political offices etc.) The user prompt also repeats the task for the model.</p>
<p>After receiving the responses, we conduct a manual sanity check to ensure that the impersonated responses adhere to the guidelines given in the prompt, i.e., do not contain the name of the speaker, any information that the response was generated by an LLM or a reason why no response was possible (for instance, due to lack of access to real-time data or for ethical reasons). This check did not flag problematic content, meaning that all responses were used in the subsequent study.</p>
</sec>
</sec>
<sec id="sec006">
<title>Study participants</title>
<p>Since the debate content we study originates from one of the most popular British topical debate programs, we recruit a representative sample of British citizens above the age of 18 using the online platform Prolific. The recruitment time frame ran from June 18th until June 26th, 2024. Participants are informed about the purpose of the study, consent to participate and receive a participation fee for compensation which lies above UK minimum wage. We recruit a total of 948 participants, who are distributed randomly across the different tracks that the study comprises, resulting in at least two judgments for each of the 1560 question/response pairs (520 original question/original response pairs, 520 original question/impersonated response pairs, and 520 pairs of original question/original responses with random speaker) in each track.</p>
</sec>
<sec id="sec007">
<title>Study design</title>
<sec id="sec008">
<title>Measures and variables.</title>
<p>To evaluate the perception of original and impersonated debate content, we elicit judgments on the following measures:</p>
<list list-type="bullet">
<list-item>
<p><italic>Authenticity</italic>: The likelihood that the response is an actual and real utterance by the speaker in a debate, i.e., the degree to which this utterance could have been contributed by that speaker in a debate. This variable measures the core aspect of our study, i.e., if people believe that a statement is genuine.</p>
</list-item>
<list-item>
<p><italic>Coherence</italic>: The logical flow of the response. This variable measures the internal reasoning structure of the response.</p>
</list-item>
<list-item>
<p><italic>Relevance</italic>: The extent to which the response addresses the question. This variable measures if the response stays on topic and conveys relevant information.</p>
</list-item>
<list-item>
<p><italic>Content</italic>: The extent to which the overall meaning of original and impersonated response is identical. This variable allows us to understand if LLM-generated responses differ from the actual responses.</p>
</list-item>
<list-item>
<p><italic>Confidence</italic>: The confidence in judging whether the response was given by a specific speaker. This variable is used as a control variable to understand if the certainty in judging debate content is affected by whether it is original or impersonated.</p>
</list-item>
<list-item>
<p><italic>Familiarity</italic>: The knowledge on a panel member based on their previous public appearances. This variable is used as a control variable to understand if familiarity with a speaker has an impact on the authenticity judgments.</p>
</list-item>
</list>
<p>For all measures we use a five-point Likert scale (for instance, ‘not authentic’ to ‘very authentic’) such that the middle point of the scale is neutral. The supplementary material contains the full description of the measures. The judgments are collected using an online survey, the design of which we present in the following.</p>
</sec>
<sec id="sec009">
<title>Study procedure.</title>
<p>Our custom-made online survey starts with the collection of demographic data on the participants, i.e., their age, gender, country of residence within the United Kingdom, and political preference. At this stage, the participants are only informed that the debate questions and responses are taken from the BBC1 show ‘Question Time’. They are not aware that some of the responses are generated by an LLM (this information is disclosed only at a later stage of the survey). This deceptive design prompts participants to believe they are judging actual debate content.</p>
<p>The participants are then randomly sampled into three tracks. <italic>Track1</italic> measures the perception of the authenticity, coherence, and relevance of a single debate response given a question. The participants are shown the question, one response, and the name of the speaker. The response is either the original response by the speaker or an LLM-generated, impersonated response, as described above. <italic>Track2</italic> augments this setting by showing original and impersonated response side-by-side: the participants see the question, the name of the speaker and both responses at the same time. Their task is to compare the responses in terms of authenticity, coherence, and relevance to the question. We remove position bias by randomly positioning original and impersonated responses on the left or the right side of the page. Track2 also measures whether the content of the impersonated responses is the same as that of the actual responses.</p>
<p>With <italic>Track3</italic> we try to understand better the factors that lead to differences in authenticity: Here, participants are shown a question, a response, the name of the speaker, and the speaker’s short biography. The biography is the same one that we provide to the LLM as part of the user prompt. For statistical reasons, we create three populations of participants. The first population sees the question, the actual speaker, their biography and the actual response. The second population sees the question, the actual speaker, their biography and the impersonated response. And the third population sees the question, the actual response from the actual speaker, but a randomly selected different public person from our data set, plus that person’s biography. All participants are asked to judge the authenticity of the responses, to rate their confidence in the judgment and to rate their familiarity with the speaker.</p>
<p>Once the participants complete their respective tracks, they participate in an exit poll. Here we ask questions regarding their familiarity with AI and chatbots, their opinion on the use of AI in public debates, and the perceived need for transparency and regulation in this setting (see more details in the supplemental material <xref ref-type="supplementary-material" rid="pone.0347757.s001">S1 Appendix</xref>). Only after the exit poll is completed, we reveal to the participants which of their judged debate responses were generated by an LLM, together with their ratings. In light of this new information, we repeat the exit poll for all participants. This allows us to see whether more insight into the quality of the generated responses affects their opinions regarding the use of AI in public debates. As a last step, the participants are then invited to provide an (optional) free-text comment regarding their answers in the exit poll.</p>
<p>Overall, each participant judges eight different responses. For Track1 and Track3 this means that each participant judges eight question/response pairs (we use rejection sampling to ensure each question/speaker pair only appears once, i.e., it is not possible for a participant to judge both the actual and the impersonated responses from a speaker to a question). For Track2 this means that each participant judges four pairs of question/original plus impersonated response. Additional details about the survey, including the exact wording of the questions, are provided in supplemental material <xref ref-type="supplementary-material" rid="pone.0347757.s001">S1 Appendix</xref>.</p>
</sec>
</sec>
<sec id="sec010">
<title>Stylistic comparison</title>
<p>We also measure stylistic differences between the original and the impersonated responses by comparing discourse-related linguistic patterns. This allows us (1) to understand if the responses share properties on the linguistic surface and (2) whether the language is related to human judgments in terms of authenticity, coherence, and relevance. The study is based on the following linguistic properties:</p>
<list list-type="bullet">
<list-item>
<p><italic>Syntactic complexity</italic>: Syntactic complexity in terms of the mean number of conjuncts, clausal modifiers of nouns, adverbial clause modifiers, clausal complements, clausal subjects and parataxis per sentence as an approximation of language complexity. [<xref ref-type="bibr" rid="pone.0347757.ref018">18</xref>]</p>
</list-item>
<list-item>
<p><italic>Nominalisations</italic>: The number of nominalisations per sentence as an approximation of the of abstractness of the language.</p>
</list-item>
<list-item>
<p><italic>Modals</italic>: The number of modal constructions (e.g., ‘definitely’, ‘potentially’) per sentence as a signal of the stance of the speakers towards their utterances. [<xref ref-type="bibr" rid="pone.0347757.ref019">19</xref>]</p>
</list-item>
<list-item>
<p><italic>Discourse markers</italic>: The number of discourse markers (e.g., ‘first’, ‘moreover’) per sentence as an approximation of the coherence of text and the use of explicit argumentative structure. [<xref ref-type="bibr" rid="pone.0347757.ref020">20</xref>]</p>
</list-item>
<list-item>
<p><italic>Epistemic markers</italic>: The number of epistemic markers (e.g., ‘I think’, ‘in my opinion’) as an indication of the commitment of a speaker to the message they convey.</p>
</list-item>
<list-item>
<p><italic>Lexical diversity</italic>: Lexical diversity measured with MTLD [<xref ref-type="bibr" rid="pone.0347757.ref021">21</xref>] as an approximation of the diversity of the used vocabulary.</p>
</list-item>
<list-item>
<p><italic>Lexical overlap</italic>: The percentage of words in the question (excluding stop words) that also appear in the response as an approximation of the influence of the question on the response.</p>
</list-item>
</list>
<p>These stylistic features are automatically extracted from the original and the generated responses using a combination of rule-based, stochastic and neural models for natural language processing. Modals, discourse markers, epistemic markers, nominalisations, and the lexical markers are normalised by the number of sentences within a response.</p>
</sec>
<sec id="sec011">
<title>Statistical analysis</title>
<p>The inter-rater reliability between the two judgments for authenticity, coherence, relevance and content is measured with Cronbach’s <inline-formula id="pone.0347757.e001"><alternatives><graphic id="pone.0347757.e001g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0347757.e001" xlink:type="simple"/><mml:math display="inline" id="equation1"><mml:mrow><mml:mi>α</mml:mi></mml:mrow></mml:math></alternatives></inline-formula> [<xref ref-type="bibr" rid="pone.0347757.ref022">22</xref>]. Additionally, we report pair-wise differences between the judgments to quantify the disagreement between participants. We exclude confidence and familiarity because we cannot expect agreement regarding a subjective self-reflection. For the subsequent statistical analysis, we map the Likert scales to the integers [−2, −1, 0, 1, 2] and compute the average rating between the two judgments for the same data point. Since the variables from our survey are based on Likert scales, we use non-parametric rank-based statistical tests.</p>
<p>In Track1 we assess the difference in authenticity, coherence, and relevance between the original responses and the impersonated responses. The track has a between-subjects design (i.e., actual and impersonated responses are rated by different participants) with data that is paired by the question and the speaker. Consequently, we use a two-sided Wilcoxon signed rank test [<xref ref-type="bibr" rid="pone.0347757.ref023">23</xref>] to determine if the difference between both populations (actual versus impersonated responses) is significant.</p>
<p>Track2 uses a within-subjects design (i.e., one participant judges both the actual and the impersonated response). We conduct a two-sided one-sample Wilcoxon signed rank test to determine if the judgments regarding authenticity, coherence, relevance, and content are significantly different from zero. For authenticity, coherence, and relevance, a significant tendency towards negative values means that the participants favour the original responses; a significant tendency towards positive values means that the generated, impersonated responses are favoured. Regarding content, a significant positive value means that the content between original and impersonated responses is similar; a negative value means that the impersonated content is different from the actual content by the speaker. We post-process the data such that the original response is always on the left and the impersonated response is always on the right.</p>
<p>With the data from Track3 we assess if the authenticity and the confidence in the rating depend on whether the speaker is real, random, or impersonated. This results in three populations. The track has a between-subjects design where the populations are paired by the question and the actual speaker. To elicit whether there is any difference between the three populations, we use a Friedman test [<xref ref-type="bibr" rid="pone.0347757.ref024">24</xref>] with a Bonferroni-Dunn post-hoc test based on pair-wise two-sided Wilcoxon signed rank tests. This determines which differences between pairs are significant. Additionally, we use familiarity judgments to understand how this affects authenticity. For this, we conduct a subgroup analysis where we split the ratings into those where the familiarity is less than 0 (i.e., ratings where the participants are not/to a limited extent/fairly familiar with the speaker) and judgments with a familiarity greater than or equal to 0 (speakers are somewhat familiar/familiar with the speaker). For the latter subgroup, we do not have paired data anymore, because we have independent raters for the three populations. For instance, the raters for the responses attributed to the actual speakers may be familiar with different speakers than the raters for the impersonated responses, leading to different subgroups. We therefore use a Kruskal-Wallis test [<xref ref-type="bibr" rid="pone.0347757.ref025">25</xref>] with Bonferroni-Dunn post-hoc tests based on pair-wise two-sided Wilcoxon–Mann–Whitney tests [<xref ref-type="bibr" rid="pone.0347757.ref026">26</xref>].</p>
<p>The statistical analysis of the linguistic surface markers is similar to the analysis of Track1 since we also have two populations for each linguistic marker (actual versus impersonated responses). Since the data from the linguistic markers does not follow a normal distribution (visual analysis of the distribution in <xref ref-type="fig" rid="pone.0347757.g005">Fig 5</xref> shows, for instance, long tails), we also use non-parametric tests, namely two-sided Wilcoxon signed rank tests to determine if differences for each of the seven linguistic markers are significant.</p>
<fig id="pone.0347757.g001" position="float"><object-id pub-id-type="doi">10.1371/journal.pone.0347757.g001</object-id><label>Fig 1</label><caption><title>Judgments when a debate question, the name of the speaker, and either the GPT-generated or the actual response by the actual speaker are shown.</title><p>Violins show a kernel density estimation of the probability distribution, the miniature box-plots depict the median, upper and lower quartiles, and the whiskers the largest/smallest value observed within 1.5 times the interquartile range of the upper/lower quartile. The statistical markers reported are the p-value of two-sided Wilcoxon signed rank tests, the effect size <italic>r</italic>, the sample sizes <italic>n</italic>, mean values <italic>M</italic> and standard deviations <italic>SD</italic>.</p></caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0347757.g001" xlink:type="simple"/></fig>
<fig id="pone.0347757.g002" position="float"><object-id pub-id-type="doi">10.1371/journal.pone.0347757.g002</object-id><label>Fig 2</label><caption><title>Judgments when a debate question, the name of the speaker, and both the actual and GPT-generated responses are shown side-by-side.</title><p>The stacked bar chart reports the percentages of the ratings that we observed. The statistical markers reported are the p-value of a two-sided one-sample Wilcoxon signed rank tests for a difference from zero, the effect size <italic>r</italic>, the sample sizes <italic>n</italic>, mean values <italic>M</italic> and standard deviations <italic>SD</italic>.</p></caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0347757.g002" xlink:type="simple"/></fig>
<fig id="pone.0347757.g003" position="float"><object-id pub-id-type="doi">10.1371/journal.pone.0347757.g003</object-id><label>Fig 3</label><caption><title>Judgments when a debate question with either the response and biography from the actual speaker, the GPT-generated response and the biography of the actual speaker, or the response from the actual speaker but the name and biography of a random public person are shown.</title><p>Violins show a kernel density estimation of the probability distribution, the miniature box-plots depict the median, upper and lower quartiles, and the whiskers the largest/smallest value observed within 1.5 times the interquartile range of the upper/lower quartile. The statistical markers reported are the p-value of the omnibus test for differences and pair-wise Bonferroni-Dunn correct two-sided post-hoc tests, the effect size <italic>r</italic>, the sample sizes <italic>n</italic>, mean values <italic>M</italic> and standard deviations <italic>SD</italic>.</p></caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0347757.g003" xlink:type="simple"/></fig>
<fig id="pone.0347757.g004" position="float"><object-id pub-id-type="doi">10.1371/journal.pone.0347757.g004</object-id><label>Fig 4</label><caption><title>Judgments whether the content of the actual response and the GPT-generated response are the same.</title><p>The actual and impersonated responses are shown side-by-side. The stacked bar chart reports the percentages of the ratings that we observed. The statistical markers reported are the p-value of a two-sided one-sample Wilcoxon signed rank tests for a difference from zero, the effect size <italic>r</italic>, the sample sizes <italic>n</italic>, mean values <italic>M</italic> and standard deviations <italic>SD</italic>.</p></caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0347757.g004" xlink:type="simple"/></fig>
<p>Thus, we conduct three statistical tests with the data of Track1, four with the data of Track2, three with the data of Track3 and seven statistical tests for the linguistic markers, i.e., a total of 17 tests. We use a conservative approach based on Bonferroni correction [<xref ref-type="bibr" rid="pone.0347757.ref027">27</xref>] to account for multiple tests and consider results as significant if the p-value of a test is less than <inline-formula id="pone.0347757.e002"><alternatives><graphic id="pone.0347757.e002g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0347757.e002" xlink:type="simple"/><mml:math display="inline" id="equation2"><mml:mrow><mml:mi>α</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>0.05</mml:mn></mml:mrow><mml:mrow><mml:mn>17</mml:mn></mml:mrow></mml:mfrac><mml:mo>≈</mml:mo><mml:mn>0.003</mml:mn></mml:mrow></mml:math></alternatives></inline-formula>. Based on the large size of our populations with 520 question/response pairs and assuming that we observe differences of 0.5 points (i.e., half a step on the Likert scales), we compute the expected statistical power as <inline-formula id="pone.0347757.e003"><alternatives><graphic id="pone.0347757.e003g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0347757.e003" xlink:type="simple"/><mml:math display="inline" id="equation3"><mml:mrow><mml:mi>β</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:math></alternatives></inline-formula>. Consequently, differences of 0.5 or larger in judgment should always be picked up by our tests and if there are no differences in judgments, there is only a 5% chance that we find a difference that is not there.</p>
<p>We report the arithmetic mean (M) and standard deviation (SD) as statistical markers for the populations. While our data is not perfectly normal, it also does not have severe outliers or multimodalities, so we prefer the clear interpretation of the arithmetic mean (M) and standard deviation (SD) to report statistical markers for populations. We use the biserial rank correlation <italic>r</italic> [<xref ref-type="bibr" rid="pone.0347757.ref028">28</xref>] to measure the effect size. Violin plots visualise the distribution of the data based on a kernel density estimation of the underlying probability distribution. The violins include miniature box plots that depict the median, upper and lower quartiles and the whiskers defined as the largest/smallest observed value at most 1.5 times the inter-quartile range away from the upper/lower quartile. Additionally, we use stacked bar charts to depict ratios of Likert scale items, where appropriate.</p>
<p>Supplemental material <xref ref-type="supplementary-material" rid="pone.0347757.s002">S2 Appendix</xref> provides details regarding the results, e.g., the demographic information of the participants. Supplemental material <xref ref-type="supplementary-material" rid="pone.0347757.s003">S3 Appendix</xref> reports on the results of additional quantitative analysis of possible confounding factors, i.e., randomisation of speakers assigned in Track3, the lengths of responses, and the influence of spelling errors. The results for all these factors are negative, i.e., it is highly unlikely that these aspects serve as alternative explanations of our findings.</p>
<p>The statistical analysis of the data is mostly implemented in Python. We use pandas 2.2.2 and numpy 1.26.4 for processing the data, pingouin 0.5.4 for the calculation of Cronbach’s <inline-formula id="pone.0347757.e004"><alternatives><graphic id="pone.0347757.e004g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0347757.e004" xlink:type="simple"/><mml:math display="inline" id="equation4"><mml:mrow><mml:mi>α</mml:mi></mml:mrow></mml:math></alternatives></inline-formula>, effect sizes <italic>r</italic> and pair-wise tests, scipy 1.13.0 for the omnibus tests, and seaborn 0.13.2 for the generation of plots. We compute the statistical power with the R package mkpower 0.9.</p>
</sec>
<sec id="sec012">
<title>Qualitative analysis</title>
<p>The qualitative analysis in this study is two-fold: On the one hand, we code the free-text answers from the exit poll of the survey using inductive coding [<xref ref-type="bibr" rid="pone.0347757.ref029">29</xref>] and have one author assign one or more codes to each answer. The codes are aimed to capture the intent of the free-text answer, e.g., convey the reason for changes in the exit poll or observations regarding the impersonated content that the participants found striking (details on the codes are in supplemental material). This is initially done for twenty answers, at which point the coding is checked by and discussed with a second author, resulting in an agreed-upon coding taxonomy. The first author then continues to code the remainder of the data. Upon completion of this coding, the second author again checks all codes and discusses the coding to achieve agreement in the same manner as for the initial set of codes. We then conduct one round of axial coding [<xref ref-type="bibr" rid="pone.0347757.ref030">30</xref>] to group related codes into categories. Same as above, the axial coding is initially conducted by one author, then checked by and discussed with a second author to achieve agreement. We do not report inter-rater agreements for this data, because all data is fully checked by two authors and all disagreements are discussed and resolved, which results in final coding on which both annotators perfectly agree.</p>
<p>On the other hand, we conduct a qualitative analysis to better understand the human judgments regarding the differences in content between actual and impersonated responses, with a specific focus on whether the stance between actual and impersonated responses differs. For this, we randomly sample 50 pairs of actual and generated responses that were judged to be different in content by the study participants. We use deductive coding to determine if (a) the two responses have the same stance, i.e., arrive at the same conclusion, argue for the same points or take the same side, (b) express a different stance, i.e., draw different conclusions or argue for another position, or (c) whether this distinction is not applicable, e.g., because the question does not require the hearers to take a stance. Additionally, we determine for each response whether it addresses the question. Similar to the coding procedure above, the coding was first done by the one author and then checked by a second author. Disagreement between the two annotators was minimal and adjudicated to achieve agreement.</p>
</sec>
<sec id="sec013">
<title>Ethics statement</title>
<p>Our study used human subjects as participants in a survey. The ethics committee of the University of Passau approved this study (ref. III/Herbold.I-07.5095/240229). The study makes use of an online survey that obtained informed written consent prior to participation in the study.</p>
</sec>
</sec>
<sec id="sec014" sec-type="results">
<title>Results</title>
<sec id="sec015">
<title>Impersonated responses are perceived as more authentic, coherent and relevant</title>
<p>The results clearly show that LLM-generated, impersonated content is judged as more authentic, coherent, and relevant than the actual debate responses. When the participants only see one question and its response (either actual or impersonated) (see <xref ref-type="fig" rid="pone.0347757.g001">Fig 1</xref>), we observe a significant difference across all three dimensions with a large effect size for authenticity (<italic>r</italic> = −0.55), relevance (<italic>r</italic> = −0.82) and coherence (<italic>r</italic> = −0.84) in the responses. When the participants directly compare an impersonated response with an original response along these dimensions (see <xref ref-type="fig" rid="pone.0347757.g002">Fig 2</xref>), the results are supported: The effect sizes for relevance (<italic>r</italic> = 0.79) and coherence (<italic>r</italic> = 0.87) remain large, but the difference in authenticity decreases to a weak effect (<italic>r</italic> = 0.28), so the gap between impersonated responses and actual responses is smaller in this setting (but it is still statistically significant). If the participants see the biographies of the speakers during rating (see <xref ref-type="fig" rid="pone.0347757.g003">Fig 3</xref>, top-left), we observe a similar effect size for the authenticity when comparing actual and generated responses than when the questions are shown side-by-side (<italic>r</italic> = 0.25). This means that the authenticity of the impersonated response is still higher than that of the actual response, even if biographies are provided.</p>
<p>An important control in our study is whether the debate content is just generally assumed to be authentic, instead of only when the response matches the common knowledge that the public has about the speaker. The data in <xref ref-type="fig" rid="pone.0347757.g003">Fig 3</xref> shows that when deliberately assigning an actual response to a randomly picked (wrong) speaker, the authenticity is significantly lower compared to when the response is assigned to the actual speaker with the actual response (<italic>r</italic> = 0.39) or the actual speaker with the impersonated response (<italic>r</italic> = 0.59). When we take into account the confidence of the participants in their rating, we do not find any significant differences in the certainty of attributing original or impersonated responses to the actual speaker, or attributing an actual response to a random speaker. If we only consider a subgroup of data where the participants are highly familiar with the speakers, this again neither affects the authenticity nor the confidence: While the significance tests for differences between actual and impersonated responses, as well as actual speakers versus random speakers, are not significant anymore, the distributions are almost exactly the same as for the full dataset. This indicates that there is no shift in distributions, but rather the effects are too small to be detected with the smaller subgroups.</p>
</sec>
<sec id="sec016">
<title>Original content is different from impersonated content</title>
<p>If the authenticity is rated high, but the content of the original response differs from the content of the impersonated response, i.e., the impersonated content is not in line with the actual statements of the person, we are faced with a situation where GenAI can be employed for targeted misinformation about the speaker’s point of view. Our results show that a significant majority of actual responses are judged to be different in content from the impersonated counterparts, though the spread in the distribution is fairly large (see <xref ref-type="fig" rid="pone.0347757.g004">Fig 4</xref>). About half of the responses are considered to be dissimilar in comparison to only about one-third of responses that are considered similar. We observe no notable pattern or correlation between the similarity of the content and the authenticity of the responses (<inline-formula id="pone.0347757.e008"><alternatives><graphic id="pone.0347757.e008g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0347757.e008" xlink:type="simple"/><mml:math display="inline" id="equation5"><mml:mrow><mml:mi>ρ</mml:mi><mml:mo>=</mml:mo><mml:mo>−</mml:mo><mml:mn>0.16</mml:mn></mml:mrow></mml:math></alternatives></inline-formula>).</p>
<p>For the qualitative analysis of the differences, we sample 50 of the total 255 pairs of actual and generated responses for which the content is judged as being dissimilar. <xref ref-type="table" rid="pone.0347757.t001">Table 1</xref> summarizes the results of the analysis. Three categories emerge: The first category comprises cases where the differences in content are about how arguments are presented, but where both the actual and the generated response arrive at the same conclusion, while also addressing the question (<italic>n</italic> = 16 for responses with a stance, <italic>n</italic> = 5 without a stance, 42%). The second category comprises cases where only the actual speaker (<italic>n</italic> = 2, 4%) or only GPT (<italic>n</italic> = 14, 28%) responds to the question and the other source dodges the question. We do not annotate these instances as exhibiting a difference in stance – one side does not elaborate on the topic of the question and therefore does not communicate a stance towards it at all, so the identification of the difference in stance is not applicable. Overall, this means in 32% of cases, the question is dodged by one source. The third (and most important category) captures cases where the stance between the actual response and the generated response is notably different (<italic>n</italic> = 13, 26%). If we extrapolated this to the full dataset, we would expect that around 13% of generated responses communicate a different stance than the actual response.</p>
<table-wrap id="pone.0347757.t001" position="float"><object-id pub-id-type="doi">10.1371/journal.pone.0347757.t001</object-id><label>Table 1</label><caption><title>Results of qualitative analysis of the differences in content.</title></caption>
<alternatives><graphic id="pone.0347757.t001g" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0347757.t001" xlink:type="simple"/><table><colgroup>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
</colgroup>
<thead>
<tr>
<th align="left">Content similarity</th>
<th align="left">Count</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left">Same stance, both address question</td>
<td align="left">16</td>
</tr>
<tr>
<td align="left">Stance not required, only generated response addresses question</td>
<td align="left">14</td>
</tr>
<tr>
<td align="left">Different stance, both address question</td>
<td align="left">13</td>
</tr>
<tr>
<td align="left">Stance not required, both address question</td>
<td align="left">5</td>
</tr>
<tr>
<td align="left">Stance not required, only actual response addresses question</td>
<td align="left">2</td>
</tr>
</tbody>
</table>
</alternatives></table-wrap>
</sec>
<sec id="sec017">
<title>Human judgment is reliable</title>
<p>We measure the inter-rater reliability with Cronbach’s <inline-formula id="pone.0347757.e009"><alternatives><graphic id="pone.0347757.e009g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0347757.e009" xlink:type="simple"/><mml:math display="inline" id="equation6"><mml:mrow><mml:mi>α</mml:mi></mml:mrow></mml:math></alternatives></inline-formula> [<xref ref-type="bibr" rid="pone.0347757.ref022">22</xref>] for the two judgments on authenticity, coherence, relevance, and content that we get for each question/response pair. The ratings are based on a five-point Likert scale. Additionally, we report the pair-wise differences between the two participants to understand which disagreements our participants have. We exclude confidence and familiarity because we cannot expect agreement regarding a subjective self-reflection. For the subsequent statistical analysis, we map the Likert scales to the integers [−2, −1, 0, 1, 2] and compute the average rating between the two judgments for the same data point. Since the measures from our survey are based on Likert scales, we use non-parametric rank-based statistical tests.</p>
<p>In all variables, we observe an overall modest agreement when measured with Cronbach’s <inline-formula id="pone.0347757.e010"><alternatives><graphic id="pone.0347757.e010g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0347757.e010" xlink:type="simple"/><mml:math display="inline" id="equation7"><mml:mrow><mml:mi>α</mml:mi></mml:mrow></mml:math></alternatives></inline-formula>, with values of at most <inline-formula id="pone.0347757.e011"><alternatives><graphic id="pone.0347757.e011g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0347757.e011" xlink:type="simple"/><mml:math display="inline" id="equation8"><mml:mrow><mml:mi>α</mml:mi><mml:mo>=</mml:mo><mml:mn>0.55</mml:mn></mml:mrow></mml:math></alternatives></inline-formula>. We analyse the data to understand which combinations of different judgments we observed and find that regarding authenticity (across all variants) there are relatively few polar differences, i.e., one participant rating an item as authentic and the other as not authentic. For relevance, coherence and content the differences are in how positive a judgment is, with small differences of a single point (e.g., ‘neutral’ instead of ‘agreement’), again showing that while the absolute ratings have some variance, the tendency regarding the judgment is typically the same for both participants. Overall, the tendency towards positive or negative judgments about a variable is fairly consistent, especially given our large sample size which is a representative cross-section of the British society.</p>
<p>We note that our method neither allows us to observe participants directly nor measures underlying aspects regarding their capabilities (e.g., their reading comprehension). In principle, if participants with different capabilities are not evenly assigned to tracks, this could lead to unobserved confounding effects. However, given our large sample size, a strong influence of such confounding effects is unlikely.</p>
</sec>
<sec id="sec018">
<title>Linguistic style is different</title>
<p>To provide a perspective different from the human evaluation, we augment our results with a comparison of the linguistic properties of the original and the generated responses (see <xref ref-type="fig" rid="pone.0347757.g005">Fig 5</xref>). This experiment yields a number of interesting insights: First, the complexity of the sentences in terms of the number of conjuncts, clausal modifiers, clausal complements, clausal subjects and parataxes as well as the use of modals such as ‘should’ and ‘must’ are not significantly different between the actual responses and the impersonated ones. Secondly, the actual responses contain more discourse markers (e.g., ‘because’, ‘therefore’) than the impersonated responses, even though with a small effect size (<italic>r</italic> = 0.24). The reason for the statistically significant difference is that there is a long tail of actual responses that contain many discourse markers, even though the peak of the distributions is the same for actual and impersonated responses.</p>
<p>Epistemic markers like ‘I think’ are used substantially more often in original responses – they are only rarely found in the impersonated statements, leading to a large effect size (<italic>r</italic> = 0.87). The rarity in generated responses is not surprising, giving that these markers indicate that the speaker has a stance on some issue, which we expect to be strongly the case in a broadcast political debate and much less so in the case of an LLM.</p>
<p>Furthermore, the impersonated responses contain more nominalisations (<italic>r</italic> = −0.88) and have a higher lexical diversity (<italic>r</italic> = −0.92), both with large effect sizes. The overlap between the words from the question and the response is higher for impersonated than actual responses, with a large effect size (<italic>r</italic> = −0.98). In fact, the distribution shows that it is not uncommon for all words from the question to appear in the impersonated responses, while this is only rarely true for the actual responses.</p>
</sec>
<sec id="sec019">
<title>Public opinion</title>
<p>Public opinion on the use of AI technology for public debates is collected in two rounds in the exit poll: First, the participants respond without prior knowledge of the data source they just rated (actual versus generated). In the second step, the source of the data is revealed and the participants are asked the same questions on the use of AI technology again.</p>
<p>The results of our exit poll prior to revealing the use of AI (see <xref ref-type="fig" rid="pone.0347757.g006">Fig 6</xref>) paint a clear picture: The participants mostly state that they are familiar with AI. Interestingly, while they mostly believe that AI cannot provide valuable contributions to public debates, they simultaneously claim that they support the use of AI if it is made explicit and if it is known how the system was developed. When it comes to the matter of regulating AI, the participants’ opinions are rather mixed, with roughly equal-sized groups favoring regulation, opposing regulation, and being undecided. After revealing the use of AI, over 90% of the participants do not change their opinion. For those who do change their opinion, we see a clear trend: The participants realize they are less familiar with AI than they thought, but also have a more favorable opinion on the use of AI in debates, while at the same time seeing a bigger need for regulation.</p>
<fig id="pone.0347757.g005" position="float"><object-id pub-id-type="doi">10.1371/journal.pone.0347757.g005</object-id><label>Fig 5</label><caption><title>Linguistic surface of actual debate responses versus impersonated debate responses.</title><p>Violins show a kernel density estimation of the probability distribution, the miniature box-plots depict the median, upper and lower quartiles, and the whiskers the largest/smallest value observed within 1.5 times the interquartile range of the upper/lower quartile. The statistical markers reported are the p-value of two-sided Wilcoxon signed rank tests, the effect size <italic>r</italic>, the sample sizes <italic>n</italic>, mean values <italic>M</italic> and standard deviations <italic>SD</italic>.</p></caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0347757.g005" xlink:type="simple"/></fig>
<fig id="pone.0347757.g006" position="float"><object-id pub-id-type="doi">10.1371/journal.pone.0347757.g006</object-id><label>Fig 6</label><caption><title>Results of the exit poll on the opinion of the participants.</title><p>The stacked bar chart reports the percentages of the ratings that we observed. The statistical markers reported are the sample sizes <italic>n</italic>, mean values <italic>M</italic> and standard deviations <italic>SD</italic>. The bar chart depicts the counts for each topic that was addressed in the free-text answers.</p></caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0347757.g006" xlink:type="simple"/></fig>
<p>The optional free-text answers (<italic>n</italic> = 248) further corroborate these results. Many participants explicitly note that they did not change their results (<italic>n</italic> = 107). However, the other free-text answers indicate that the changes in opinions are caused by the confrontation with the capabilities of AI through the survey. Participants often mention that the impersonated responses are better than the human responses (<italic>n</italic> = 66) or that the quality of the impersonated responses is higher (<italic>n</italic> = 32). A few participants noted that the high coherence in the impersonated responses made them sceptical, leading them to believe that AI was possibly used in the survey (<italic>n</italic> = 7) and that this advantage over the humans can be explained by the live debate setting where the actual panellists do not have time to carefully prepare their responses (<italic>n</italic> = 5). One participant even notes that this advantage of AI means that AI could be used to train humans for debates. Some participants note that they are not able to distinguish between AI and humans at all (<italic>n</italic> = 26). There are also a few comments noting negative aspects regarding the quality of the generated responses (<italic>n</italic> = 4) or that AI was worse than the humans (<italic>n</italic> = 1), but these are outliers.</p>
<p>Another aspect that is stressed in the comments is the requirement to regulate the use of AI (<italic>n</italic> = 62), especially with respect to transparency. Particularly, many participants (<italic>n</italic> = 39) express concerns regarding the potential for deceptive use of AI in debates and the associated risks, some even note feelings of fear, shock, and worry (<italic>n</italic> = 17). However, some participants express positive emotions like surprise and amazement given the strong capabilities of AI (<italic>n</italic> = 16). When it comes to the use of AI technology in debates, some participants argue that AI’s convincing performance indicates potential for its use in debates (<italic>n</italic> = 36), while others question the general concept of AI debaters (<italic>n</italic> = 23). For instance, participants are uncertain about AI’s ability to represent party opinions and are concerned that involvement of AI will undermine the value of debates as forums for meaningful discussion between people.</p>
</sec>
</sec>
<sec id="sec020">
<title>Discussion and conclusion</title>
<p>Our results demonstrate that <bold>modern AI based on LLMs is able to provide high-quality impersonated debate content that is perceived as authentic when attributed to actual people</bold>. We also find indications that people in general rate the impersonated content to be slightly more authentic than the actual human debate responses. In addition, the impersonated responses are judged as more coherent and relevant than actual responses. While the lower coherence can be attributed to the panellists being under scrutiny in a nationally broadcast political debate programme, the higher relevance of the LLM-generated responses indicates that the LLM tends to stay more on-topic than human speakers. Our analysis of responses judged as different in content supports this, as we find that the panel members do not address the question more often than the LLM. While we rule out length and grammatical errors as possible sources for differences in authenticity, we cannot rule out that there are unobserved factors introduced by our experiment design. Interestingly, the authenticity is not negatively affected by the notable differences in the linguistic surface of the responses. GPT clearly has its own unique style defined by a diverse vocabulary and an avoidance of epistemic markers, which is, however, not noticed by our study participants. There does not seem to be a problem with an uncanny valley [<xref ref-type="bibr" rid="pone.0347757.ref031">31</xref>], which would make the participants feel uncomfortable with the impersonated responses.</p>
<p>Even though most of <bold>our participants stated that they are familiar with AI, they did not expect AI to have generated these responses and therefore underestimated the capabilities of modern generative AI</bold>. Being confronted with the AI’s ability to generate convincing debate contributions elicited different reactions from the participants including evidence-driven discussions of the merits of AI, negative emotional responses driven by concerns over the potential for misuse, and positive emotional responses to technological progress demonstrated by the AI’s capabilities. Overall, the participants’ encounter with AI via our survey seemed to increase their appreciation for generative technology in some ways, while also highlighting the need for regulation of such tools when used in a debate context.</p>
<p>When questioned about the merits of AI, the participants expressed a strong belief that AI can be a valuable tool, but they have heterogeneous views on the need for regulation and restrictions on use. However, on the matters of transparency, the public perspective is clear: <bold>Over 85% of participants think that AI use has to be made explicit and that information on how the AI was developed needs to be shared</bold>.</p>
<p>In general, the risks that are implied by our findings are severe. While it has been previously established that LLMs are capable of generating persuasive misinformation [<xref ref-type="bibr" rid="pone.0347757.ref032">32</xref>] and that the automated and human detection of such misinformation is unreliable [<xref ref-type="bibr" rid="pone.0347757.ref033">33</xref>], the results of our study add another layer of concern: We demonstrate that LLMs can generate authentic information by impersonating specific people, meaning that LLM-powered misinformation campaigns can go beyond targeting general topics and societal group, but can be made to target individual people by imitating their contributions to public discourse. Furthermore, the potential for AI to generate responses that do not only successfully imitate a politician, but also push a specific political agenda, warrants a larger-scale further exploration and an assessment of the associated risks. We note that our setting did not study this targeted misinformation, i.e., we did not prescribe the position the LLM should express. Future work needs to study if LLMs are still perceived as authentic when used in such a targeted manner. Since the dissemination of excerpts from political statements via social networks is a common form of political communication [<xref ref-type="bibr" rid="pone.0347757.ref034">34</xref>], it is easy to spread such generated statements at scale. Therefore, content moderation to identify and remove undisclosed AI-generated statements will be crucial [<xref ref-type="bibr" rid="pone.0347757.ref035">35</xref>]. Our own results suggest that a current model [<xref ref-type="bibr" rid="pone.0347757.ref036">36</xref>] can be used for such content moderation (accuracy of 89% on the task of classifying responses into impersonated or actual). However, more sophisticated approaches may be able to fool such detectors [<xref ref-type="bibr" rid="pone.0347757.ref037">37</xref>].</p>
<p>Overall, the implications of our results for the communication of political content are devastating: <bold>Threat actors can easily use LLMs to pollute public information spheres with fake, but authentic-sounding, political statements</bold>, for instance, to sow confusion about the original statement and to invent talking points. If this is further combined with deep fakes that are already known to be able to generate reliable authentic voices and videos of public people [<xref ref-type="bibr" rid="pone.0347757.ref038">38</xref>], the risk for society is enormous.</p>
</sec>
<sec id="sec021" sec-type="supplementary-material">
<title>Supporting information</title>
<supplementary-material id="pone.0347757.s001" mimetype="application/pdf" position="float" xlink:href="info:doi/10.1371/journal.pone.0347757.s001" xlink:type="simple">
<label>S1 Appendix</label>
<caption>
<title>Additional details for the survey.</title>
<p>(PDF)</p>
</caption>
</supplementary-material>
<supplementary-material id="pone.0347757.s002" mimetype="application/pdf" position="float" xlink:href="info:doi/10.1371/journal.pone.0347757.s002" xlink:type="simple">
<label>S2 Appendix</label>
<caption>
<title>Additional details for the results.</title>
<p>(PDF)</p>
</caption>
</supplementary-material>
<supplementary-material id="pone.0347757.s003" mimetype="application/pdf" position="float" xlink:href="info:doi/10.1371/journal.pone.0347757.s003" xlink:type="simple">
<label>S3 Appendix</label>
<caption>
<title>Additional analysis of alternative explanations for results.</title>
<p>(PDF)</p>
</caption>
</supplementary-material>
</sec>
</body>
<back>
<ref-list>
<title>References</title>
<ref id="pone.0347757.ref001"><label>1</label><mixed-citation publication-type="book" xlink:type="simple"><collab>OpenAI</collab>. <source>GPT-4 Technical Report</source>; <year>2023</year>. arXiv:2303.08774.</mixed-citation></ref>
<ref id="pone.0347757.ref002"><label>2</label><mixed-citation publication-type="book" xlink:type="simple"><name name-style="western"><surname>Anthropic</surname> <given-names>AI</given-names></name>. <source>The Claude 3 Model Family: Opus, Sonnet, Haiku</source>. Technical Report; <year>2024</year>.</mixed-citation></ref>
<ref id="pone.0347757.ref003"><label>3</label><mixed-citation publication-type="other" xlink:type="simple">Team G, Anil R, Borgeaud S, Wu Y, Alayrac JB, Yu J, <etal>et al</etal>. <article-title>Gemini: a family of highly capable multimodal models.</article-title> arXiv:231211805 [Preprint]. <year>2023</year>.</mixed-citation></ref>
<ref id="pone.0347757.ref004"><label>4</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Ziegler</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Kalliamvakou</surname> <given-names>E</given-names></name>, <name name-style="western"><surname>Li</surname> <given-names>XA</given-names></name>, <name name-style="western"><surname>Rice</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Rifkin</surname> <given-names>D</given-names></name>, <name name-style="western"><surname>Simister</surname> <given-names>S</given-names></name>, <etal>et al</etal>. <article-title>Measuring GitHub Copilot’s impact on productivity</article-title>. <source>Commun ACM</source>. <year>2024</year>;<volume>67</volume>(<issue>3</issue>):<fpage>54</fpage>–<lpage>63</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1145/3633453" xlink:type="simple">10.1145/3633453</ext-link></comment></mixed-citation></ref>
<ref id="pone.0347757.ref005"><label>5</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Herbold</surname> <given-names>S</given-names></name>, <name name-style="western"><surname>Hautli-Janisz</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Heuer</surname> <given-names>U</given-names></name>, <name name-style="western"><surname>Kikteva</surname> <given-names>Z</given-names></name>, <name name-style="western"><surname>Trautsch</surname> <given-names>A</given-names></name>. <article-title>A large-scale comparison of human-written versus ChatGPT-generated essays</article-title>. <source>Sci Rep</source>. <year>2023</year>;<volume>13</volume>(<issue>1</issue>):<fpage>18617</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1038/s41598-023-45644-9" xlink:type="simple">10.1038/s41598-023-45644-9</ext-link></comment> <object-id pub-id-type="pmid">37903836</object-id></mixed-citation></ref>
<ref id="pone.0347757.ref006"><label>6</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Katz</surname> <given-names>DM</given-names></name>, <name name-style="western"><surname>Bommarito</surname> <given-names>MJ</given-names></name>, <name name-style="western"><surname>Gao</surname> <given-names>S</given-names></name>, <name name-style="western"><surname>Arredondo</surname> <given-names>P</given-names></name>. <article-title>GPT-4 passes the bar exam</article-title>. <source>Philos Trans A Math Phys Eng Sci</source>. <year>2024</year>;<volume>382</volume>(<issue>2270</issue>):<fpage>20230254</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1098/rsta.2023.0254" xlink:type="simple">10.1098/rsta.2023.0254</ext-link></comment> <object-id pub-id-type="pmid">38403056</object-id></mixed-citation></ref>
<ref id="pone.0347757.ref007"><label>7</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Shanahan</surname> <given-names>M</given-names></name>, <name name-style="western"><surname>McDonell</surname> <given-names>K</given-names></name>, <name name-style="western"><surname>Reynolds</surname> <given-names>L</given-names></name>. <article-title>Role play with large language models</article-title>. <source>Nature</source>. <year>2023</year>;<volume>623</volume>(<issue>7987</issue>):<fpage>493</fpage>–<lpage>8</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1038/s41586-023-06647-8" xlink:type="simple">10.1038/s41586-023-06647-8</ext-link></comment> <object-id pub-id-type="pmid">37938776</object-id></mixed-citation></ref>
<ref id="pone.0347757.ref008"><label>8</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Bhandarkar</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Wilson</surname> <given-names>R</given-names></name>, <name name-style="western"><surname>Swarup</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Woodard</surname> <given-names>D.</given-names></name> <article-title>Emulating author style: a feasibility study of prompt-enabled text stylization with off-the-shelf LLMs.</article-title> In: <name name-style="western"><surname>Deshpande</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Hwang</surname> <given-names>E</given-names></name>, <name name-style="western"><surname>Murahari</surname> <given-names>V</given-names></name>, <name name-style="western"><surname>Park</surname> <given-names>JS</given-names></name>, <name name-style="western"><surname>Yang</surname> <given-names>D</given-names></name>, <name name-style="western"><surname>Sabharwal</surname> <given-names>A</given-names></name>, <etal>et al</etal>., editors. <conf-name>Proceedings of the 1st Workshop on Personalization of Generative AI Systems (PERSONALIZE 2024)</conf-name>. <publisher-loc>St. Julians, Malta</publisher-loc>: <publisher-name>Association for Computational Linguistics</publisher-name>; <year>2024</year>. p. <fpage>76</fpage>–<lpage>82</lpage>. Available from: <ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/2024.personalize-1.6" xlink:type="simple">https://aclanthology.org/2024.personalize-1.6</ext-link></mixed-citation></ref>
<ref id="pone.0347757.ref009"><label>9</label><mixed-citation publication-type="other" xlink:type="simple">Simmons G. <article-title>Moral mimicry: large language models produce moral rationalizations tailored to political identity.</article-title> arXiv:220912106 [Preprint]. <year>2022</year>.</mixed-citation></ref>
<ref id="pone.0347757.ref010"><label>10</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Hackenburg</surname> <given-names>K</given-names></name>, <name name-style="western"><surname>Ibrahim</surname> <given-names>L</given-names></name>, <name name-style="western"><surname>Tappin</surname> <given-names>BM</given-names></name>, <name name-style="western"><surname>Tsakiris</surname> <given-names>M</given-names></name>. <article-title>Comparing the persuasiveness of role-playing large language models and human experts on polarized US political issues</article-title>. <source>OSF Preprints</source>. <year>2023</year>;<volume>10</volume>.</mixed-citation></ref>
<ref id="pone.0347757.ref011"><label>11</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Salewski</surname> <given-names>L</given-names></name>, <name name-style="western"><surname>Alaniz</surname> <given-names>S</given-names></name>, <name name-style="western"><surname>Rio-Torto</surname> <given-names>I</given-names></name>, <name name-style="western"><surname>Schulz</surname> <given-names>E</given-names></name>, <name name-style="western"><surname>Akata</surname> <given-names>Z</given-names></name>. <article-title>In-context impersonation reveals Large Language Models’ strengths and biases</article-title>. <source>Adv Neural Inf Process Syst</source>. <year>2024</year>;<volume>36</volume>.</mixed-citation></ref>
<ref id="pone.0347757.ref012"><label>12</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Bai</surname> <given-names>H</given-names></name>, <name name-style="western"><surname>Voelkel</surname> <given-names>J</given-names></name>, <name name-style="western"><surname>Eichstaedt</surname> <given-names>J</given-names></name>, <name name-style="western"><surname>Willer</surname> <given-names>R</given-names></name>. <article-title>Artificial intelligence can persuade humans on political issues.</article-title> Preprint. <year>2023</year>. Available from: <ext-link ext-link-type="uri" xlink:href="http://dx.doi.org/10.21203/rs.3.rs-3238396/v1" xlink:type="simple">http://dx.doi.org/10.21203/rs.3.rs-3238396/v1</ext-link></mixed-citation></ref>
<ref id="pone.0347757.ref013"><label>13</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Santurkar</surname> <given-names>S</given-names></name>, <name name-style="western"><surname>Durmus</surname> <given-names>E</given-names></name>, <name name-style="western"><surname>Ladhak</surname> <given-names>F</given-names></name>, <name name-style="western"><surname>Lee</surname> <given-names>C</given-names></name>, <name name-style="western"><surname>Liang</surname> <given-names>P</given-names></name>, <name name-style="western"><surname>Hashimoto</surname> <given-names>T</given-names></name>. Whose opinions do language models reflect? <year>2023</year>. Available from: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2303.17548" xlink:type="simple">https://arxiv.org/abs/2303.17548</ext-link>. arXiv:2303.17548.</mixed-citation></ref>
<ref id="pone.0347757.ref014"><label>14</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Matz</surname> <given-names>SC</given-names></name>, <name name-style="western"><surname>Teeny</surname> <given-names>JD</given-names></name>, <name name-style="western"><surname>Vaid</surname> <given-names>SS</given-names></name>, <name name-style="western"><surname>Peters</surname> <given-names>H</given-names></name>, <name name-style="western"><surname>Harari</surname> <given-names>GM</given-names></name>, <name name-style="western"><surname>Cerf</surname> <given-names>M</given-names></name>. <article-title>The potential of generative AI for personalized persuasion at scale</article-title>. <source>Sci Rep</source>. <year>2024</year>;<volume>14</volume>(<issue>1</issue>):<fpage>4692</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1038/s41598-024-53755-0" xlink:type="simple">10.1038/s41598-024-53755-0</ext-link></comment> <object-id pub-id-type="pmid">38409168</object-id></mixed-citation></ref>
<ref id="pone.0347757.ref015"><label>15</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Simchon</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Edwards</surname> <given-names>M</given-names></name>, <name name-style="western"><surname>Lewandowsky</surname> <given-names>S</given-names></name>. <article-title>The persuasive effects of political microtargeting in the age of generative artificial intelligence</article-title>. <source>PNAS Nexus</source>. <year>2024</year>;<volume>3</volume>(<issue>2</issue>):pgae035. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/pnasnexus/pgae035" xlink:type="simple">10.1093/pnasnexus/pgae035</ext-link></comment> <object-id pub-id-type="pmid">38328785</object-id></mixed-citation></ref>
<ref id="pone.0347757.ref016"><label>16</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Hautli-Janisz</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Kikteva</surname> <given-names>Z</given-names></name>, <name name-style="western"><surname>Siskou</surname> <given-names>W</given-names></name>, <name name-style="western"><surname>Gorska</surname> <given-names>K</given-names></name>, <name name-style="western"><surname>Becker</surname> <given-names>R</given-names></name>, <name name-style="western"><surname>Reed</surname> <given-names>C.</given-names></name> <article-title>QT30: a corpus of argument and conflict in broadcast debate.</article-title> In: <name name-style="western"><surname>Calzolari</surname> <given-names>N</given-names></name>, <name name-style="western"><surname>Béchet</surname> <given-names>F</given-names></name>, <name name-style="western"><surname>Blache</surname> <given-names>P</given-names></name>, <name name-style="western"><surname>Choukri</surname> <given-names>K</given-names></name>, <name name-style="western"><surname>Cieri</surname> <given-names>C</given-names></name>, <name name-style="western"><surname>Declerck</surname> <given-names>T</given-names></name>, <etal>et al</etal>., editors. <conf-name>Proceedings of the Thirteenth Language Resources and Evaluation Conference</conf-name>. <publisher-loc>Marseille, France</publisher-loc>: <publisher-name>European Language Resources Association</publisher-name>; <year>2022</year>. p. <fpage>3291</fpage>–<lpage>300</lpage>. Available from: <ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/2022.lrec-1.352" xlink:type="simple">https://aclanthology.org/2022.lrec-1.352</ext-link></mixed-citation></ref>
<ref id="pone.0347757.ref017"><label>17</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Zellers</surname> <given-names>R</given-names></name>, <name name-style="western"><surname>Holtzman</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Bisk</surname> <given-names>Y</given-names></name>, <name name-style="western"><surname>Farhadi</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Choi</surname> <given-names>Y.</given-names></name> <article-title>HellaSwag: can a machine really finish your sentence?</article-title> In: <name name-style="western"><surname>Korhonen</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Traum</surname> <given-names>D</given-names></name>, <name name-style="western"><surname>Màrquez</surname> <given-names>L</given-names></name>, editors. <conf-name>Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics</conf-name>. <publisher-loc>Florence, Italy</publisher-loc>: <publisher-name>Association for Computational Linguistics</publisher-name>; <year>2019</year>. p. <fpage>4791</fpage>–<lpage>800</lpage>. Available from: <ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/P19-1472" xlink:type="simple">https://aclanthology.org/P19-1472</ext-link> <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.18653/v1/P19-1472" xlink:type="simple">https://doi.org/10.18653/v1/P19-1472</ext-link></mixed-citation></ref>
<ref id="pone.0347757.ref018"><label>18</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Weiss</surname> <given-names>Z</given-names></name>, <name name-style="western"><surname>Riemenschneider</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Schröter</surname> <given-names>P</given-names></name>, <name name-style="western"><surname>Meurers</surname> <given-names>D</given-names></name>. <article-title>Computationally modeling the impact of task-appropriate language complexity and accuracy on human grading of German essays</article-title>. <conf-name>Proceedings of the Fourteenth Workshop on Innovative Use of NLP for Building Educational Applications</conf-name>. <publisher-loc>Florence, Italy</publisher-loc>: <publisher-name>Association for Computational Linguistics</publisher-name>; <year>2019</year>. p. <fpage>30</fpage>–<lpage>45</lpage>. Available from: <ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/W19-4404" xlink:type="simple">https://aclanthology.org/W19-4404</ext-link> <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.18653/v1/W19-4404" xlink:type="simple">https://doi.org/10.18653/v1/W19-4404</ext-link></mixed-citation></ref>
<ref id="pone.0347757.ref019"><label>19</label><mixed-citation publication-type="other" xlink:type="simple">Siskou W, Friedrich L, Eckhard S, Espinoza I, Hautli-Janisz A. <article-title>Measuring plain language in public service encounters</article-title>. <conf-name>Proceedings of the 2nd Workshop on Computational Linguistics for Political Text Analysis (CPSS-2022)</conf-name>. <conf-loc>Potsdam, Germany</conf-loc>; <year>2022</year>.</mixed-citation></ref>
<ref id="pone.0347757.ref020"><label>20</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Lenk</surname> <given-names>U</given-names></name>. <article-title>Discourse markers and global coherence in conversation</article-title>. <source>J Pragmat</source>. <year>1998</year>;<volume>30</volume>(<issue>2</issue>):<fpage>245</fpage>–<lpage>57</lpage>. Available from: <ext-link ext-link-type="uri" xlink:href="https://www.sciencedirect.com/science/article/pii/S0378216698000277" xlink:type="simple">https://www.sciencedirect.com/science/article/pii/S0378216698000277</ext-link> <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1016/S0378-2166(98)00027-7" xlink:type="simple">https://doi.org/10.1016/S0378-2166(98)00027-7</ext-link></mixed-citation></ref>
<ref id="pone.0347757.ref021"><label>21</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>McCarthy</surname> <given-names>PM</given-names></name>, <name name-style="western"><surname>Jarvis</surname> <given-names>S</given-names></name>. <article-title>MTLD, vocd-D, and HD-D: a validation study of sophisticated approaches to lexical diversity assessment</article-title>. <source>Behav Res Methods</source>. <year>2010</year>;<volume>42</volume>(<issue>2</issue>):<fpage>381</fpage>–<lpage>92</lpage>.</mixed-citation></ref>
<ref id="pone.0347757.ref022"><label>22</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Cronbach</surname> <given-names>LJ</given-names></name>. <article-title>Coefficient alpha and the internal structure of tests</article-title>. <source>Psychometrika</source>. <year>1951</year>;<volume>16</volume>(<issue>3</issue>):<fpage>297</fpage>–<lpage>334</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1007/bf02310555" xlink:type="simple">10.1007/bf02310555</ext-link></comment></mixed-citation></ref>
<ref id="pone.0347757.ref023"><label>23</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Wilcoxon</surname> <given-names>F</given-names></name>. <article-title>Individual comparisons by ranking methods</article-title>. <source>Biom Bull</source>. <year>1945</year>;<volume>1</volume>(<issue>6</issue>):<fpage>80</fpage>–<lpage>3</lpage>.</mixed-citation></ref>
<ref id="pone.0347757.ref024"><label>24</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Friedman</surname> <given-names>M</given-names></name>. <article-title>A Correction: The use of ranks to avoid the assumption of normality implicit in the analysis of variance</article-title>. <source>J Am Stat Assoc</source>. <year>1939</year>;<volume>34</volume>(<issue>205</issue>):<fpage>109</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.2307/2279169" xlink:type="simple">10.2307/2279169</ext-link></comment></mixed-citation></ref>
<ref id="pone.0347757.ref025"><label>25</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Kruskal</surname> <given-names>WH</given-names></name>, <name name-style="western"><surname>Wallis</surname> <given-names>WA</given-names></name>. <article-title>Errata: use of ranks in one-criterion variance analysis</article-title>. <source>J Am Stat Assoc</source>. <year>1953</year>;<volume>48</volume>(<issue>264</issue>):<fpage>907</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.2307/2281082" xlink:type="simple">10.2307/2281082</ext-link></comment></mixed-citation></ref>
<ref id="pone.0347757.ref026"><label>26</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Mann</surname> <given-names>HB</given-names></name>, <name name-style="western"><surname>Whitney</surname> <given-names>DR</given-names></name>. <article-title>On a test of whether one of two random variables is stochastically larger than the other</article-title>. <source>Ann Math Statist</source>. <year>1947</year>;<volume>18</volume>(<issue>1</issue>):<fpage>50</fpage>–<lpage>60</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1214/aoms/1177730491" xlink:type="simple">10.1214/aoms/1177730491</ext-link></comment></mixed-citation></ref>
<ref id="pone.0347757.ref027"><label>27</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Bonferroni</surname> <given-names>C</given-names></name>. <article-title>Teoria statistica delle classi e calcolo delle probabilita</article-title>. <source>Pubbl R Ist Sup Sci Econ Com Firenze</source>. <year>1936</year>;<volume>8</volume>:<fpage>3</fpage>–<lpage>62</lpage>.</mixed-citation></ref>
<ref id="pone.0347757.ref028"><label>28</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Cureton</surname> <given-names>EE</given-names></name>. <article-title>Rank-biserial correlation</article-title>. <source>Psychometrika</source>. <year>1956</year>;<volume>21</volume>(<issue>3</issue>):<fpage>287</fpage>–<lpage>90</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1007/BF02289138" xlink:type="simple">10.1007/BF02289138</ext-link></comment></mixed-citation></ref>
<ref id="pone.0347757.ref029"><label>29</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Thomas</surname> <given-names>DR</given-names></name>. <article-title>A General inductive approach for analyzing qualitative evaluation data</article-title>. <source>Am J Eval</source>. <year>2006</year>;<volume>27</volume>(<issue>2</issue>):<fpage>237</fpage>–<lpage>46</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1177/1098214005283748" xlink:type="simple">10.1177/1098214005283748</ext-link></comment></mixed-citation></ref>
<ref id="pone.0347757.ref030"><label>30</label><mixed-citation publication-type="book" xlink:type="simple"><name name-style="western"><surname>Corbin</surname> <given-names>J</given-names></name>, <name name-style="western"><surname>Strauss</surname> <given-names>A</given-names></name>. <source>Basics of qualitative research: techniques and procedures for developing grounded theory</source>. <publisher-name>Sage publications</publisher-name>; <year>2014</year>.</mixed-citation></ref>
<ref id="pone.0347757.ref031"><label>31</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Mori</surname> <given-names>M</given-names></name>. <article-title>Bukimi no tani [The uncanny valley]</article-title>. <source>Energy</source>. <year>1970</year>;<volume>7</volume>:<fpage>33</fpage>.</mixed-citation></ref>
<ref id="pone.0347757.ref032"><label>32</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Zhou</surname> <given-names>J</given-names></name>, <name name-style="western"><surname>Zhang</surname> <given-names>Y</given-names></name>, <name name-style="western"><surname>Luo</surname> <given-names>Q</given-names></name>, <name name-style="western"><surname>Parker</surname> <given-names>AG</given-names></name>, <name name-style="western"><surname>De Choudhury</surname> <given-names>M</given-names></name>. <article-title>Synthetic lies: understanding AI-generated misinformation and evaluating algorithmic and human solutions</article-title>. <conf-name>Proceedings of the 2023 CHI Conference on Human Factors in Computing Systems. CHI ’23</conf-name>. <publisher-loc>New York, NY, USA</publisher-loc>: <publisher-name>Association for Computing Machinery</publisher-name>; <year>2023</year>. p. <fpage>1</fpage>–<lpage>20</lpage>. Available from: <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1145/3544548.3581318" xlink:type="simple">10.1145/3544548.3581318</ext-link></comment></mixed-citation></ref>
<ref id="pone.0347757.ref033"><label>33</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Jiang</surname> <given-names>B</given-names></name>, <name name-style="western"><surname>Tan</surname> <given-names>Z</given-names></name>, <name name-style="western"><surname>Nirmal</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Liu</surname> <given-names>H</given-names></name>. <article-title>Disinformation detection: an evolving challenge in the age of LLMs</article-title>. <conf-name>Proceedings of the 2024 SIAM International Conference on Data Mining (SDM)</conf-name>. <publisher-name>SIAM Publications Library</publisher-name>; <year>2024</year>. p. <fpage>427</fpage>–<lpage>35</lpage>. Available from: <ext-link ext-link-type="uri" xlink:href="https://epubs.siam.org/doi/abs/10.1137/1.9781611978032.50" xlink:type="simple">https://epubs.siam.org/doi/abs/10.1137/1.9781611978032.50</ext-link>. arXiv: <ext-link ext-link-type="uri" xlink:href="https://epubs.siam.org/doi/pdf/10.1137/1.9781611978032.50" xlink:type="simple">https://epubs.siam.org/doi/pdf/10.1137/1.9781611978032.50</ext-link></mixed-citation></ref>
<ref id="pone.0347757.ref034"><label>34</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Osatuyi</surname> <given-names>B</given-names></name>. <article-title>Information sharing on social media sites</article-title>. <source>Comput Hum Behav</source>. <year>2013</year>;<volume>29</volume>(<issue>6</issue>):<fpage>2622</fpage>–<lpage>31</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1016/j.chb.2013.07.001" xlink:type="simple">10.1016/j.chb.2013.07.001</ext-link></comment></mixed-citation></ref>
<ref id="pone.0347757.ref035"><label>35</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Clark</surname> <given-names>CJ</given-names></name>, <name name-style="western"><surname>Jussim</surname> <given-names>L</given-names></name>, <name name-style="western"><surname>Frey</surname> <given-names>K</given-names></name>, <name name-style="western"><surname>Stevens</surname> <given-names>ST</given-names></name>, <name name-style="western"><surname>Al-Gharbi</surname> <given-names>M</given-names></name>, <name name-style="western"><surname>Aquino</surname> <given-names>K</given-names></name>, <etal>et al</etal>. <article-title>Prosocial motives underlie scientific censorship by scientists: a perspective and research agenda</article-title>. <source>Proc Natl Acad Sci U S A</source>. <year>2023</year>;<volume>120</volume>(<issue>48</issue>):e2301642120. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1073/pnas.2301642120" xlink:type="simple">10.1073/pnas.2301642120</ext-link></comment> <object-id pub-id-type="pmid">37983511</object-id></mixed-citation></ref>
<ref id="pone.0347757.ref036"><label>36</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Guo</surname> <given-names>B</given-names></name>, <name name-style="western"><surname>Zhang</surname> <given-names>X</given-names></name>, <name name-style="western"><surname>Wang</surname> <given-names>Z</given-names></name>, <name name-style="western"><surname>Jiang</surname> <given-names>M</given-names></name>, <name name-style="western"><surname>Nie</surname> <given-names>J</given-names></name>, <name name-style="western"><surname>Ding</surname> <given-names>Y</given-names></name>, <etal>et al</etal>. How close is ChatGPT to human experts? Comparison corpus, evaluation, and detection; <year>2023</year>. Available from: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2301.07597" xlink:type="simple">https://arxiv.org/abs/2301.07597</ext-link>. arXiv:2301.07597.</mixed-citation></ref>
<ref id="pone.0347757.ref037"><label>37</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Sun</surname> <given-names>Y</given-names></name>, <name name-style="western"><surname>He</surname> <given-names>J</given-names></name>, <name name-style="western"><surname>Cui</surname> <given-names>L</given-names></name>, <name name-style="western"><surname>Lei</surname> <given-names>S</given-names></name>, <name name-style="western"><surname>Lu</surname> <given-names>CT</given-names></name>. Exploring the deceptive power of LLM-generated fake news: a study of real-world detection challenges; <year>2024</year>. Available from: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2403.18249" xlink:type="simple">https://arxiv.org/abs/2403.18249.</ext-link> arXiv:2403.18249.</mixed-citation></ref>
<ref id="pone.0347757.ref038"><label>38</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Farid</surname> <given-names>H</given-names></name>. <article-title>Creating, using, misusing, and detecting deep fakes</article-title>. <source>J Online Trust Saf</source>. <year>2022</year>;<volume>1</volume>(<issue>4</issue>).</mixed-citation></ref>
</ref-list>
</back>
<sub-article article-type="aggregated-review-documents" id="pone.0347757.r001" specific-use="decision-letter">
<front-stub>
<article-id pub-id-type="doi">10.1371/journal.pone.0347757.r001</article-id>
<title-group>
<article-title>Decision Letter 0</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name name-style="western"><surname>Carrasco-Farré</surname>
<given-names>Carlos</given-names>
</name>
<role>Academic Editor</role>
</contrib>
</contrib-group>
<permissions>
<copyright-year>2026</copyright-year>
<copyright-holder>Carlos Carrasco-FarréCarlos Carrasco-Farré</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">
<license-p>This is an open access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">Creative Commons Attribution License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited., which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited.</license-p></license>
</permissions>
<related-object document-id="10.1371/journal.pone.0347757" document-id-type="doi" document-type="article" id="rel-obj001" link-type="peer-reviewed-article"/>
<custom-meta-group>
<custom-meta>
<meta-name>Submission Version</meta-name>
<meta-value>0</meta-value>
</custom-meta>
</custom-meta-group>
</front-stub>
<body>
<p><named-content content-type="letter-date">19 Oct 2025</named-content></p>
<p>--&gt;PONE-D-25-15257--&gt;--&gt;LLM-impersonated debate contributions are more authentic, relevant and coherent than their original: A representative study using BBC1’s Question Time--&gt;--&gt;PLOS ONE</p>
<p>Dear Dr. Herbold,</p>
<p>Thank you for submitting your manuscript to <italic>PLOS ONE</italic>   . I am pleased to inform you that, following peer review, the reviewers and I found your study to be original, clearly written, and addressing a timely and important topic. Both reviewers recognized the relevance and ambition of your work and appreciated its methodological clarity and potential contribution.. I am pleased to inform you that, following peer review, the reviewers and I found your study to be original, clearly written, and addressing a timely and important topic. Both reviewers recognized the relevance and ambition of your work and appreciated its methodological clarity and potential contribution.</p>
<p>==============================</p>
<p>However, they also identified several issues that need to be addressed before the manuscript can be considered for publication. Based on their feedback, I am inviting you to <bold>submit a major revision</bold>   . Below, I summarize the main points raised by the reviewers and provide some guidance on how to proceed.. Below, I summarize the main points raised by the reviewers and provide some guidance on how to proceed.</p>
<p>&lt;h3 data-end="1014" data-start="953"&gt;<bold>1. Contextual and interpretive alignment (Reviewer 1)</bold>   &lt;/h3&gt;&lt;/h3&gt;</p>
<p>Reviewer 1 commends the design and analysis but raises concerns about the contextual asymmetry between the original human responses and the LLM-generated ones. Since the human content was transcribed from a live television programme (“Question Time”), it may include disfluencies, references, or transcription artifacts that affect perceptions differently than the directly generated LLM text. The reviewer suggests adding a post-hoc robustness check or qualitative analysis to demonstrate that these contextual or transcription-related differences do not drive the main effects.</p>
<p>Relatedly, Reviewer 1 encourages you to expand the discussion of content differences between human and impersonated responses. Specifically, the paper would benefit from showing <italic>how</italic>    these differ substantively; for example, by analyzing whether impersonations diverge in stance or argumentation, rather than merely in surface-level content. This would clarify whether the framing of “misrepresentation” is warranted.these differ substantively; for example, by analyzing whether impersonations diverge in stance or argumentation, rather than merely in surface-level content. This would clarify whether the framing of “misrepresentation” is warranted.</p>
<p>Finally, the reviewer notes that the reliability of human ratings is modest and recommends greater transparency about inter-rater variability, or, if feasible, additional averaging or collection of ratings to reduce subjective noise.</p>
<p>Minor notes include adding a brief explanation that <italic>Question Time</italic>    is a television programme, clarifying “random speaker” in the methods, and checking for potential answer length differences that could influence results.is a television programme, clarifying “random speaker” in the methods, and checking for potential answer length differences that could influence results.</p>
<p>&lt;h3 data-end="2561" data-start="2496"&gt;<bold>2. Conceptual and methodological refinements (Reviewer 2)</bold>   &lt;/h3&gt;&lt;/h3&gt;</p>
<p>Reviewer 2 finds the study compelling but identifies several areas where methodological clarification is needed:</p>
<p><list list-type="bullet"><list-item><p><bold>Construct clarity:</bold>    The meaning of The meaning of <italic>authenticity</italic>    should be defined more precisely, whether it captures authorship likelihood, realism, or plausibility. Reviewers recommend clarifying how this construct differs from related ones (e.g., coherence or content similarity) and ensuring that participants understood it consistently.should be defined more precisely, whether it captures authorship likelihood, realism, or plausibility. Reviewers recommend clarifying how this construct differs from related ones (e.g., coherence or content similarity) and ensuring that participants understood it consistently.</p>
</list-item>
<list-item>
<p><bold>Experimental design:</bold>    Provide more detail on the randomization and balancing of stimuli across conditions. If possible, include or reference any inter-rater checks used during the manual screening of generated responses.Provide more detail on the randomization and balancing of stimuli across conditions. If possible, include or reference any inter-rater checks used during the manual screening of generated responses.</p>
</list-item>
<list-item>
<p><bold>Measurement approach:</bold>    Consider complementing subjective Likert judgments with an objective or behavioral indicator (e.g., response times or accuracy), or at least acknowledge the limitations of relying solely on self-reported data.Consider complementing subjective Likert judgments with an objective or behavioral indicator (e.g., response times or accuracy), or at least acknowledge the limitations of relying solely on self-reported data.</p>
</list-item>
<list-item>
<p><bold>Statistical analysis:</bold>    The reviewer suggests using non-parametric or mixed-effects models better suited to ordinal and nested data, reporting medians and interquartile ranges, and revising effect size measures accordingly. You might also re-evaluate inter-rater reliability using intraclass correlation coefficients.The reviewer suggests using non-parametric or mixed-effects models better suited to ordinal and nested data, reporting medians and interquartile ranges, and revising effect size measures accordingly. You might also re-evaluate inter-rater reliability using intraclass correlation coefficients.</p>
</list-item>
<list-item>
<p><bold>Interpretation:</bold>    Finally, temper the strength of the claims about deception or indistinguishability between AI and human responses. Reviewer 2 encourages reframing your findings as perceptual rather than deceptive effects, potentially driven by linguistic fluency or stylistic uniformity.Finally, temper the strength of the claims about deception or indistinguishability between AI and human responses. Reviewer 2 encourages reframing your findings as perceptual rather than deceptive effects, potentially driven by linguistic fluency or stylistic uniformity.</p></list-item></list></p>
<p>Both reviewers found your study promising and well written. Their feedback is intended to help you enhance the rigor, clarity, and interpretive precision of your work. I encourage you to take this as a positive step toward publication; since the paper already has a strong foundation and, with the recommended revisions, could make a meaningful contribution to ongoing debates on authenticity, impersonation, and LLM-mediated communication.</p>
<p>==============================</p>
<p>Please submit your revised manuscript by Dec 03 2025 11:59PM. If you will need more time than this to complete your revisions, please reply to this message or contact the journal office at <email xlink:type="simple">plosone@plos.org</email>. When you're ready to submit your revision, log on to <ext-link ext-link-type="uri" xlink:href="https://www.editorialmanager.com/pone/" xlink:type="simple">https://www.editorialmanager.com/pone/</ext-link> and select the 'Submissions Needing Revision' folder to locate your manuscript file.. When you're ready to submit your revision, log on to <ext-link ext-link-type="uri" xlink:href="https://www.editorialmanager.com/pone/" xlink:type="simple">https://www.editorialmanager.com/pone/</ext-link> and select the 'Submissions Needing Revision' folder to locate your manuscript file.</p>
<p>Please include the following items when submitting your revised manuscript:</p>
<p>--&gt;</p>
<p><list list-type="bullet"><list-item><p>A rebuttal letter that responds to each point raised by the academic editor and reviewer(s). You should upload this letter as a separate file labeled 'Response to Reviewers'.</p>
</list-item>
<list-item>
<p>A marked-up copy of your manuscript that highlights changes made to the original version. You should upload this as a separate file labeled 'Revised Manuscript with Track Changes'.</p>
</list-item>
<list-item>
<p>An unmarked version of your revised paper without tracked changes. You should upload this as a separate file labeled 'Manuscript'.</p></list-item></list></p>
<p>If you would like to make changes to your financial disclosure, please include your updated statement in your cover letter. Guidelines for resubmitting your figure files are available below the reviewer comments at the end of this letter.</p>
<p>If applicable, we recommend that you deposit your laboratory protocols in protocols.io to enhance the reproducibility of your results. Protocols.io assigns your protocol its own identifier (DOI) so that it can be cited independently in the future. For instructions see: <ext-link ext-link-type="uri" xlink:href="https://journals.plos.org/plosone/s/submission-guidelines#loc-laboratory-protocols" xlink:type="simple">https://journals.plos.org/plosone/s/submission-guidelines#loc-laboratory-protocols</ext-link>. Additionally, PLOS ONE offers an option for publishing peer-reviewed Lab Protocol articles, which describe protocols hosted on protocols.io. Read more information on sharing protocols at . Additionally, PLOS ONE offers an option for publishing peer-reviewed Lab Protocol articles, which describe protocols hosted on protocols.io. Read more information on sharing protocols at <ext-link ext-link-type="uri" xlink:href="https://plos.org/protocols?utm_medium=editorial-email&amp;utm_source=authorletters&amp;utm_campaign=protocols" xlink:type="simple">https://plos.org/protocols?utm_medium=editorial-email&amp;utm_source=authorletters&amp;utm_campaign=protocols</ext-link>..</p>
<p>We look forward to receiving your revised manuscript.</p>
<p>Kind regards,</p>
<p>Carlos Carrasco-Farré</p>
<p>Academic Editor</p>
<p>PLOS ONE</p>
<p><bold>Journal Requirements:</bold></p>
<p>--&gt;1. When submitting your revision, we need you to address these additional requirements.--&gt;--&gt; --&gt;--&gt;Please ensure that your manuscript meets PLOS ONE's style requirements, including those for file naming. The PLOS ONE style templates can be found at --&gt;--&gt;<ext-link ext-link-type="uri" xlink:href="https://journals.plos.org/plosone/s/file?id=wjVg/PLOSOne_formatting_sample_main_body.pdf" xlink:type="simple">https://journals.plos.org/plosone/s/file?id=wjVg/PLOSOne_formatting_sample_main_body.pdf</ext-link> and --&gt;--&gt;<ext-link ext-link-type="uri" xlink:href="https://journals.plos.org/plosone/s/file?id=ba62/PLOSOne_formatting_sample_title_authors_affiliations.pdf" xlink:type="simple">https://journals.plos.org/plosone/s/file?id=ba62/PLOSOne_formatting_sample_title_authors_affiliations.pdf</ext-link>--&gt;--&gt; --&gt;--&gt;2. Please update your submission to use the PLOS LaTeX template. The template and more information on our requirements for LaTeX submissions can be found at <ext-link ext-link-type="uri" xlink:href="http://journals.plos.org/plosone/s/latex" xlink:type="simple">http://journals.plos.org/plosone/s/latex</ext-link>.--&gt;--&gt; --&gt;--&gt;3. We note that the grant information you provided in the ‘Funding Information’ and ‘Financial Disclosure’ sections do not match. --&gt;--&gt; --&gt;--&gt;When you resubmit, please ensure that you provide the correct grant numbers for the awards you received for your study in the ‘Funding Information’ section.--&gt;--&gt; --&gt;--&gt;4. Thank you for stating the following financial disclosure: --&gt;--&gt;A.H. work was partially funded by the VolkswagenStiftung under grant Az. 98544 ‘Deliberation Laboratory’--&gt;--&gt;URL: <ext-link ext-link-type="uri" xlink:href="https://www.volkswagenstiftung.de/" xlink:type="simple">https://www.volkswagenstiftung.de/</ext-link>--&gt;--&gt;The funders had no role in the study whatsoever.  --&gt;--&gt; --&gt;--&gt;Please state what role the funders took in the study.  If the funders had no role, please state: "The funders had no role in study design, data collection and analysis, decision to publish, or preparation of the manuscript." --&gt;--&gt;If this statement is not correct you must amend it as needed. --&gt;--&gt;Please include this amended Role of Funder statement in your cover letter; we will change the online submission form on your behalf.--&gt;--&gt; --&gt;--&gt;5. Thank you for stating the following in the Acknowledgments Section of your manuscript: --&gt;--&gt;The work reported on in this paper was partially funded by the VolkswagenStiftung under grant Az. 98544 ‘Deliberation Laboratory’--&gt;--&gt; --&gt;--&gt;We note that you have provided funding information that is not currently declared in your Funding Statement. However, funding information should not appear in the Acknowledgments section or other areas of your manuscript. We will only publish funding information present in the Funding Statement section of the online submission form. --&gt;--&gt;Please remove any funding-related text from the manuscript and let us know how you would like to update your Funding Statement. Currently, your Funding Statement reads as follows: --&gt;--&gt;A.H. work was partially funded by the VolkswagenStiftung under grant Az. 98544 ‘Deliberation Laboratory’--&gt;--&gt;URL: <ext-link ext-link-type="uri" xlink:href="https://www.volkswagenstiftung.de/" xlink:type="simple">https://www.volkswagenstiftung.de/</ext-link>--&gt;--&gt;The funders had no role in the study whatsoever. --&gt;--&gt; --&gt;--&gt;Please include your amended statements within your cover letter; we will change the online submission form on your behalf.--&gt;--&gt; --&gt;--&gt;6. Thank you for uploading your study's underlying data set. Unfortunately, the repository you have noted in your Data Availability statement does not qualify as an acceptable data repository according to PLOS's standards.--&gt;--&gt; --&gt;--&gt;At this time, please upload the minimal data set necessary to replicate your study's findings to a stable, public repository (such as figshare or Dryad) and provide us with the relevant URLs, DOIs, or accession numbers that may be used to access these data. For a list of recommended repositories and additional information on PLOS standards for data deposition, please see <ext-link ext-link-type="uri" xlink:href="https://journals.plos.org/plosone/s/recommended-repositories" xlink:type="simple">https://journals.plos.org/plosone/s/recommended-repositories</ext-link>.--&gt;--&gt; --&gt;--&gt;7. Please include your full ethics statement in the ‘Methods’ section of your manuscript file. In your statement, please include the full name of the IRB or ethics committee who approved or waived your study, as well as whether or not you obtained informed written or verbal consent. If consent was waived for your study, please include this information in your statement as well.--&gt;--&gt; --&gt;--&gt;8. If the reviewer comments include a recommendation to cite specific previously published works, please review and evaluate these publications to determine whether they are relevant and should be cited. There is no requirement to cite these works unless the editor has indicated otherwise.</p>
<p>[Note: HTML markup is below. Please do not edit.]</p>
<p>Reviewers' comments:</p>
<p>Reviewer's Responses to Questions--&gt;</p>
<p>--&gt;<bold>Comments to the Author</bold></p>
<p>1. Is the manuscript technically sound, and do the data support the conclusions?</p>
<p>The manuscript must describe a technically sound piece of scientific research with data that supports the conclusions. Experiments must have been conducted rigorously, with appropriate controls, replication, and sample sizes. The conclusions must be drawn appropriately based on the data presented.--&gt;</p>
<p>Reviewer #1: Yes</p>
<p>Reviewer #2: Yes</p>
<p>**********</p>
<p>--&gt;2. Has the statistical analysis been performed appropriately and rigorously?--&gt;</p>
<p>Reviewer #1: Yes</p>
<p>Reviewer #2: Yes</p>
<p>**********</p>
<p>--&gt;3. Have the authors made all data underlying the findings in their manuscript fully available?</p>
<p>The <ext-link ext-link-type="uri" xlink:href="http://www.plosone.org/static/policies.action#sharing" xlink:type="simple">PLOS Data policy</ext-link> requires authors to make all data underlying the findings described in their manuscript fully available without restriction, with rare exception (please refer to the Data Availability Statement in the manuscript PDF file). The data should be provided as part of the manuscript or its supporting information, or deposited to a public repository. For example, in addition to summary statistics, the data points behind means, medians and variance measures should be available. If there are restrictions on publicly sharing data—e.g. participant privacy or use of data from a third party—those must be specified.requires authors to make all data underlying the findings described in their manuscript fully available without restriction, with rare exception (please refer to the Data Availability Statement in the manuscript PDF file). The data should be provided as part of the manuscript or its supporting information, or deposited to a public repository. For example, in addition to summary statistics, the data points behind means, medians and variance measures should be available. If there are restrictions on publicly sharing data—e.g. participant privacy or use of data from a third party—those must be specified.--&gt;</p>
<p>Reviewer #1: Yes</p>
<p>Reviewer #2: No</p>
<p>**********</p>
<p>--&gt;4. Is the manuscript presented in an intelligible fashion and written in standard English?</p>
<p>PLOS ONE does not copyedit accepted manuscripts, so the language in submitted articles must be clear, correct, and unambiguous. Any typographical or grammatical errors should be corrected at revision, so please note any specific errors here.--&gt;</p>
<p>Reviewer #1: Yes</p>
<p>Reviewer #2: Yes</p>
<p>**********</p>
<p>--&gt;5. Review Comments to the Author</p>
<p>Please use the space provided to explain your answers to the questions above. You may also include additional comments for the author, including concerns about dual publication, research ethics, or publication ethics. (Please upload your review as an attachment if it exceeds 20,000 characters)--&gt;</p>
<p><bold>Reviewer #1:</bold>    I found this paper interesting to read and easy to follow. The experimental design and methods were well motivated and clearly explained. The statistical analysis is sound. The results were generally well presented.I found this paper interesting to read and easy to follow. The experimental design and methods were well motivated and clearly explained. The statistical analysis is sound. The results were generally well presented.</p>
<p>While my overall opinion on the paper leans positive, I have three main concerns, which I believe could at least be partly addressed in revisions:</p>
<p>1) QT is a television programme, where panellists give live answers to audience questions. This means that the real human responses that form the basis of this paper are transcribed from speech that occurred in a specific context. The LLM, on the other hand, generated text directly in a decontextualised setting, as provided by the system prompt (lines 84ff).</p>
<p>The authors acknowledge this discrepancy (e.g. line 443) but do not meaningfully engage with it. Fundamentally, human responses are not being evaluated in the setting and modality for which they were intended.</p>
<p>The paper would be stronger if it demonstrated that this discrepancy does not affect results. For example, human responses may include markers (such as references to other panellists names, disfluencies) that are perceived as inauthentic, incoherent, etc. when transcribed and decontextualised, but not when spoken on the programme. There may also be transcription errors, which LLMs would not have. This seems in scope for post-hoc analysis.</p>
<p>2) The paper finds that original content is different from impersonated content and concludes from this that AI may be used to generate targeted misinformation about the speaker’s point of view (lines 339ff). However, there is no analysis into *how* content differs in human vs LLM-impersonated content, and whether this difference does in fact correspond to “misrepresentation”. For this, it would be necessary to show that the LLM-impersonated answers differ from the human answers in their issue-specific stance.</p>
<p>It seems very plausible that LLMs follow different lines of argumentation but ultimately make similar points, which may (hypothetically) even be endorsed by the original speaker. Asking only about “difference in content”, which may be explained in many ways, cannot capture this. Consequently, the framing of the paper as relating to misinformation and deception seems only partly warranted.</p>
<p>3) The paper claims that human judgments are reliable but does not provide strong evidence for this. Agreement is modest at best (lines 359 following), and the supplementary variables show low agreement on several variables. Clearly, there is substantial noise in the data from subjective rating differences, which could be addressed by collecting and averaging across more ratings. If not, it would still be useful to be more open and specific about the (lack of) human agreement in the main body of the paper.</p>
<p>Minor notes:</p>
<p>- I found no mention of answer length in the paper. It seems very plausible that there is a difference in length between human and LLM answers, which may in turn affect at least some of the analyses. Some normalisation would likely be necessary?</p>
<p>- It may help the (non-UK) reader to note in the Data section that Question Time is a TV programme. The only mention of TV / television right now, I believe, is in the intro.</p>
<p>- On first reading, I was a bit confused by lines 117-119: “random speaker” may need brief explanation. Of course, this becomes clear later.</p>
<p><bold>Reviewer #2:</bold>    The manuscript explores the capacity of large language models (LLMs) to impersonate public figures in political debates. The topic is timely and uses an interesting dataset. The paper is well written and methodologically ambitious. However, the strength of the conclusions is weakened by shortcomings in the experimental design and in the statistical analysis. The findings are interesting but should be interpreted with greater caution, given the limitations discussed below.The manuscript explores the capacity of large language models (LLMs) to impersonate public figures in political debates. The topic is timely and uses an interesting dataset. The paper is well written and methodologically ambitious. However, the strength of the conclusions is weakened by shortcomings in the experimental design and in the statistical analysis. The findings are interesting but should be interpreted with greater caution, given the limitations discussed below.</p>
<p>1. Experimental Design</p>
<p>The overall three-track design is conceptually sound allowing the author to assess perceptions under different conditions. Nonetheless, the construct of authenticity is insufficiently defined. It remains unclear whether participants interpreted it as authorship likelihood, realism, or plausibility, which may conflate distinct psychological dimensions. Moreover, each participant evaluated only a few items, which limits the reliability of individual responses. Besides, more details about the randomization of questions and speakers is required to ensure data is balanced. The manual inspection of generated responses lacks formal coding criteria or inter-rater checks, reducing transparency and reproducibility.</p>
<p>Suggestions for improvement:</p>
<p>Clarify the exact wording and intended meaning of each measure, particularly “authenticity.” Provide evidence that the randomization of stimuli was balanced across conditions. Include or reference inter-rater checks for the manual screening of generated responses.</p>
<p>2. Measures and Data Collection</p>
<p>The study relies entirely on self-reported Likert-scale judgments. This approach is valid for perceptual studies, but complementary behavioral or comprehension-based measures (e.g., response time, detection accuracy) would have strengthened construct validity. Some measures—especially “content similarity”—may overlap conceptually with “authenticity,” potentially introducing redundancy.</p>
<p>Suggestions for improvement:</p>
<p>Consider clarifying the independence of constructs and, if possible, provide an additional objective indicator of perceptual discrimination between real and impersonated responses. Provide information about possible moderator factors.</p>
<p>3. Statistical Analysis</p>
<p>The paper presents a detailed description of the statistical workflow and justifies the use of non-parametric tests. However, several inconsistencies remain. The analysis averages ordinal Likert data and reports means, standard deviations, and Cohen’s d, assuming interval properties. This is not strictly correct for ranked data and can inflate effect sizes. Inter-rater reliability (reported Cronbach’s α) indicates moderate agreement at best, suggesting considerable noise in the judgments. The treatment of each outcome variable (authenticity, coherence, relevance, content) as independent is also problematic, as they are conceptually and statistically correlated. Finally, while some corrections are applied, the overall analytical structure would benefit from mixed-effects or ordinal regression models that account for the nested and repeated nature of the data.</p>
<p>Suggestions for improvement:</p>
<p>Report medians and interquartile ranges alongside or instead of means. Use effect size measures suited for ordinal data (e.g., rank-biserial correlation or other relevant measure.) Reassess reliability using intraclass correlation or a mixed-model framework. Consider a multivariate or mixed-effects approach to handle correlated measures and participant/item dependencies.</p>
<p>4. Interpretation of Results</p>
<p>The finding that impersonated responses are perceived as more authentic, coherent, and relevant than the originals is statistically supported but likely overstated. Given the modest reliability and limited item sampling, the conclusion that participants “cannot discern” AI-generated content should be presented more cautiously. It is plausible that judgments reflect linguistic fluency rather than genuine attribution of authorship.</p>
<p>Suggestions for improvement:</p>
<p>Moderate the interpretive claims and emphasize the perceptual, rather than deceptive, nature of the results. Discuss alternative explanations such as stylistic consistency or linguistic polish in LLM outputs.</p>
<p>**********</p>
<p>--&gt;6. PLOS authors have the option to publish the peer review history of their article (<ext-link ext-link-type="uri" xlink:href="https://journals.plos.org/plosone/s/editorial-and-peer-review-process#loc-peer-review-history" xlink:type="simple">what does this mean?</ext-link>). If published, this will include your full peer review and any attached files.). If published, this will include your full peer review and any attached files.</p>
<p>If you choose “no”, your identity will remain anonymous but your review may still be made public.</p>
<p><bold>Do you want your identity to be public for this peer review?</bold>    For information about this choice, including consent withdrawal, please see our For information about this choice, including consent withdrawal, please see our <ext-link ext-link-type="uri" xlink:href="https://www.plos.org/privacy-policy" xlink:type="simple">Privacy Policy</ext-link>..--&gt;</p>
<p>Reviewer #1: No</p>
<p>Reviewer #2: No</p>
<p>**********</p>
<p>[NOTE: If reviewer comments were submitted as an attachment file, they will be attached to this email and accessible via the submission site. Please log into your account, locate the manuscript record, and check for the action link "View Attachments". If this link does not appear, there are no attachment files.]</p>
<p>While revising your submission, please upload your figure files to the Preflight Analysis and Conversion Engine (PACE) digital diagnostic tool, <ext-link ext-link-type="uri" xlink:href="https://pacev2.apexcovantage.com/" xlink:type="simple">https://pacev2.apexcovantage.com/</ext-link>. PACE helps ensure that figures meet PLOS requirements. To use PACE, you must first register as a user. Registration is free. Then, login and navigate to the UPLOAD tab, where you will find detailed instructions on how to use the tool. If you encounter any issues or have any questions when using PACE, please email PLOS at . PACE helps ensure that figures meet PLOS requirements. To use PACE, you must first register as a user. Registration is free. Then, login and navigate to the UPLOAD tab, where you will find detailed instructions on how to use the tool. If you encounter any issues or have any questions when using PACE, please email PLOS at <email xlink:type="simple">figures@plos.org</email>. Please note that Supporting Information files do not need this step.. Please note that Supporting Information files do not need this step.</p>
</body>
</sub-article>
<sub-article article-type="author-comment" id="pone.0347757.r002">
<front-stub>
<article-id pub-id-type="doi">10.1371/journal.pone.0347757.r002</article-id>
<title-group>
<article-title>Author response to Decision Letter 1</article-title>
</title-group>
<related-object document-id="10.1371/journal.pone.0347757" document-id-type="doi" document-type="peer-reviewed-article" id="rel-obj002" link-type="rebutted-decision-letter" object-id="10.1371/journal.pone.0347757.r001" object-id-type="doi" object-type="decision-letter"/>
<custom-meta-group>
<custom-meta>
<meta-name>Submission Version</meta-name>
<meta-value>1</meta-value>
</custom-meta>
</custom-meta-group>
</front-stub>
<body>
<p><named-content content-type="author-response-date">25 Nov 2025</named-content></p>
<p>Response to Reviews</p>
<p>We thank the two reviewers, as well as the editor, for their feedback. Based on the comments, we prepared a revision of our manuscript for your consideration. To aid the reviewers, all changes are marked in blue within the updated manuscript.</p>
<p>Editor</p>
<p>Comment E.1: Reviewer 1 comments the design and analysis but raises concerns about the contextual asymmetry between the original human responses and the LLM-generated ones. Since the human content was transcribed from a live television programme (“Question Time”), it may include disfluencies, references, or transcription artifacts that affect perceptions differently than the directly generated LLM text. The reviewer suggests adding a post-hoc robustness check or qualitative analysis to demonstrate that these contextual or transcription-related differences do not drive the main effects.</p>
<p>Response E.1: We address this issue by conducting additional checks such as automated spell checking and a manual analysis of the responses to rule out such effects. The additional results are reported in the new supplemental material S3.</p>
<p>For more details, we refer to Response R1.1.</p>
<p>Comment E.2: Relatedly, Reviewer 1 encourages you to expand the discussion of content differences between human and impersonated responses. Specifically, the paper would benefit from showing how these differ substantively; for example, by analyzing whether impersonations diverge in stance or argumentation, rather than merely in surface-level content. This would clarify whether the framing of “misrepresentation” is warranted.</p>
<p>Response E.2: Based on these understandable concerns, we conduct additional manual analysis, in which we compare the stance conveyed by the actual and the generated response. While there are indeed cases in which the differences are rather in how a stance is conveyed, we also find a non-negligible number (26%) of the samples that we checked, where the differences in content were indeed differences in the stance that was conveyed.</p>
<p>For more details, we refer to Response 1.2.</p>
<p>Comment E.3: Finally, the reviewer notes that the reliability of human ratings is modest and recommends greater transparency about inter-rater variability, or, if feasible, additional averaging or collection of ratings to reduce subjective noise.</p>
<p>Response E.3: While we acknowledge this concern, we want to highlight that the differences in the judgments are neither random (i.e., completely unreliable) nor polar (i.e., opposite judgments), but rather on neighboring items on the Likert scale. We discuss these concerns in detail in the “Human judgment is reliable” section, which dedicates a whole paragraph to this issue. While we acknowledge that additional data collection could possibly help, we do not believe that there will be substantial changes, given our already very large sample size of 520 question/response pairs.</p>
<p>For more details, we refer to Response 1.3.</p>
<p>Comment E.4: Minor notes include adding a brief explanation that Question Time is a television programme, clarifying “random speaker” in the methods, and checking for potential answer length differences that could influence results.</p>
<p>Response E.4: We now address all these concerns explicitly in the paper, including an additional analysis on the random speaker assignments and the influence of answer lengths.</p>
<p>For more details, we refer to Responses 1.4, 1.5, and 1.6.</p>
<p>Comment E.5: 2. Conceptual and methodological refinements (Reviewer 2)</p>
<p>Reviewer 2 finds the study compelling but identifies several areas where methodological clarification is needed:</p>
<p>Construct clarity: The meaning of authenticity should be defined more precisely, whether it captures authorship likelihood, realism, or plausibility. Reviewers recommend clarifying how this construct differs from related ones (e.g., coherence or content similarity) and ensuring that participants understood it consistently.</p>
<p>Response E.5: We clarify our definition for authenticity as requested.</p>
<p>For more details, refer to Response 2.1.</p>
<p>Comment E.6: Experimental design: Provide more detail on the randomization and balancing of stimuli across conditions. If possible, include or reference any inter-rater checks used during the manual screening of generated responses.</p>
<p>Response E.6: We acknowledge that we should have been clearer regarding these issues. We almost always use the full data set and avoid sampling. The only case where we sample is the random assignment of speakers, as well as the new qualitative analysis of answers we already discussed above. We provide additional analysis for this aspect. We also failed to clarify why no inter-rater agreements are reported for our own manual analysis: all cases in which the second rater disagreed with the original judgment by the first rater were discussed and agreement was reached in all cases.</p>
<p>For more details, we refer to Responses 2.3 and 2.4.</p>
<p>Comment E.7: Measurement approach: Consider complementing subjective Likert judgments with an objective or behavioral indicator (e.g., response times or accuracy), or at least acknowledge the limitations of relying solely on self-reported data.</p>
<p>Response E.7: While this would yield insightful additional data, such data cannot be reliably obtained with the only survey method we used. While there is possibly an impact because of unobserved confounding effects, this is highly unlikely given our sample size. Still, we acknowledge this limitation in the “Human judgments are reliable” section.</p>
<p>For more details we refer to Response 2.5.</p>
<p>Comment E.8: Statistical analysis: The reviewer suggests using non-parametric or mixed-effects models better suited to ordinal and nested data, reporting medians and interquartile ranges, and revising effect size measures accordingly. You might also re-evaluate inter-rater reliability using intraclass correlation coefficients.</p>
<p>Response E.8: We carefully considered all suggestions by the reviewer and partially updated our methods. We revised the effect sizes to be non-parametric (almost no changes, one increase in in effect strength). All non-parametric markers are now reported next to the parametric mean and standard deviation in the appendix and we explain why we kept the mean and standard deviation in the main body of the paper. We did not switch to a different statistical model (i.e., no variant of a linear model), because we directly want to observe how the variables change and not how the combination of variables explains a change in dependent variable.</p>
<p>For more details, we refer to Response 2.6.</p>
<p>Comment E.9: Interpretation: Finally, temper the strength of the claims about deception or indistinguishability between AI and human responses. Reviewer 2 encourages reframing your findings as perceptual rather than deceptive effects, potentially driven by linguistic fluency or stylistic uniformity.</p>
<p>Response E.9: While our additional analysis further supports our results, we acknowledge that some wordings were too strong in the initial submission as our results are limited to the perception of our participants. We consequently edited the wording in key locations of the paper such as the abstract and the results section.</p>
<p>For more details, we refer to Response 2.7.</p>
<p>Journal Requirements:</p>
<p>Comment J.1: When submitting your revision, we need you to address these additional requirements.</p>
<p>Please ensure that your manuscript meets PLOS ONE's style requirements, including those for file naming. The PLOS ONE style templates can be found at</p>
<p><ext-link ext-link-type="uri" xlink:href="https://journals.plos.org/plosone/s/file?id=wjVg/PLOSOne_formatting_sample_main_body.pdf" xlink:type="simple">https://journals.plos.org/plosone/s/file?id=wjVg/PLOSOne_formatting_sample_main_body.pdf</ext-link> and</p>
<p><ext-link ext-link-type="uri" xlink:href="https://journals.plos.org/plosone/s/file?id=ba62/PLOSOne_formatting_sample_title_authors_affiliations.pdf" xlink:type="simple">https://journals.plos.org/plosone/s/file?id=ba62/PLOSOne_formatting_sample_title_authors_affiliations.pdf</ext-link></p>
<p>Response J.1: We updated all file names and types to match the requirements.</p>
<p>Comment J.2: Please update your submission to use the PLOS LaTeX template. The template and more information on our requirements for LaTeX submissions can be found at <ext-link ext-link-type="uri" xlink:href="http://journals.plos.org/plosone/s/latex" xlink:type="simple">http://journals.plos.org/plosone/s/latex</ext-link>.</p>
<p>Response J.2: We updated our submission to use the latest version of the LaTeX style file (Version 3.7 from Aug 2025)</p>
<p>Comment J.3: We note that the grant information you provided in the ‘Funding Information’ and ‘Financial Disclosure’ sections do not match.</p>
<p>When you resubmit, please ensure that you provide the correct grant numbers for the awards you received for your study in the ‘Funding Information’ section.</p>
<p>Response J.3: Thank you for reporting this. We now clarify the matter in the cover letter.</p>
<p>Comment J.4: Thank you for stating the following financial disclosure:</p>
<p>A.H. work was partially funded by the VolkswagenStiftung under grant Az. 98544 ‘Deliberation Laboratory’</p>
<p>URL: <ext-link ext-link-type="uri" xlink:href="https://www.volkswagenstiftung.de/" xlink:type="simple">https://www.volkswagenstiftung.de/</ext-link></p>
<p>The funders had no role in the study whatsoever.</p>
<p>Please state what role the funders took in the study. If the funders had no role, please state: "The funders had no role in study design, data collection and analysis, decision to publish, or preparation of the manuscript."</p>
<p>If this statement is not correct you must amend it as needed.</p>
<p>Please include this amended Role of Funder statement in your cover letter; we will change the online submission form on your behalf.</p>
<p>Response J.4: The funders had no role in the study and we clarify this in the cover letter.</p>
<p>Comment J.5: Thank you for stating the following in the Acknowledgments Section of your manuscript:</p>
<p>The work reported on in this paper was partially funded by the VolkswagenStiftung under grant Az. 98544 ‘Deliberation Laboratory’</p>
<p>We note that you have provided funding information that is not currently declared in your Funding Statement. However, funding information should not appear in the Acknowledgments section or other areas of your manuscript. We will only publish funding information present in the Funding Statement section of the online submission form.</p>
<p>Please remove any funding-related text from the manuscript and let us know how you would like to update your Funding Statement. Currently, your Funding Statement reads as follows:</p>
<p>A.H. work was partially funded by the VolkswagenStiftung under grant Az. 98544 ‘Deliberation Laboratory’</p>
<p>URL: <ext-link ext-link-type="uri" xlink:href="https://www.volkswagenstiftung.de/" xlink:type="simple">https://www.volkswagenstiftung.de/</ext-link></p>
<p>The funders had no role in the study whatsoever.</p>
<p>Please include your amended statements within your cover letter; we will change the online submission form on your behalf.</p>
<p>Response J.5: Thank you for this clarification. We provide the amended statements in our cover letter and drop the acknowledgements section from the manuscript.</p>
<p>Comment J.6: Thank you for uploading your study's underlying data set. Unfortunately, the repository you have noted in your Data Availability statement does not qualify as an acceptable data repository according to PLOS's standards.</p>
<p>At this time, please upload the minimal data set necessary to replicate your study's findings to a stable, public repository (such as figshare or Dryad) and provide us with the relevant URLs, DOIs, or accession numbers that may be used to access these data. For a list of recommended repositories and additional information on PLOS standards for data deposition, please see <ext-link ext-link-type="uri" xlink:href="https://journals.plos.org/plosone/s/recommended-repositories" xlink:type="simple">https://journals.plos.org/plosone/s/recommended-repositories</ext-link>.</p>
<p>Response J.6: We noticed that the DOI-reference to our long-term archive at Zenodo was missing the last character. This is now fixed and the new Data Availability statement is:</p>
<p>“All materials are available online in the form of a replication package that contains the data: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.5281/zenodo.12698364" xlink:type="simple">https://doi.org/10.5281/zenodo.12698364</ext-link>”</p>
<p>Comment J.7: Please include your full ethics statement in the ‘Methods’ section of your manuscript file. In your statement, please include the full name of the IRB or ethics committee who approved or waived your study, as well as whether or not you obtained informed written or verbal consent. If consent was waived for your study, please include this information in your statement as well.</p>
<p>Response J.7: We now have an additional subsection at the end of the Methods section with the full ethics statement. We extend the ethics statement to clarify that the obtained consent was written.</p>
<p>Comment J.8: If the reviewer comments include a recommendation to cite specific previously published works, please review and evaluate these publications to determine whether they are relevant and should be cited. There is no requirement to cite these works unless the editor has indicated otherwise.</p>
<p>Response J.8: Thank you for this comment. There were no recommendations for additional references in the reviews.</p>
<p>Reviewer 1</p>
<p>Comment R1.1: QT is a television programme, where panellists give live answers to audience questions. This means that the real human responses that form the basis of this paper are transcribed from speech that occurred in a specific context. The LLM, on the other hand, generated text directly in a decontextualised setting, as provided by the system prompt (lines 84ff).</p>
<p>The authors acknowledge this discrepancy (e.g. line 443) but do not meaningfully engage with it. Fundamentally, human responses are not being evaluated in the setting and modality for which they were intended.</p>
<p>The paper would be stronger if it demonstrated that this discrepancy does not affect results. For example, human responses may include markers (such as references to other panellists names, disfluencies) that are perceived as inauthentic, incoherent, etc. when transcribed and decontextualised, but not when spoken on the programme. There may also be transcription errors, which LLMs would not have. This seems in scope for post-hoc analysis.</p>
<p>Response R1.1: This is a good point and would be a viable alternative explanation for our results. We add additional analysis regarding this (and other) alternative hypothesis to the supplemental material in Appendix S3. We decided to exclude this from the main body of the paper in order to keep the presentation crisp and only add a reference to this new supplemental material to the end of the Section “Statistical analysis”. In summary, errors have no correlation with judgments. While transcriptions have more errors (2.4 on average vs. 0.26 for generated responses), most errors are minor, e.g., missing commas between clauses of compound sentences. In 10% of the original responses, we manually checked if actual responses contain references to other panelists and we do not find such cases. We add this information in Section “Original Debate Content.”</p>
<p>Comment R1.2: The paper finds that original content is different from impersonated content and concludes from this that AI may be used to generate targeted misinformation about the speaker’s point of view (lines 339ff). However, there is no analysis into *how* content differs in human vs LLM-impersonated content, and whether this difference does in fact correspond to “misrepresentation”. For this, it would be necessary to show that the LLM-impersonated answers differ from the human answers in their issue-specific stance.</p>
<p>It seems very plausible that LLMs follow different lines of argumentation but ultimately make similar points, which may (hypothetically) even be endorsed by the original speaker. Asking only about “difference in content”, which may be explained in many ways, cannot capture this. Consequently, the framing of the paper as relating to misinformation and deception seems only partly warranted.</p>
<p>Response R1.2: Thank you for this comment. We acknowledge that our definition of the term “content” as “difference in overall meaning” is too broad to directly justify these statements. To mitigate this, we include additional qualitative analysis on the share of data that was judged as different in content, indicated by a negative mean value of the “content” judgments by the study participants. The goal of this qualitative analysis is to establish whether the differences in content are indeed about misrepresentation. For this, we focus on the stance, i.e., if the response arrives at the same conclusion, argues for the same points or takes the same stance. This focus excludes the use of rhetorical devices as other means to convey a point, which may be used to make misrepresentations authentic. Additionally, we check if the respons</p>
<supplementary-material id="pone.0347757.s004" mimetype="application/pdf" position="float" xlink:href="info:doi/10.1371/journal.pone.0347757.s004" xlink:type="simple">
<label>Attachment</label>
<caption>
<p>Submitted filename: <named-content content-type="submitted-filename">PLOS-One-QT-GPT-Response.pdf</named-content></p>
</caption>
</supplementary-material>
</body>
</sub-article>
<sub-article article-type="aggregated-review-documents" id="pone.0347757.r003" specific-use="decision-letter">
<front-stub>
<article-id pub-id-type="doi">10.1371/journal.pone.0347757.r003</article-id>
<title-group>
<article-title>Decision Letter 1</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name name-style="western"><surname>Hassan</surname>
<given-names>Mohammad Salah</given-names>
</name>
<role>Academic Editor</role>
</contrib>
</contrib-group>
<permissions>
<copyright-year>2026</copyright-year>
<copyright-holder>Mohammad Salah HassanMohammad Salah Hassan</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">
<license-p>This is an open access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">Creative Commons Attribution License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited., which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited.</license-p></license>
</permissions>
<related-object document-id="10.1371/journal.pone.0347757" document-id-type="doi" document-type="article" id="rel-obj003" link-type="peer-reviewed-article"/>
<custom-meta-group>
<custom-meta>
<meta-name>Submission Version</meta-name>
<meta-value>1</meta-value>
</custom-meta>
</custom-meta-group>
</front-stub>
<body>
<p><named-content content-type="letter-date">23 Mar 2026</named-content></p>
<p>--&gt;PONE-D-25-15257R1--&gt;--&gt;LLM-impersonated debate contributions are more authentic, relevant and coherent than their original: A representative study using BBC1’s Question Time--&gt;--&gt;PLOS One</p>
<p>Dear Dr. Herbold,</p>
<p>Thank you for submitting your manuscript to PLOS ONE. After careful consideration, we feel that it has merit but does not fully meet PLOS ONE’s publication criteria as it currently stands. Therefore, we invite you to submit a revised version of the manuscript that addresses the points raised during the review process.--&gt;--&gt; --&gt;--&gt;</p>
<p>Please submit your revised manuscript by May 07 2026 11:59PM. If you will need more time than this to complete your revisions, please reply to this message or contact the journal office at <email xlink:type="simple">plosone@plos.org</email>. When you're ready to submit your revision, log on to <ext-link ext-link-type="uri" xlink:href="https://www.editorialmanager.com/pone/" xlink:type="simple">https://www.editorialmanager.com/pone/</ext-link> and select the 'Submissions Needing Revision' folder to locate your manuscript file.. When you're ready to submit your revision, log on to <ext-link ext-link-type="uri" xlink:href="https://www.editorialmanager.com/pone/" xlink:type="simple">https://www.editorialmanager.com/pone/</ext-link> and select the 'Submissions Needing Revision' folder to locate your manuscript file.</p>
<p>Please include the following items when submitting your revised manuscript:--&gt;</p>
<p><list list-type="bullet"><list-item><p>A letter that responds to each point raised by the academic editor and reviewer(s). You should upload this letter as a separate file labeled 'Response to Reviewers'.</p>
</list-item>
<list-item>
<p>A marked-up copy of your manuscript that highlights changes made to the original version. You should upload this as a separate file labeled 'Revised Manuscript with Track Changes'.</p>
</list-item>
<list-item>
<p>An unmarked version of your revised paper without tracked changes. You should upload this as a separate file labeled 'Manuscript'.</p></list-item></list></p>
<p>--&gt;If you would like to make changes to your financial disclosure, please include your updated statement in your cover letter. Guidelines for resubmitting your figure files are available below the reviewer comments at the end of this letter.</p>
<p>If applicable, we recommend that you deposit your laboratory protocols in protocols.io to enhance the reproducibility of your results. Protocols.io assigns your protocol its own identifier (DOI) so that it can be cited independently in the future. For instructions see: <ext-link ext-link-type="uri" xlink:href="https://journals.plos.org/plosone/s/submission-guidelines#loc-laboratory-protocols" xlink:type="simple">https://journals.plos.org/plosone/s/submission-guidelines#loc-laboratory-protocols</ext-link>. Additionally, PLOS ONE offers an option for publishing peer-reviewed Lab Protocol articles, which describe protocols hosted on protocols.io. Read more information on sharing protocols at . Additionally, PLOS ONE offers an option for publishing peer-reviewed Lab Protocol articles, which describe protocols hosted on protocols.io. Read more information on sharing protocols at <ext-link ext-link-type="uri" xlink:href="https://plos.org/protocols?utm_medium=editorial-email&amp;utm_source=authorletters&amp;utm_campaign=protocols" xlink:type="simple">https://plos.org/protocols?utm_medium=editorial-email&amp;utm_source=authorletters&amp;utm_campaign=protocols</ext-link>..</p>
<p>We look forward to receiving your revised manuscript.</p>
<p>Kind regards,</p>
<p>Mohammad Salah Hassan, Ph.D</p>
<p>Academic Editor</p>
<p>PLOS One</p>
<p>Journal Requirements:</p>
<p>1. If the reviewer comments include a recommendation to cite specific previously published works, please review and evaluate these publications to determine whether they are relevant and should be cited. There is no requirement to cite these works unless the editor has indicated otherwise.</p>
<p>2. Please review your reference list to ensure that it is complete and correct. If you have cited papers that have been retracted, please include the rationale for doing so in the manuscript text, or remove these references and replace them with relevant current references. Any changes to the reference list should be mentioned in the rebuttal letter that accompanies your revised manuscript. If you need to cite a retracted article, indicate the article’s retracted status in the References list and also include a citation and full reference for the retraction notice.</p>
<p>Additional Editor Comments :</p>
<p>Dear Authors,</p>
<p>The revised manuscript, PONE-D-25-15257R1, entitled “LLM-impersonated debate contributions are more authentic, relevant and coherent than their original: A representative study using BBC1’s Question Time,” has now been reviewed.</p>
<p>Please accept my sincere apologies for the delay in reaching this decision, as one of the invited reviewers did not reply and additional time was therefore needed to secure a further review.</p>
<p>Reviewer 1 recommends acceptance and confirms that the authors have adequately addressed the comments raised in the previous round. The reviewer considers the manuscript technically sound, the statistical analysis appropriate, the data availability satisfactory, and the language clear and intelligible.</p>
<p>A few very minor points remain, but these should be treated as final editorial clean-up rather than substantive revisions. In line with the reviewer’s observations, the authors should clarify the limitation concerning the mismatch in context and modality between the original human responses and the LLM-generated responses, and moderate the “threat” framing where the wording may still appear stronger than warranted. In addition, a few minor editorial inconsistencies should be corrected in the final files, including residual typographical and spacing issues, consistency of the manuscript title across all submitted documents, and the final Data Availability link/DOI.</p>
<p>These minor matters do not affect the overall positive recommendation. Based on the reviewer’s report and my own assessment, I recommend acceptance of the manuscript subject to final minor editorial polishing.</p>
<p>Kind regards,</p>
<p>Mohammed Salah Alazzawi, PhD</p>
<p>[Note: HTML markup is below. Please do not edit.]</p>
<p>Reviewers' comments:</p>
<p>Reviewer's Responses to Questions</p>
<p>--&gt;<bold>Comments to the Author</bold></p>
<p>1. If the authors have adequately addressed your comments raised in a previous round of review and you feel that this manuscript is now acceptable for publication, you may indicate that here to bypass the “Comments to the Author” section, enter your conflict of interest statement in the “Confidential to Editor” section, and submit your "Accept" recommendation.--&gt;</p>
<p>Reviewer #1: All comments have been addressed</p>
<p>Reviewer #3: All comments have been addressed</p>
<p>**********</p>
<p>--&gt;2. Is the manuscript technically sound, and do the data support the conclusions?</p>
<p>The manuscript must describe a technically sound piece of scientific research with data that supports the conclusions. Experiments must have been conducted rigorously, with appropriate controls, replication, and sample sizes. The conclusions must be drawn appropriately based on the data presented.--&gt;</p>
<p>Reviewer #1: Yes</p>
<p>Reviewer #3: Yes</p>
<p>**********</p>
<p>--&gt;3. Has the statistical analysis been performed appropriately and rigorously? --&gt;</p>
<p>Reviewer #1: Yes</p>
<p>Reviewer #3: Yes</p>
<p>**********</p>
<p>--&gt;4. Have the authors made all data underlying the findings in their manuscript fully available?</p>
<p>The <ext-link ext-link-type="uri" xlink:href="http://www.plosone.org/static/policies.action#sharing" xlink:type="simple">PLOS Data policy</ext-link> requires authors to make all data underlying the findings described in their manuscript fully available without restriction, with rare exception (please refer to the Data Availability Statement in the manuscript PDF file). The data should be provided as part of the manuscript or its supporting information, or deposited to a public repository. For example, in addition to summary statistics, the data points behind means, medians and variance measures should be available. If there are restrictions on publicly sharing data—e.g. participant privacy or use of data from a third party—those must be specified.requires authors to make all data underlying the findings described in their manuscript fully available without restriction, with rare exception (please refer to the Data Availability Statement in the manuscript PDF file). The data should be provided as part of the manuscript or its supporting information, or deposited to a public repository. For example, in addition to summary statistics, the data points behind means, medians and variance measures should be available. If there are restrictions on publicly sharing data—e.g. participant privacy or use of data from a third party—those must be specified.--&gt;</p>
<p>Reviewer #1: Yes</p>
<p>Reviewer #3: Yes</p>
<p>**********</p>
<p>--&gt;5. Is the manuscript presented in an intelligible fashion and written in standard English?</p>
<p>PLOS ONE does not copyedit accepted manuscripts, so the language in submitted articles must be clear, correct, and unambiguous. Any typographical or grammatical errors should be corrected at revision, so please note any specific errors here.--&gt;</p>
<p>Reviewer #1: Yes</p>
<p>Reviewer #3: Yes</p>
<p>**********</p>
<p>--&gt;6. Review Comments to the Author</p>
<p>Please use the space provided to explain your answers to the questions above. You may also include additional comments for the author, including concerns about dual publication, research ethics, or publication ethics. (Please upload your review as an attachment if it exceeds 20,000 characters)--&gt;</p>
<p>Reviewer #1: I thank the authors for their detailed response to my first review. My concerns have largely been addressed, and I am happy to recommend acceptance. However, I would encourage the authors to engage with the following points:</p>
<p>1) I still believe that the mismatch in context and modality between the original human responses and the LLM-generated responses cannot be ruled out as at least a partial explanation of results. Length and prevalence of spelling errors may not individually have strong correlations with perceptions of authenticity etc., but their combination (and other unobserved factors) may still partially explain human judgments. A true like-for-like comparison would compare human-written textual responses to LLM-written responses. Since this comparison cannot be made, I think this should be clearly acknowledged as a limitation in the main body discussion.</p>
<p>2) I would encourage the authors to be clearer in their “threat model”: What exactly is the new threat created by LLM impersonation? And to what extent do the experiments in this paper measure this new threat? For example, I would argue that not all “made-up content” is equally problematic – LLM impersonation poses a threat only if the LLM-generated content *misrepresents* the speaker’s views *while still being perceived as authentic*. This is not something the paper tests for! The small qualitative analysis shows that there are 13 of 50 cases where the LLM-generated response misaligns in stance, but this is not enough to make claims about general LLM tendency to misrepresent, or how this misrepresentation impacts authenticity judgments. Personally, I would therefore tone down the “threat” language throughout the paper and simply provide a sober description of results related to authenticity, and pointers to open questions for future work.</p>
<p>Minor notes:</p>
<p>- line 18: typo and missing whitespace</p>
<p>- line 184: missing whitespace</p>
<p>Reviewer #3: This paper has been significantly improved. Thank you very much for your efforts in making the manuscript stronger and more polished after the revision.</p>
<p>**********</p>
<p>--&gt;7. PLOS authors have the option to publish the peer review history of their article (<ext-link ext-link-type="uri" xlink:href="https://journals.plos.org/plosone/s/editorial-and-peer-review-process#loc-peer-review-history" xlink:type="simple">what does this mean?</ext-link>). If published, this will include your full peer review and any attached files.). If published, this will include your full peer review and any attached files.</p>
<p>If you choose “no”, your identity will remain anonymous but your review may still be made public.</p>
<p><bold>Do you want your identity to be public for this peer review?</bold>    For information about this choice, including consent withdrawal, please see our For information about this choice, including consent withdrawal, please see our <ext-link ext-link-type="uri" xlink:href="https://www.plos.org/privacy-policy" xlink:type="simple">Privacy Policy</ext-link>..--&gt;</p>
<p>Reviewer #1: No</p>
<p>Reviewer #3: No</p>
<p>**********</p>
<p>[NOTE: If reviewer comments were submitted as an attachment file, they will be attached to this email and accessible via the submission site. Please log into your account, locate the manuscript record, and check for the action link "View Attachments". If this link does not appear, there are no attachment files.]</p>
<p>To ensure your figures meet our technical requirements, please review our figure guidelines: <ext-link ext-link-type="uri" xlink:href="https://journals.plos.org/plosone/s/figures" xlink:type="simple">https://journals.plos.org/plosone/s/figures</ext-link></p>
<p>You may also use PLOS’s free figure tool, NAAS, to help you prepare publication quality figures: <ext-link ext-link-type="uri" xlink:href="https://journals.plos.org/plosone/s/figures#loc-tools-for-figure-preparation" xlink:type="simple">https://journals.plos.org/plosone/s/figures#loc-tools-for-figure-preparation</ext-link>.</p>
<p>NAAS will assess whether your figures meet our technical requirements by comparing each figure against our figure specifications.</p>
<p>--&gt;</p>
</body>
</sub-article>
<sub-article article-type="author-comment" id="pone.0347757.r004">
<front-stub>
<article-id pub-id-type="doi">10.1371/journal.pone.0347757.r004</article-id>
<title-group>
<article-title>Author response to Decision Letter 2</article-title>
</title-group>
<related-object document-id="10.1371/journal.pone.0347757" document-id-type="doi" document-type="peer-reviewed-article" id="rel-obj004" link-type="rebutted-decision-letter" object-id="10.1371/journal.pone.0347757.r003" object-id-type="doi" object-type="decision-letter"/>
<custom-meta-group>
<custom-meta>
<meta-name>Submission Version</meta-name>
<meta-value>2</meta-value>
</custom-meta>
</custom-meta-group>
</front-stub>
<body>
<p><named-content content-type="author-response-date">25 Mar 2026</named-content></p>
<p>We fixed all reported minor issues. For the two comments from R1, we added the following to the main body of the manuscript in the Section "Discussion and conclusion" (1) to highlight the limitation due to possible unobservable factors affecting our results and (2) to frame the generalizability more carefully:</p>
<p>(1) "While we rule out length and grammatical errors as possible sources for differences in authenticity, we cannot rule out that there are unobserved factors introduced by our experiment design."</p>
<p>(2) "We note that our setting did not study this targeted misinformation, i.e., we did not prescribe the position the LLM should express. Future work needs to study if LLMs are still perceived as authentic when used in such a targeted manner."</p>
</body>
</sub-article>
<sub-article article-type="editor-report" id="pone.0347757.r005" specific-use="decision-letter">
<front-stub>
<article-id pub-id-type="doi">10.1371/journal.pone.0347757.r005</article-id>
<title-group>
<article-title>Decision Letter 2</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name name-style="western"><surname>Hassan</surname>
<given-names>Mohammad Salah</given-names>
</name>
<role>Academic Editor</role>
</contrib>
</contrib-group>
<permissions>
<copyright-year>2026</copyright-year>
<copyright-holder>Mohammad Salah HassanMohammad Salah Hassan</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">
<license-p>This is an open access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">Creative Commons Attribution License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited., which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited.</license-p></license>
</permissions>
<related-object document-id="10.1371/journal.pone.0347757" document-id-type="doi" document-type="article" id="rel-obj005" link-type="peer-reviewed-article"/>
<custom-meta-group>
<custom-meta>
<meta-name>Submission Version</meta-name>
<meta-value>2</meta-value>
</custom-meta>
</custom-meta-group>
</front-stub>
<body>
<p><named-content content-type="letter-date">7 Apr 2026</named-content></p>
<p>LLM-impersonated debate contributions are more authentic, relevant and coherent than their original: A representative study using BBC1’s Question Time</p>
<p>PONE-D-25-15257R2</p>
<p>Dear Dr. Authors,</p>
<p>We’re pleased to inform you that your manuscript has been judged scientifically suitable for publication and will be formally accepted for publication once it meets all outstanding technical requirements.</p>
<p>Within one week, you’ll receive an e-mail detailing the required amendments. When these have been addressed, you’ll receive a formal acceptance letter and your manuscript will be scheduled for publication.</p>
<p>An invoice will be generated when your article is formally accepted. Please note, if your institution has a publishing partnership with PLOS and your article meets the relevant criteria, all or part of your publication costs will be covered. Please make sure your user information is up-to-date by logging into Editorial Manager at <ext-link ext-link-type="uri" xlink:href="https://www.editorialmanager.com/pone/" xlink:type="simple">Editorial Manager®</ext-link> and clicking the ‘Update My Information' link at the top of the page. For questions related to billing, please contact  and clicking the ‘Update My Information' link at the top of the page. For questions related to billing, please contact <ext-link ext-link-type="uri" xlink:href="https://plos.my.site.com/s/" xlink:type="simple">billing support</ext-link>..</p>
<p>If your institution or institutions have a press office, please notify them about your upcoming paper to help maximize its impact. If they’ll be preparing press materials, please inform our press team as soon as possible -- no later than 48 hours after receiving the formal acceptance. Your manuscript will remain under strict press embargo until 2 pm Eastern Time on the date of publication. For more information, please contact onepress@plos.org.</p>
<p>Kind regards,</p>
<p>Mohammad Salah Hassan, Ph.D</p>
<p>Academic Editor</p>
<p>PLOS One</p>
<p>Additional Editor Comments (optional):</p>
<p>Dear Authors,</p>
<p>Thank you for submitting the revised version of your manuscript, “LLM-impersonated debate contributions are more authentic, relevant and coherent than their original: A representative study using BBC1’s Question Time.”</p>
<p>The manuscript has improved substantially, and the concerns raised in the previous round have been adequately addressed.</p>
<p>I am pleased to inform you that I recommend acceptance of the manuscript for publication in PLOS ONE.</p>
<p>Kind regards,</p>
<p>Reviewers' comments:</p>
</body>
</sub-article>
<sub-article article-type="editor-report" id="pone.0347757.r006" specific-use="acceptance-letter">
<front-stub>
<article-id pub-id-type="doi">10.1371/journal.pone.0347757.r006</article-id>
<title-group>
<article-title>Acceptance letter</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name name-style="western"><surname>Hassan</surname>
<given-names>Mohammad Salah</given-names>
</name>
<role>Academic Editor</role>
</contrib>
</contrib-group>
<permissions>
<copyright-year>2026</copyright-year>
<copyright-holder>Mohammad Salah HassanMohammad Salah Hassan</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">
<license-p>This is an open access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">Creative Commons Attribution License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited., which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited.</license-p></license>
</permissions>
<related-object document-id="10.1371/journal.pone.0347757" document-id-type="doi" document-type="article" id="rel-obj006" link-type="peer-reviewed-article"/>
</front-stub>
<body>
<p>PONE-D-25-15257R2</p>
<p>PLOS One</p>
<p>Dear Dr. Herbold,</p>
<p>I'm pleased to inform you that your manuscript has been deemed suitable for publication in PLOS One. Congratulations! Your manuscript is now being handed over to our production team.</p>
<p>At this stage, our production department will prepare your paper for publication. This includes ensuring the following:</p>
<p>* All references, tables, and figures are properly cited</p>
<p>* All relevant supporting information is included in the manuscript submission,</p>
<p>* There are no issues that prevent the paper from being properly typeset</p>
<p>You will receive further instructions from the production team, including instructions on how to review your proof when it is ready. Please keep in mind that we are working through a large volume of accepted articles, so please give us a few days to review your paper and let you know the next and final steps.</p>
<p>Lastly, if your institution or institutions have a press office, please let them know about your upcoming paper now to help maximize its impact. If they'll be preparing press materials, please inform our press team within the next 48 hours. Your manuscript will remain under strict press embargo until 2 pm Eastern Time on the date of publication. For more information, please contact onepress@plos.org.</p>
<p>You will receive an invoice from PLOS for your publication fee after your manuscript has reached the completed accept phase. If you receive an email requesting payment before acceptance or for any other service, this may be a phishing scheme. Learn how to identify phishing emails and protect your accounts at <ext-link ext-link-type="uri" xlink:href="https://explore.plos.org/phishing" xlink:type="simple">https://explore.plos.org/phishing</ext-link>.</p>
<p>If we can help with anything else, please email us at customercare@plos.org.</p>
<p>Thank you for submitting your work to PLOS ONE and supporting open access.</p>
<p>Kind regards,</p>
<p>PLOS ONE Editorial Office Staff</p>
<p>on behalf of</p>
<p>Dr. Mohammad Salah Hassan</p>
<p>Academic Editor</p>
<p>PLOS One</p>
</body>
</sub-article>
</article>