<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.1d3 20150301//EN" "http://jats.nlm.nih.gov/publishing/1.1d3/JATS-journalpublishing1.dtd">
<article article-type="research-article" dtd-version="1.1d3" xml:lang="en" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="nlm-ta">PLoS ONE</journal-id>
<journal-id journal-id-type="publisher-id">plos</journal-id>
<journal-id journal-id-type="pmc">plosone</journal-id>
<journal-title-group>
<journal-title>PLOS ONE</journal-title>
</journal-title-group>
<issn pub-type="epub">1932-6203</issn>
<publisher>
<publisher-name>Public Library of Science</publisher-name>
<publisher-loc>San Francisco, CA USA</publisher-loc>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">PONE-D-18-16878</article-id>
<article-id pub-id-type="doi">10.1371/journal.pone.0205999</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Research Article</subject>
</subj-group>
<subj-group subj-group-type="Discipline-v3"><subject>People and places</subject><subj-group><subject>Population groupings</subject><subj-group><subject>Age groups</subject><subj-group><subject>Children</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3"><subject>People and places</subject><subj-group><subject>Population groupings</subject><subj-group><subject>Families</subject><subj-group><subject>Children</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3"><subject>Biology and life sciences</subject><subj-group><subject>Psychology</subject><subj-group><subject>Behavior</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3"><subject>Social sciences</subject><subj-group><subject>Psychology</subject><subj-group><subject>Behavior</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3"><subject>Engineering and technology</subject><subj-group><subject>Mechanical engineering</subject><subj-group><subject>Robotics</subject><subj-group><subject>Robots</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3"><subject>Engineering and technology</subject><subj-group><subject>Mechanical engineering</subject><subj-group><subject>Robotics</subject><subj-group><subject>Robotic behavior</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3"><subject>Biology and life sciences</subject><subj-group><subject>Anatomy</subject><subj-group><subject>Head</subject><subj-group><subject>Face</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3"><subject>Medicine and health sciences</subject><subj-group><subject>Anatomy</subject><subj-group><subject>Head</subject><subj-group><subject>Face</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3"><subject>Biology and life sciences</subject><subj-group><subject>Psychology</subject><subj-group><subject>Behavior</subject><subj-group><subject>Recreation</subject><subj-group><subject>Games</subject></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3"><subject>Social sciences</subject><subj-group><subject>Psychology</subject><subj-group><subject>Behavior</subject><subj-group><subject>Recreation</subject><subj-group><subject>Games</subject></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3"><subject>Computer and information sciences</subject><subj-group><subject>Data acquisition</subject></subj-group></subj-group><subj-group subj-group-type="Discipline-v3"><subject>Engineering and technology</subject><subj-group><subject>Equipment</subject><subj-group><subject>Optical equipment</subject><subj-group><subject>Cameras</subject></subj-group></subj-group></subj-group></subj-group></article-categories>
<title-group>
<article-title>The PInSoRo dataset: Supporting the data-driven study of child-child and child-robot social dynamics</article-title>
<alt-title alt-title-type="running-head">The PInSoRo dataset of child-child and child-robot social dynamics</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes" xlink:type="simple">
<contrib-id authenticated="true" contrib-id-type="orcid">http://orcid.org/0000-0002-3391-8876</contrib-id>
<name name-style="western">
<surname>Lemaignan</surname> <given-names>Séverin</given-names></name>
<role content-type="http://credit.casrai.org/">Conceptualization</role>
<role content-type="http://credit.casrai.org/">Investigation</role>
<role content-type="http://credit.casrai.org/">Methodology</role>
<role content-type="http://credit.casrai.org/">Software</role>
<role content-type="http://credit.casrai.org/">Supervision</role>
<role content-type="http://credit.casrai.org/">Writing – original draft</role>
<role content-type="http://credit.casrai.org/">Writing – review &amp; editing</role>
<xref ref-type="aff" rid="aff001"><sup>1</sup></xref>
<xref ref-type="corresp" rid="cor001">*</xref>
</contrib>
<contrib contrib-type="author" xlink:type="simple">
<name name-style="western">
<surname>Edmunds</surname> <given-names>Charlotte E. R.</given-names></name>
<role content-type="http://credit.casrai.org/">Data curation</role>
<role content-type="http://credit.casrai.org/">Formal analysis</role>
<role content-type="http://credit.casrai.org/">Investigation</role>
<role content-type="http://credit.casrai.org/">Methodology</role>
<role content-type="http://credit.casrai.org/">Writing – review &amp; editing</role>
<xref ref-type="aff" rid="aff002"><sup>2</sup></xref>
</contrib>
<contrib contrib-type="author" xlink:type="simple">
<name name-style="western">
<surname>Senft</surname> <given-names>Emmanuel</given-names></name>
<role content-type="http://credit.casrai.org/">Software</role>
<role content-type="http://credit.casrai.org/">Writing – review &amp; editing</role>
<xref ref-type="aff" rid="aff002"><sup>2</sup></xref>
</contrib>
<contrib contrib-type="author" xlink:type="simple">
<name name-style="western">
<surname>Belpaeme</surname> <given-names>Tony</given-names></name>
<role content-type="http://credit.casrai.org/">Conceptualization</role>
<role content-type="http://credit.casrai.org/">Funding acquisition</role>
<role content-type="http://credit.casrai.org/">Supervision</role>
<xref ref-type="aff" rid="aff002"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff003"><sup>3</sup></xref>
</contrib>
</contrib-group>
<aff id="aff001">
<label>1</label>
<addr-line>Bristol Robotics Lab, University of the West of England, Bristol, United Kingdom</addr-line>
</aff>
<aff id="aff002">
<label>2</label>
<addr-line>Centre for Robotics and Neural Systems, University of Plymouth, Plymouth, United Kingdom</addr-line>
</aff>
<aff id="aff003">
<label>3</label>
<addr-line>IDLab – imec, Ghent University, Ghent, Belgium</addr-line>
</aff>
<contrib-group>
<contrib contrib-type="editor" xlink:type="simple">
<name name-style="western">
<surname>Goodman</surname> <given-names>Michael L.</given-names></name>
<role>Editor</role>
<xref ref-type="aff" rid="edit1"/>
</contrib>
</contrib-group>
<aff id="edit1">
<addr-line>University of Texas Medical Branch at Galveston, UNITED STATES</addr-line>
</aff>
<author-notes>
<fn fn-type="conflict" id="coi001">
<p>The authors have declared that no competing interests exist.</p>
</fn>
<corresp id="cor001">* E-mail: <email xlink:type="simple">severin.lemaignan@brl.ac.uk</email></corresp>
</author-notes>
<pub-date pub-type="collection">
<year>2018</year>
</pub-date>
<pub-date pub-type="epub">
<day>19</day>
<month>10</month>
<year>2018</year>
</pub-date>
<volume>13</volume>
<issue>10</issue>
<elocation-id>e0205999</elocation-id>
<history>
<date date-type="received">
<day>5</day>
<month>6</month>
<year>2018</year>
</date>
<date date-type="accepted">
<day>4</day>
<month>10</month>
<year>2018</year>
</date>
</history>
<permissions>
<copyright-year>2018</copyright-year>
<copyright-holder>Lemaignan et al</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">
<license-p>This is an open access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">Creative Commons Attribution License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited.</license-p>
</license>
</permissions>
<self-uri content-type="pdf" xlink:href="info:doi/10.1371/journal.pone.0205999"/>
<abstract>
<p>The study of the fine-grained social dynamics between children is a methodological challenge, yet a good understanding of how social interaction between children unfolds is important not only to Developmental and Social Psychology, but recently has become relevant to the neighbouring field of Human-Robot Interaction (HRI). Indeed, child-robot interactions are increasingly being explored in domains which require longer-term interactions, such as healthcare and education. For a robot to behave in an appropriate manner over longer time scales, its behaviours have to be contingent and meaningful to the unfolding relationship. Recognising, interpreting and generating sustained and engaging social behaviours is as such an important—and essentially, open—research question. We believe that the recent progress of machine learning opens new opportunities in terms of both analysis and synthesis of complex social dynamics. To support these approaches, we introduce in this article a novel, open dataset of child social interactions, designed with data-driven research methodologies in mind. Our data acquisition methodology relies on an engaging, methodologically sound, but purposefully underspecified <italic>free-play</italic> interaction. By doing so, we capture a rich set of behavioural patterns occurring in natural social interactions between children. The resulting dataset, called the PInSoRo dataset, comprises 45+ hours of hand-coded recordings of social interactions between 45 child-child pairs and 30 child-robot pairs. In addition to annotations of social constructs, the dataset includes fully calibrated video recordings, 3D recordings of the faces, skeletal informations, full audio recordings, as well as game interactions.</p>
</abstract>
<funding-group>
<award-group id="award001">
<funding-source>
<institution-wrap>
<institution-id institution-id-type="funder-id">http://dx.doi.org/10.13039/100010661</institution-id>
<institution>Horizon 2020 Framework Programme</institution>
</institution-wrap>
</funding-source>
<award-id>657227</award-id>
<principal-award-recipient>
<contrib-id authenticated="true" contrib-id-type="orcid">http://orcid.org/0000-0002-3391-8876</contrib-id>
<name name-style="western">
<surname>Lemaignan</surname> <given-names>Séverin</given-names></name>
</principal-award-recipient>
</award-group>
<award-group id="award002">
<funding-source>
<institution-wrap>
<institution-id institution-id-type="funder-id">http://dx.doi.org/10.13039/100010661</institution-id>
<institution>Horizon 2020 Framework Programme</institution>
</institution-wrap>
</funding-source>
<award-id>688014</award-id>
<principal-award-recipient>
<name name-style="western">
<surname>Belpaeme</surname> <given-names>Tony</given-names></name>
</principal-award-recipient>
</award-group>
<funding-statement>This work was primarily funded by the European Union H2020 "Donating Robots a Theory of Mind" project (grant id #657227) awarded to SL. It received additional funding from the European Union H2020 "Second Language Tutoring using Social Robots" project (grant id #688014), awarded to TB. The funders had no role in study design, data collection and analysis, decision to publish, or preparation of the manuscript.</funding-statement>
</funding-group>
<counts>
<fig-count count="11"/>
<table-count count="4"/>
<page-count count="19"/>
</counts>
<custom-meta-group>
<custom-meta id="data-availability">
<meta-name>Data Availability</meta-name>
<meta-value>The dataset is freely available to any interested researcher. Due to ethical and data protection regulations, the dataset is however made available in two forms: - a public, Creative Commons licensed, version that does not include any video material of the children (no video nor audio streams), and hosted on the Zenodo open-data platform: <ext-link ext-link-type="uri" xlink:href="https://zenodo.org/record/1043508" xlink:type="simple">https://zenodo.org/record/1043508</ext-link>. - the complete version that includes all video streams is freely available as well, but interested researchers must first fill a data protection form. The detail of the procedure are available online: <ext-link ext-link-type="uri" xlink:href="https://freeplay-sandbox.github.io/application" xlink:type="simple">https://freeplay-sandbox.github.io/application</ext-link>.</meta-value>
</custom-meta>
</custom-meta-group>
</article-meta>
</front>
<body>
<sec id="sec001" sec-type="intro">
<title>Introduction</title>
<sec id="sec002">
<title>Studying social interactions</title>
<p>Studying social interactions requires a social <italic>situation</italic> that effectively elicits interactions between the participants. Such a situation is typically scaffolded by a social task, and consequently, the nature of this task influences in fundamental ways the kind of interactions that might be observed and analysed. In particular, the socio-cognitive tasks commonly found in both the experimental psychology and human-robot interaction (HRI) literature often have a narrow focus: because they aim at studying one (or a few) specific social or cognitive skills in isolation and in a controlled manner, these tasks are typically conceptually simple and highly constrained (for instance, object hand-over tasks; perspective-taking tasks; etc.). While these focused endeavours are important and necessary, they do not adequately reflect the complexity and dynamics of real-world, natural interactions (as discussed by Baxter et al. in [<xref ref-type="bibr" rid="pone.0205999.ref001">1</xref>], in the context of HRI). Consequently, we need to investigate richer interactions, scaffolded by socio-cognitive tasks that:</p>
<list list-type="bullet">
<list-item>
<p>are long enough and varied enough to elicit a large range of interaction situations;</p>
</list-item>
<list-item>
<p>foster rich multi-modal interactions, such as simultaneous speech, gesture, and gaze behaviours;</p>
</list-item>
<list-item>
<p>are not over-specified, in order to maximise natural, non-contrived behaviours;</p>
</list-item>
<list-item>
<p>evidence complex social dynamics, such as rhythmic coupling, joint attention, implicit turn-taking;</p>
</list-item>
<list-item>
<p>include a level of non-determinism and unpredictability.</p>
</list-item>
</list>
<p specific-use="continuation">The challenge lies in designing a social task that exhibits these features <italic>while maintaining</italic> essential scientific properties (repeatability; replicability; robust metrics) as well as good practical properties (not requiring unique or otherwise very costly experimental environments; not requiring very specific hardware or robotic platform; easy deployment; short enough experimental sessions to allow for large groups of participants).</p>
<p>Looking specifically at social interactions amongst children, we present in the next section our take on this challenge, and we introduce a novel task of free play. The task is designed to elicit rich, complex, varied social interactions while supporting rigorous scientific methodologies, and is well suited for studying both child-child and child-robot interactions.</p>
</sec>
<sec id="sec003">
<title>Social play</title>
<p>Our interaction paradigm is based on free and playful interactions (hereafter, <italic>free play</italic>) in what we call a <italic>sandboxed environment</italic>. In other words, while the interaction is free (participants are not directed to perform any particular task beyond playing), the activity is both <italic>scaffolded</italic> and <italic>constrained</italic> by the setup mediating the interaction (a large interactive table), in a similar way to children freely playing with sand within the boundaries of a sandpit. Consequently, while participants engage in open-ended and non-directed activity, the play situation is framed to be easily reproducible as well as practical to record and analyse.</p>
<p>This initial description frames the socio-cognitive interactions that might be observed and studied: playful, dyadic, face-to-face interactions. While gestures and manipulations (including joint manipulations) play an important role in this paradigm, the participants do not typically move much during the interaction. Because it builds on play, this paradigm is also primarily targeted to practitioners in the field of child-child or child-robot social interactions.</p>
<p>The choice of a playful interaction is supported by the wealth of social situations and social behaviours that play elicits (see for instance parts 3 and 4 of [<xref ref-type="bibr" rid="pone.0205999.ref002">2</xref>]). Most of the research in this field builds on the early work of Parten who established five <italic>stages of play</italic> [<xref ref-type="bibr" rid="pone.0205999.ref003">3</xref>], corresponding to different stages of development, and accordingly associated with typical age ranges: (<italic>a</italic>) <italic>solitary (independent) play</italic> (age 2-3): child playing separately from others, with no reference to what others are doing; (<italic>b</italic>) <italic>onlooker play</italic> (age 2.5-3.5): child watching others play; may engage in conversation but not engage in doing; true focus on the children at play; (<italic>c</italic>) <italic>parallel play</italic> (also called adjacent play, social co-action, age 2.5-3.5): children playing with similar objects, clearly beside others but not with them; (<italic>d</italic>) <italic>associative play</italic> (age 3-4): child playing with others without organization of play activity; initiating or responding to interaction with peers; (<italic>e</italic>) <italic>cooperative play</italic> (age 4+): coordinating one’s behavior with that of a peer; everyone has a role, with the emergence of a sense of belonging to a group; beginning of “team work.”</p>
<p>These five stages of play have been extensively discussed and refined over the last century, yet remain remarkably widely accepted. It must be noted that the age ranges are only indicative. In particular, most of the early behaviours still occur at times by older children.</p>
</sec>
<sec id="sec004">
<title>Machine learning, robots and social behaviours</title>
<p>The data-driven study of social mechanisms is still an emerging field, and only limited literature is available.</p>
<p>The use of interaction datasets to teach artificial agents (robots) how to socially behave has been previously explored, and can be considered as the extension of the traditional learning from demonstration (LfD) paradigms to social interactions [<xref ref-type="bibr" rid="pone.0205999.ref004">4</xref>, <xref ref-type="bibr" rid="pone.0205999.ref005">5</xref>]. However, existing research focuses on low-level identification or generation of brief, isolated behaviours, including social gestures [<xref ref-type="bibr" rid="pone.0205999.ref006">6</xref>] and gazing behaviours [<xref ref-type="bibr" rid="pone.0205999.ref007">7</xref>].</p>
<p>Based on a human-human interaction dataset, Liu et al. [<xref ref-type="bibr" rid="pone.0205999.ref008">8</xref>] have investigated machine learning approaches to learn longer interaction sequences. Using unsupervised learning, they train a robot to act as a shop-keeper, generating both speech and socially acceptable motions. Their approach remains task-specific, and they report only limited success. They however emphasise the “life-likeness” of the generated behaviours.</p>
<p>This burgeoning interest in the research community for the data-driven study of social responses is however impaired by the lack of structured research efforts. In particular, there is only limited availability of large and open datasets of social interactions, suitable for machine-learning applications.</p>
<p>One such dataset is the <italic>Multimodal Dyadic Behavior Dataset</italic> (<italic>MMDB</italic>, [<xref ref-type="bibr" rid="pone.0205999.ref009">9</xref>]). It comprises of 160 sessions of 3 to 5 minute child-adult interactions. During these interactions, the experimenter plays with toddlers (1.5 to 2.5 years old) in a semi-structured manner. The dataset includes video streams of the faces and the room, audio, physiological data (electrodermal activity) as well as manual annotations of specific behaviours (like gaze to the examiner, laughter, pointing). This dataset focuses on very young children during short, adult-driven interactions. As such, it does not include episodes of naturally-occurring social interactions between peers, and the diversity of said interactions is limited. Besides, the lack of intrinsic and extrinsic camera calibration information in the dataset prevent the automatic extraction and labeling of key interaction features (like mutual gaze).</p>
<p>Another recent dataset, the <italic>Tower Game Dataset</italic> [<xref ref-type="bibr" rid="pone.0205999.ref010">10</xref>], focuses specifically on rich dyadic social interactions. The dataset comprises of 39 adults recorded over a total of 112 annotated sessions of 3 min in average. The participants are instructed to jointly construct a tower using wooden blocks. Interestingly, the participants are not allowed to talk to maximise the amount of non-verbal communication. The skeletons and faces of the participants are recorded, and the dataset is manually annotated with so-called <italic>Essential Social Interaction Predicates</italic> (ESIPs): rhythmic coupling (entrainment or attunement), mimicry (behavioral matching), movement simultaneity, kinematic turn taking patterns, joint attention. This dataset does not appear to be publicly available on-line.</p>
<p>The UE-HRI dataset [<xref ref-type="bibr" rid="pone.0205999.ref011">11</xref>] is another recently published (2017) dataset of social interactions, focusing solely on human-robot interactions. 54 adult participants were recorded (duration M = 7.7min) during spontaneous dialogues with a Pepper robot. The interactions took place in a public space, and include both one-to-one and multi-party interactions. The resulting dataset includes audio and video recordings from the robot perspective, as well as manual annotations of the levels of engagement. It is publicly available.</p>
<p>PInSoRo, our dataset, shares some of the aims of the <italic>Tower Game</italic> and <italic>UE-HRI</italic> datasets, with however significant differences. Contrary to these two datasets, our target population are children. We also put a strong focus on naturally occurring, real-world social behaviours. Furthermore, as presented in the following sections, we record much longer interactions (up to 40 minutes) of free play interactions, capturing a wider range of socio-cognitive behaviours. We did not place any constraints on the permissible communication modalities, and the recordings were manually annotated with a focus on social constructs.</p>
</sec>
</sec>
<sec id="sec005" sec-type="materials|methods">
<title>Material and methods</title>
<sec id="sec006">
<title>The free-play sandbox task</title>
<p>As previously introduced, the <italic>free-play sandbox</italic> task is based on face-to-face free-play interactions, mediated by a large, horizontal touchscreen. Pairs of children (or alternatively, one child and one robot) are invited to freely draw and interact with items displayed on an interactive table, without any explicit goals set by the experimenter (<xref ref-type="fig" rid="pone.0205999.g001">Fig 1</xref>). The task is designed so that children can engage in open-ended and non-directive play. Yet, it is sufficiently constrained to be suitable for recording, and allows the reproduction of social behaviour by an artificial agent in comparable conditions.</p>
<fig id="pone.0205999.g001" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0205999.g001</object-id>
<label>Fig 1</label>
<caption>
<title>The free-play social interactions sandbox: Two children or one child and one robot (as pictured here) interacted in a free-play situation, by drawing and manipulating items on a touchscreen.</title>
<p>Children were facing each other and sit on cushions. Each child wore a bright sports bib, either purple or yellow, to facilitate later identification.</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0205999.g001" xlink:type="simple"/>
</fig>
<p>Specifically, the free-play sandbox follows the <italic>sandtray</italic> paradigm [<xref ref-type="bibr" rid="pone.0205999.ref012">12</xref>]: a large touchscreen (60cm × 33cm, with multitouch support) is used as an interactive surface. The two players, facing each other, play together, moving interactive items or drawing on the surface if they wish so (<xref ref-type="fig" rid="pone.0205999.g002">Fig 2</xref>). The background image depicts a generic empty environment, with different symbolic colours (water, grass, beach, bushes…). By drawing on top of the background picture, the children can change the environment to their liking. The players do not have any particular task to complete, they are simply invited to freely play. They can play for as long as they wish. However, for practical reasons, we had to limit the sessions to a maximum of 40 minutes.</p>
<fig id="pone.0205999.g002" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0205999.g002</object-id>
<label>Fig 2</label>
<caption>
<title>Example of a possible game situation.</title>
<p>Game items (animals, characters…) can be dragged over the whole play area, while the background picture can be painted over by picking a colour. In this example, the top player is played by a robot.</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0205999.g002" xlink:type="simple"/>
</fig>
<p>Even though the children do typically move a little, the task is fundamentally a face-to-face, spatially delimited, interaction, and as such simplifies the data collection. In fact, the children’s faces were successfully detected in 98% of the over 2 million frames recorded during the PInSoRo dataset acquisition campaign.</p>
<sec id="sec007">
<title>Experimental conditions</title>
<p>The PInSoRo dataset aims to establish two experimental baselines for the free-play sandbox task: the ‘human social interactions’ baseline on one hand (child–child condition), an ‘asocial’ baseline on the other hand (child–<italic>non-social</italic> robot condition). These two baselines aim to characterise the qualitative and quantitative bounds of the spectrum of social interactions and dynamics that can be observed in this situation.</p>
<p>In the <italic>child-child</italic> condition, a diverse set of social interactions and social dynamics were expected to be observed, ranging from little social interactions (for instance, with shy children) to strong, positive interactions (for instance, good friends), to hostility (children who do not get along very well).</p>
<p>In the <italic>asocial</italic> condition, one child was replaced by an autonomous robot. The robot was purposefully programmed to be <italic>asocial</italic>. It autonomously played with the game items as a child would (although it did not perform any drawing action), but avoided all social interactions: no social gaze, no verbal interaction, no reaction to child-initiated game actions.</p>
<p>From the perspective of social psychology, this condition provides a baseline for the social interactions and dynamics at play (or the lack thereof) when the social communication channel is severed between the agents, while maintaining a similar social setting (face-to-face interaction; free-play activity).</p>
<p>From the perspective of human-robot interaction and artificial intelligence in general, the child–‘asocial robot’ condition provides a baseline to contrast with for yet-to-be-created richer social and behavioural AI policies.</p>
</sec>
<sec id="sec008">
<title>Hardware apparatus</title>
<p>The interactive table was based on a 27” Samsung All-In-One computer (quad core i7-3770T, 8GB RAM) running Ubuntu Linux and equipped with a fast 1TB SSD hard-drive. The computer was held horizontally in a custom aluminium frame standing 26cm above the floor. All the cameras were connected to the computer via USB-3. The computer performed all the data acquisition using ROS Kinetic (<ext-link ext-link-type="uri" xlink:href="http://www.ros.org/" xlink:type="simple">http://www.ros.org/</ext-link>). The same computer was also running the game interface on its touch-enabled screen (60cm × 33cm), making the whole system standalone and easy to deploy.</p>
<p>The children’s faces were recorded using two short range (0.2m to 1.2m) Intel RealSense SR300 RGB-D cameras placed at the corners of the touchscreen (<xref ref-type="fig" rid="pone.0205999.g001">Fig 1</xref>) and tilted to face the children. The cameras were rigidly mounted on custom 3D-printed brackets. This enabled a precise measurement of their 6D pose relative to the touchscreen (extrinsic calibration).</p>
<p>Audio was recorded from the same SR300 cameras (one mono audio stream was recorded for each child, from the camera facing him or her).</p>
<p>Finally, a third RGB camera (the RGB stream of a Microsoft Kinect One, the <italic>environment camera</italic> in <xref ref-type="fig" rid="pone.0205999.g001">Fig 1</xref>) recorded the whole interaction setting. This third video stream was intended to support human coders while annotating the interaction, and was not precisely calibrated.</p>
<p>In the child-robot condition, a Softbank Robotics’ Nao robot was used. The robot remained in standing position during the entire play interaction. The actual starting position of the robot with respect to the interactive table was recalibrated before each session by flashing a 2D fiducial marker on the touchscreen, from which the robot could compute its physical location.</p>
</sec>
<sec id="sec009">
<title>Software apparatus</title>
<p>The software-side of the free-play sandbox is entirely open-source (source code: <ext-link ext-link-type="uri" xlink:href="https://github.com/freeplay-sandbox/" xlink:type="simple">https://github.com/freeplay-sandbox/</ext-link>). It was implemented using two main frameworks: Qt QML (<ext-link ext-link-type="uri" xlink:href="http://doc.qt.io/qt-5/qtquick-index.html" xlink:type="simple">http://doc.qt.io/qt-5/qtquick-index.html</ext-link>) for the user interface (UI) of the game (<xref ref-type="fig" rid="pone.0205999.g002">Fig 2</xref>), and the <italic>Robot Operating System</italic> (ROS) for the modular implementation of the data processing and behaviour generation pipelines, as well as for the recordings of the various datastreams (<xref ref-type="fig" rid="pone.0205999.g004">Fig 4</xref>). The graphical interface interacts with the decisional pipeline over a bidirectional QML-ROS bridge that was developed for that purpose (source code available from the same link).</p>
<p><xref ref-type="fig" rid="pone.0205999.g003">Fig 3</xref> presents the complete software architecture of the sandbox as used in the child-robot condition (in the child-child condition, robot-related modules were simply not started).</p>
<fig id="pone.0205999.g003" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0205999.g003</object-id>
<label>Fig 3</label>
<caption>
<title>Software architecture of the free-play sandbox (data flows <italic>from</italic> orange dots <italic>to</italic> blue dots).</title>
<p>Left nodes interact with the interactive table hardware (game interface (1) and camera drivers (2)). The green nodes in the centre implement the behaviour of the robot (play policy (3) and robot behaviours (4)). Several helper nodes are available to provide for instance a segmentation of the children drawings into zones (5) or A* motion planning for the robot to move in-game items (6). Nodes are implemented in Python (except for the game interface, developed in QML) and inter-process communication relies on ROS. 6D poses are managed and exchanged via ROS TF.</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0205999.g003" xlink:type="simple"/>
</fig>
</sec>
<sec id="sec010">
<title>Robot control</title>
<p>As previously described, one child was replaced by a robot in the child-robot condition. Our software stack allowed for the robot to be used in two modes of operations: either autonomous (selecting actions based on pre-programmed play policies), or controlled by a human operator (so-called <italic>Wizard-of-Oz</italic> mode of operation).</p>
<p>For the purpose of the PInSoRo dataset, the robot behaviour was fully autonomous, yet coded to be purposefully <italic>asocial</italic> (no social gaze, no verbal interaction, no reaction to child-initiated game actions). The simple action policy that we implemented consisted in the robot choosing a random game item (in its reach), and moving that item to a predefined zone on the map (e.g. if the robot could reach the crocodile figure, it would attempt to drag it to a blue, i.e. water, zone). The robot did not physically drag the item on the touchscreen: it relied on a A* motion planner to find an adequate path, sent the resulting path to the touchscreen GUI to animate the displacement of the item, and moved its arm in a synchronized fashion using the inverse kinematics solver provided with the robot’s software development kit (SDK).</p>
<p>In the Wizard-of-Oz mode of operation, the experimenter would remotely control the robot through a tablet application developed for this purpose (Figs <xref ref-type="fig" rid="pone.0205999.g003">3</xref>–<xref ref-type="fig" rid="pone.0205999.g011">11</xref>). The tablet exactly mirrored the game state, and the experimenter dragged the game items on the tablet as would the child on the touchscreen. On release, the robot would again mimic the dragging motion on the touchscreen, moving an object to a new location. This mode of operation, while useful to conduct controlled studies, was not used for the dataset acquisition.</p>
<fig id="pone.0205999.g004" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0205999.g004</object-id>
<label>Fig 4</label>
<caption>
<title>The free-play sandbox, viewed at runtime within ROS RViz.</title>
<p>Simple computer vision was used to segment the background drawings into zones (visible on the right panel). The poses and bounding boxes of the interactive items were broadcast as well, and turned into an occupancy map, used to plan the robot’s arm motion. The individual pictured in this figure has given written informed consent (as outlined in PLOS consent form) to appear.</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0205999.g004" xlink:type="simple"/>
</fig>
<fig id="pone.0205999.g005" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0205999.g005</object-id>
<label>Fig 5</label>
<caption>
<title>The coding scheme used for annotating social interactions occurring during free-play episodes.</title>
<p>Three main axis were studied: task engagement, social engagement and social attitude.</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0205999.g005" xlink:type="simple"/>
</fig>
<fig id="pone.0205999.g006" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0205999.g006</object-id>
<label>Fig 6</label>
<caption>
<title>2D skeletons, including facial landmarks and hand details are automatically extracted using the OpenPose library [<xref ref-type="bibr" rid="pone.0205999.ref018">18</xref>].</title>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0205999.g006" xlink:type="simple"/>
</fig>
<fig id="pone.0205999.g007" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0205999.g007</object-id>
<label>Fig 7</label>
<caption>
<title>Screenshot of the dedicated tool developed for rapid annotation of the social interactions.</title>
<p>The annotators used a secondary screen (tablet) with buttons (layout similar to <xref ref-type="fig" rid="pone.0205999.g005">Fig 5</xref>) to record the social constructs. Figure edited for legibility (timeline enlarged) and to mask out one of the children’ face. The right individual pictured in this figure has given written informed consent (as outlined in PLOS consent form) to appear.</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0205999.g007" xlink:type="simple"/>
</fig>
<fig id="pone.0205999.g008" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0205999.g008</object-id>
<label>Fig 8</label>
<caption>
<title>Density distribution of the durations of the interactions for the two conditions.</title>
<p>Interactions in the child-robot condition were generally shorter than the child-child interactions. Interactions in the child-child condition followed a bi-modal distribution, with one mode centered around minute 15 (similar to the child-robot one) and one, much longer mode, at minute 37.</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0205999.g008" xlink:type="simple"/>
</fig>
<fig id="pone.0205999.g009" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0205999.g009</object-id>
<label>Fig 9</label>
<caption>
<title>Repartition of annotations over the dataset (in total duration of recordings annotated with a given construct).</title>
<p>The three classes of constructs (task engagement, social engagement, social attitude) and the two conditions (child-child and child-robot) are plotted separately.</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0205999.g009" xlink:type="simple"/>
</fig>
<fig id="pone.0205999.g010" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0205999.g010</object-id>
<label>Fig 10</label>
<caption>
<title>Mean time (and standard deviation) that each construct has been annotated in each recording.</title>
<p>The large standard deviations reflect the broad range of group dynamics captured in the dataset.</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0205999.g010" xlink:type="simple"/>
</fig>
<fig id="pone.0205999.g011" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0205999.g011</object-id>
<label>Fig 11</label>
<caption>
<title>Percentage of observations for each constructs with respect the children’s age.</title>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0205999.g011" xlink:type="simple"/>
</fig>
</sec>
<sec id="sec011">
<title>Experiment manager</title>
<p>We developed as well a dedicated web-based interface (usually accessed from a tablet) for the experimenter to manage the whole experiment and data acquisition procedure (Figs <xref ref-type="fig" rid="pone.0205999.g003">3</xref>–<xref ref-type="fig" rid="pone.0205999.g010">10</xref>). This interface ensured that all the required software modules were running; it allowed the experimenter to check the status of each of them and, if needed, to start/stop/restart any of them. It also helped managing the data collection campaign by providing a convenient interface to record the participants’ demographics, resetting the game interface after each session, and automatically enforcing the acquisition protocol (presented in <xref ref-type="table" rid="pone.0205999.t001">Table 1</xref>).</p>
<table-wrap id="pone.0205999.t001" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0205999.t001</object-id>
<label>Table 1</label>
<caption>
<title>Data acquisition protocol.</title>
</caption>
<alternatives>
<graphic id="pone.0205999.t001g" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0205999.t001" xlink:type="simple"/>
<table border="0" frame="box" rules="all">
<colgroup>
<col align="left" valign="middle"/>
</colgroup>
<tbody>
<tr>
<td align="left"><bold>Greetings</bold> <italic><bold>(about 5 min)</bold></italic><break/>
<list list-type="bullet">
<list-item><p>explain the purpose of the study: showing robots how children play</p></list-item>
<list-item><p>briefly present a Nao robot: the robot stands up, gives a short message (<italic>Today I’ll be watching you playing</italic> in the child-child condition;<italic>Today I’ll be playing with you</italic> in the child-robot condition), and sits down.</p></list-item>
<list-item><p>place children on cushions</p></list-item>
<list-item><p>complete demographics on the tablet</p></list-item>
<list-item><p>remind the children that they can withdraw at anytime</p></list-item></list></td>
</tr>
<tr>
<td align="left"><bold>Gaze tracking task</bold> <italic><bold>(40 sec)</bold></italic><break/>children are instructed to closely watch a small picture of a rocket that moves randomly on the screen. Recorded data is used to train a eye-tracker post-hoc.</td>
</tr>
<tr>
<td align="left"><bold>Tutorial</bold> <italic><bold>(1-2 min)</bold></italic><break/>explain how to interact with the game, ensure the children are confident with the manipulation/drawing.</td>
</tr>
<tr>
<td align="left"><bold>Free-play task (up to 40 min)</bold><break/>
<list list-type="bullet">
<list-item><p>initial prompt: <italic>“Just to remind you, you can use the animals or draw. Whatever you like. If you run out of ideas, there’s also an ideas box. For example, the first one is a zoo. You could draw a zoo or tell a story. When you get bored or don’t want to play anymore, just let me know.”</italic></p></list-item>
<list-item><p>let children play</p></list-item>
<list-item><p>once they wish to stop, stop recording</p></list-item></list></td>
</tr>
<tr>
<td align="left"><bold>Debriefing</bold> <italic><bold>(about 2 min)</bold></italic><break/>
<list list-type="bullet">
<list-item><p>answer possible questions from the children</p></list-item>
<list-item><p>give small reward (e.g. stickers) as a thank you</p></list-item></list></td>
</tr>
</tbody>
</table>
</alternatives>
</table-wrap>
</sec>
</sec>
<sec id="sec012">
<title>Coding of the social interactions</title>
<p>Our aim is to provide insights on the social dynamics, and as such we annotated the dataset using a combination of three coding schemes for social interactions that reuse and adapt established social scales. Our resulting coding scheme (<xref ref-type="fig" rid="pone.0205999.g005">Fig 5</xref>) looked specifically at three axis: the level of <italic>task engagement</italic> (that distinguishes between <italic>focused</italic>, <italic>task oriented</italic> behaviours, and <italic>disengaged</italic>—yet sometimes highly social – behaviours); the level of social engagement (reusing Parten’s stages of play, but at a fine temporal granularity); the social attitude (that encoded attitudes like <italic>supportive</italic>, <italic>aggressive</italic>, <italic>dominant</italic>, <italic>annoyed</italic>, etc).</p>
<sec id="sec013">
<title>Task engagement</title>
<p>The first axis of our coding scheme aimed at making a broad distinction between ‘on-task’ behaviours (even though the free-play sandbox did not explicitly require the children to perform a specific task, they were still engaged in an underlying task: to play with the game) and ‘off-task’ behaviours. We called ‘on-task’ behaviours <italic>goal oriented</italic>: they encompassed considered, planned actions (that might be social or not). <italic>Aimless</italic> behaviours (with respect to the task) encompassed opposite behaviours: being silly, chatting about unrelated matters, having a good laugh, etc. These <italic>Aimless</italic> behaviours were in fact often highly social, and played an important role in establishing trust and cooperation between the peers. In that sense, we considered them as as important as on-task behaviours.</p>
</sec>
<sec id="sec014">
<title>Social engagement: Parten’s stages of play at micro-level</title>
<p>In our scheme, we characterised <italic>Social engagement</italic> by building upon Parten’s stages of play [<xref ref-type="bibr" rid="pone.0205999.ref003">3</xref>]. These five stages of play are normally used to characterise rather long sequences (at least several minutes) of social interactions. In our coding scheme, we applied them at the level of each of the micro-sequences of the interactions: one child is drawing and the other is observing was labelled as <italic>solitary play</italic> for the former child, <italic>on-looker</italic> behaviour for the later; the two children discuss what to do next: this sequence was annotated as a <italic>cooperative</italic> behaviour; etc.</p>
<p>We chose this fine-grained coding of social engagement to enable proper analyses of the internal dynamics of a long sequence of social interaction.</p>
</sec>
<sec id="sec015">
<title>Social attitude</title>
<p>The constructs related to the social <italic>attitude</italic> of the children derived from the <italic>Social Communication Coding System</italic> (SCCS) proposed by Olswang et al. [<xref ref-type="bibr" rid="pone.0205999.ref013">13</xref>]. The SCCS consists in 6 mutually exclusive constructs characterising social communication (<italic>hostile</italic>; <italic>pro-social</italic>; <italic>assertive</italic>; <italic>passive</italic>; <italic>adult seeking</italic>; <italic>irrelevant</italic>) and were specifically created to characterise children’s communication in a classroom setting.</p>
<p>We transposed these constructs from the communication domain to the general behavioural domain, keeping the <italic>pro-social</italic>, <italic>hostile</italic> (whose scope we broadened in <italic>adversarial</italic>), <italic>assertive</italic> (i.e. dominant), and <italic>passive</italic> constructs. In our scheme, the <italic>adult seeking</italic> and <italic>irrelevant</italic> constructs belong to Task Engagement axis.</p>
<p>Finally, we added the construct <italic>Frustrated</italic> to describe children who are reluctant or refuse to engage in a specific phase of interaction because of a perceived lack of fairness or attention from their peer, or because they fail at achieving a particular task (like a drawing).</p>
</sec>
</sec>
<sec id="sec016">
<title>Protocol</title>
<p>We adhered to the acquisition protocol described in <xref ref-type="table" rid="pone.0205999.t001">Table 1</xref> with all participants. To ease later identification, each child was also given a different and brightly coloured sports bib to wear.</p>
<p>Importantly, during the <italic>Greetings</italic> stage, we showed the robot both moving and speaking (for instance, “Hello, I’m Nao. Today I’ll be playing with you. Exciting!” while waving at the children). This was of particular importance in the child-robot condition, as it set the children’s expectations in term of the capabilities of the robot: the robot could in principle speak, move, and even behave in a social way.</p>
<p>Also, the game interface of the free-play sandbox offered a tutorial mode, used to ensure the children know how to manipulate items on a touchscreen and draw. In our experience, this never was an issue for children.</p>
</sec>
<sec id="sec017">
<title>Data collection</title>
<p><xref ref-type="table" rid="pone.0205999.t002">Table 2</xref> lists the raw datastreams that were collected during the game. By relying on ROS for the data acquisition (and in particular the <monospace>rosbag</monospace> tool), we ensured all the datastreams were synchronised, timestamped, and, where appropriate, came with calibration information (for the cameras mainly). For the PInSoRo dataset, cameras were configured to stream in qHD resolution (960×540 pixels) in an attempt to balance high enough resolution with tractable file size. It resulted in bag files weighting ≈1GB per minute.</p>
<table-wrap id="pone.0205999.t002" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0205999.t002</object-id>
<label>Table 2</label>
<caption>
<title>List of raw datastreams available in the PInSoRo dataset.</title>
<p>Each datastream is timestamped with a synchronised clock to facilitate later analysis.</p>
</caption>
<alternatives>
<graphic id="pone.0205999.t002g" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0205999.t002" xlink:type="simple"/>
<table border="0" frame="box" rules="all">
<colgroup>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
</colgroup>
<thead>
<tr>
<th align="left">Domain</th>
<th align="left">Type</th>
<th align="left">Details</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" rowspan="3">child 1</td>
<td align="left">audio</td>
<td align="left">16kHz, mono, semi-directional</td>
</tr>
<tr>
<td align="left">face (RGB)</td>
<td align="left">qHD (960×540), 30Hz</td>
</tr>
<tr>
<td align="left">face (depth)</td>
<td align="left">VGA (640×480), 30Hz</td>
</tr>
<tr>
<td align="left" rowspan="3">child 2</td>
<td align="left">audio</td>
<td align="left">16kHz, mono, semi-directional</td>
</tr>
<tr>
<td align="left">face (RGB)</td>
<td align="left">qHD (960×540), 30Hz</td>
</tr>
<tr>
<td align="left">face (depth)</td>
<td align="left">VGA (640×480), 30Hz</td>
</tr>
<tr>
<td align="left">environment</td>
<td align="left">RGB</td>
<td align="left">qHD (960×540), 29.7Hz</td>
</tr>
<tr>
<td align="left" rowspan="3">game interactions</td>
<td align="left">background drawing (RGB)</td>
<td align="left">4Hz</td>
</tr>
<tr>
<td align="left">finger touches</td>
<td align="left">6 points multi-touch, 10Hz</td>
</tr>
<tr>
<td align="left">game items pose</td>
<td align="left">TF frames, 10Hz</td>
</tr>
<tr>
<td align="left" rowspan="2">other</td>
<td align="left" colspan="2">static transforms between touchscreen and facial cameras</td>
</tr>
<tr>
<td align="left" colspan="2">cameras calibration informations</td>
</tr>
</tbody>
</table>
</alternatives>
</table-wrap>
<p>Besides audio and video streams, user interactions with the game were monitored and recorded as well. The background drawings produced by the children were recorded. They were also segmented according to their colours, and the contours of resulting regions were extracted and recorded. The positions of all manipulable game items were recorded (as ROS TF frames), as well as every touch on the touchscreen.</p>
</sec>
<sec id="sec018">
<title>Data post-processing</title>
<p><xref ref-type="table" rid="pone.0205999.t003">Table 3</xref> summarises the post-processed datastreams that are made available alongside the raw datastreams.</p>
<table-wrap id="pone.0205999.t003" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0205999.t003</object-id>
<label>Table 3</label>
<caption>
<title>List of post-processed datastreams available in the PInSoRo dataset.</title>
<p>With the exception of social annotations, all the data was automatically computed from the raw datastreams at 30Hz.</p>
</caption>
<alternatives>
<graphic id="pone.0205999.t003g" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0205999.t003" xlink:type="simple"/>
<table border="0" frame="box" rules="all">
<colgroup>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
</colgroup>
<thead>
<tr>
<th align="left">Domain</th>
<th align="left">Type</th>
<th align="left">Details</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" rowspan="7">children</td>
<td align="left" rowspan="4">face</td>
<td align="left">70 facial landmarks (2D)</td>
</tr>
<tr>
<td align="left">17 facial action-units</td>
</tr>
<tr>
<td align="left">head pose estimation (TF frame)</td>
</tr>
<tr>
<td align="left">gaze estimation (TF frame)</td>
</tr>
<tr>
<td align="left" rowspan="2">skeleton</td>
<td align="left">18 points body pose (2D)</td>
</tr>
<tr>
<td align="left">20 points hand tracking (2D, only when visible)</td>
</tr>
<tr>
<td align="left">audio</td>
<td align="left">INTERSPEECH’s 16 low-level descriptors</td>
</tr>
<tr>
<td align="left">annotations</td>
<td align="left" colspan="2">timestamped annotations of social behaviours and remarkable events</td>
</tr>
</tbody>
</table>
</alternatives>
</table-wrap>
<sec id="sec019">
<title>Audio processing</title>
<p>Audio features were automatically extracted using the OpenSMILE toolkit [<xref ref-type="bibr" rid="pone.0205999.ref014">14</xref>]. We used a 33ms-wide time windows in order to match the cameras FPS. We extracted the INTERSPEECH 2009 Emotion Challenge standardised features [<xref ref-type="bibr" rid="pone.0205999.ref015">15</xref>]. These are a range of prosodic, spectral and voice quality features that are arguably the most common features we might want to use for emotion recognition [<xref ref-type="bibr" rid="pone.0205999.ref016">16</xref>]. For a full list, please see [<xref ref-type="bibr" rid="pone.0205999.ref015">15</xref>]. As no reliable speech recognition engine for children voice could be found [<xref ref-type="bibr" rid="pone.0205999.ref017">17</xref>], audio recordings were not automatically transcribed.</p>
</sec>
<sec id="sec020">
<title>Facial landmarks, action-units, skeletons, gaze</title>
<p>Offline post-processing was performed on the images obtained from the cameras. We relied on the CMU OpenPose library [<xref ref-type="bibr" rid="pone.0205999.ref018">18</xref>] to extract for each child the upper-body skeleton (18 points), 70 facial landmarks including the pupil position, as well as the hands’ skeleton (<xref ref-type="fig" rid="pone.0205999.g006">Fig 6</xref>).</p>
<p>This skeletal information was extracted from the RGB streams of each of the three cameras, for every frame. It is stored alongside the main data in an easy-to-parse JSON file.</p>
<p>For each frame, 17 action units, with accompanying confidence levels, were also extracted using the OpenFace library [<xref ref-type="bibr" rid="pone.0205999.ref019">19</xref>]. The action-units recognised by OpenFace and provided alongside the data are AU01, AU02, AU04, AU05, AU06, AU07, AU09, AU10, AU12, AU14, AU15, AU17, AU20, AU23, AU25, AU26, AU28 and AU45 (classification following <ext-link ext-link-type="uri" xlink:href="https://www.cs.cmu.edu/~face/facs.htm" xlink:type="simple">https://www.cs.cmu.edu/~face/facs.htm</ext-link>).</p>
<p>Gaze was also estimated, using two techniques. First, head pose estimation was performed following [<xref ref-type="bibr" rid="pone.0205999.ref020">20</xref>], and used to estimate gaze pose. While this technique is effective to segment pose at a coarse level (i.e. gaze on interactive table vs. gaze on other child/robot vs. gaze on experimenter), it offers limited accuracy when tracking the precise gaze location on the surface of the interactive table (due to not tracking the eye pupils).</p>
<p>We complemented head pose estimation with a neural network (a simple 7-layers, fully connected, multi-layer perceptron with ReLU activations and 64 units per layer), implemented with the Caffe framework (source available here: <ext-link ext-link-type="uri" xlink:href="https://github.com/severin-lemaignan/visual_tracking_caffe" xlink:type="simple">https://github.com/severin-lemaignan/visual_tracking_caffe</ext-link>).</p>
<p>The network trained from a ground truth mapping between the children’ faces and 2D gaze coordinates. Training data is obtained by asking the children to follow a target on the screen for a short period of time before starting the main free play activity (see protocol, <xref ref-type="table" rid="pone.0205999.t001">Table 1</xref>). The position of the target provides the ground truth (x, y) coordinates of the gaze on the screen. For each frame, the network is then fed a feature vector comprising 32 facial and skeletal (x, y) points of interest relevant to gaze estimation (namely, the 2D location of the pupils, eye contours, eyebrows, nose, neck, shoulders and ears). The training dataset comprises 80% of the fully randomized dataset (123711 frames) and the testing dataset the remaining 20% (30927 frames). Using this technique, we measured a gaze location error of 12.8% on our test data between the ground truth location of the target on the screen and the estimated gaze location (i.e. ±9cm over the 70cm-wide touchscreen). The same pre-trained network is then used to provide gaze estimation during the remainder of the free play activity.</p>
</sec>
<sec id="sec021">
<title>Video coding</title>
<p>The coding was performed post-hoc with the help of a dedicated annotation tool (<xref ref-type="fig" rid="pone.0205999.g007">Fig 7</xref>) which is part of the free-play sandbox toolbox. This tool can replay and randomly seek in the three video streams, synchronised with the recorded state of the game (including the drawings as they were created). An interactive timeline displaying the annotations is also displayed.</p>
<p>The annotation tool offers a remote interface for the annotator (made of large buttons, and visually similar to <xref ref-type="fig" rid="pone.0205999.g005">Fig 5</xref>) that is typically displayed on a tablet and allow the simultaneous coding of the behaviours of the two children. Usual video coding practices (double-coding of a portion of the dataset and calculation of an inter-judge agreement score) were followed.</p>
</sec>
</sec>
</sec>
<sec id="sec022">
<title>Results—The PInSoRo dataset</title>
<p>Using the free-play sandbox methodology, we have acquired a large dataset of social interactions between either pairs of children or one child and one robot. The data collection took place over a period of 3 months during Spring 2017.</p>
<p>In total, 120 children were recorded for a total duration of 45 hours and 48 minutes of data collection. These 120 children (see demographics in <xref ref-type="table" rid="pone.0205999.t004">Table 4</xref>; sample drawn from local schools) were randomly assigned to one of two conditions: the child-child condition (90 children, 45 pairs) and a child-robot condition (30 children). The sample sizes were balanced in favour of the child-child condition as the social dynamics that we ultimately want to capture are much richer in this condition.</p>
<table-wrap id="pone.0205999.t004" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0205999.t004</object-id>
<label>Table 4</label>
<caption>
<title>Descriptive statistics for the children.</title>
</caption>
<alternatives>
<graphic id="pone.0205999.t004g" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0205999.t004" xlink:type="simple"/>
<table border="0" frame="box" rules="all">
<colgroup>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
</colgroup>
<thead>
<tr>
<th align="left">Condition</th>
<th align="left">Age Mean</th>
<th align="left">Age SD</th>
<th align="left"># girls</th>
<th align="left"># boys</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left">Whole group</td>
<td align="char" char=".">6.4</td>
<td align="char" char=".">1.3</td>
<td align="left">55</td>
<td align="left">65</td>
</tr>
<tr>
<td align="left">Child-child</td>
<td align="char" char=".">6.3</td>
<td align="char" char=".">1.4</td>
<td align="left">42</td>
<td align="left">48</td>
</tr>
<tr>
<td align="left">Child-robot</td>
<td align="char" char=".">6.9</td>
<td align="char" char=".">0.9</td>
<td align="left">12</td>
<td align="left">18</td>
</tr>
</tbody>
</table>
</alternatives>
</table-wrap>
<p>In both conditions, and after a short tutorial, the children were simply invited to freely play with the sandbox, for as long as they wished (with a cap at 40 min; cf. protocol in <xref ref-type="table" rid="pone.0205999.t001">Table 1</xref>).</p>
<p>In the child-child condition, 45 free-play interactions (i.e. 90 children) were recorded with a mean duration M = 24.15 min (standard deviation SD = 11.25 min). In the child-robot condition, 30 children were recorded, M = 19.18 min (SD = 10 min).</p>
<p><xref ref-type="fig" rid="pone.0205999.g008">Fig 8</xref> presents the density distributions of the durations of the interactions for the two baselines. The distributions show that (1) the vast majority of children engaged easily and for non-trivial amounts of time with the task; (2) the task led to a wide range of levels of commitment, which is desirable: it supports the claim that the free-play sandbox is an effective paradigm to observe a range of different social behaviours; (3) many long interactions (&gt;30 min) were observed, which is especially desirable to study social dynamics.</p>
<p>The distribution of the child-robot interaction durations shows that these interactions are generally shorter. This was expected as the robot’s asocial behaviour was designed to be less engaging. Often, the child and the robot were found to be playing side-by-side—in some case for rather long periods of time—without interacting at all (solitary play).</p>
<p>Over the whole dataset, the children faces were detected on 98% of the images, which validates the positioning of the camera with respect to the children to record facial features.</p>
<sec id="sec023">
<title>Annotations</title>
<p>Five expert annotators performed the dataset annotation. Each annotator received one hour of training by the experimenters, and were compensated for their work.</p>
<p>In total, 13289 annotations of social dynamics were produced, resulting in an average of 149 annotations per record (SD = 136), which equates to an average of 4.2 annotations/min (SD = 2.1), and an average duration of annotated episodes of 48.8 sec (SD = 33.3). <xref ref-type="fig" rid="pone.0205999.g009">Fig 9</xref> shows the repartition of the annotation corpus over the different constructs presented in <xref ref-type="fig" rid="pone.0205999.g005">Fig 5</xref>. <xref ref-type="fig" rid="pone.0205999.g010">Fig 10</xref> shows the mean annotation time and standard deviation per recording for each construct.</p>
<p>Overall, 23% of the dataset was double-coded. Inter-coder agreement was found to be 51.8% (SD = 16.8) for task engagement annotations; 46.1% (SD = 24.2) for social engagement; 56.6% (SD = 22.9) for social attitude.</p>
<p>These values are relatively low (only partial agreement amongst coders). This was expected, as annotating social interactions beyond surface behaviours is indeed generally difficult. The observable, objective behaviours are typically the result of a superposition of the complex and non-observable underlying cognitive and emotional states. As such, these deeper socio-cognitive states can only be indirectly observed, and their labelling is typically error prone.</p>
<p>However, this is not anticipated to be a major issue for data-driven analyses, as machine learning algorithms are typically trained to estimate probability distributions. As such, divergences in human interpretations of a given social episode will simply be reflected in the probability distribution of the learnt model.</p>
<p>When looking at social behaviours with respect to age groups, expected behavioural trends are observed (<xref ref-type="fig" rid="pone.0205999.g011">Fig 11</xref>): <italic>adult seeking</italic> goes down when children get older; more <italic>cooperative</italic> play is observed with older children, while more <italic>parallel</italic> play takes place with younger ones. In constrast, the social attitudes appear evenly distributed amongst age groups.</p>
</sec>
<sec id="sec024">
<title>Dataset availability and data protection</title>
<p>All data has been collected by researchers at the University of Plymouth, under a protocol approved by the university ethics committee. The parents of the participants explicitly consented in writing to sharing of their child’s video and audio with the research community. The data does not contain any identifying information, except the participant’s images. The child’s age and gender are also available. The parents of the children in this manuscript have given written informed consent (as outlined in PLOS consent form) to publish these case details.</p>
<p>The dataset is freely available to any interested researcher. Due to ethical and data protection regulations, the dataset is however made available in two forms: a public, Creative Commons licensed, version that does not include any video material of the children (no video nor audio streams), and hosted on the Zenodo open-data platform: <ext-link ext-link-type="uri" xlink:href="https://zenodo.org/record/1043508" xlink:type="simple">https://zenodo.org/record/1043508</ext-link>. The complete version that includes all video streams is freely available as well, but interested researchers must first fill a data protection form. The detail of the procedure are available online: <ext-link ext-link-type="uri" xlink:href="https://freeplay-sandbox.github.io/application" xlink:type="simple">https://freeplay-sandbox.github.io/application</ext-link>.</p>
</sec>
</sec>
<sec id="sec025">
<title>Discussion of the free-play sandbox</title>
<p>The free-play sandbox elicits a loosely structured form of play: the actual play situations are not known beforehand and might change several times during the interaction; the game actions, even though based on one primary interaction modality (touches on the interactive table), are varied and unlimited (especially when considering the drawings); the social interactions between participants are multi-modal (speech, body postures, gestures, facial expressions, etc.) and unconstrained. This loose structure creates a fecund environment for children to express a range of complex, dynamics, natural social behaviours that are not tied to an overly constructed social situation. The diversity of the social behaviours that we have been able to capture can indeed been seen in Figs <xref ref-type="fig" rid="pone.0205999.g009">9</xref> and <xref ref-type="fig" rid="pone.0205999.g011">11</xref>.</p>
<p>Yet, the interaction is nonetheless structured. First, the physical bounds of the interactive table limit the play area to a well defined and relatively small area. As a consequence, children are mostly static (they are sitting in front of the table) and their primary form of physical interaction is based on 2D manipulations on a screen.</p>
<p>Second, the game items themselves (visible in <xref ref-type="fig" rid="pone.0205999.g002">Fig 2</xref>) structure the game scenarios. They are iconic characters (animals or children) with strong semantics associated to them (such as ‘crocodiles like water and eat children’). The game background, with its recognizable zones, also elicit a particular type of games (like building a zoo or pretending to explore the savannah).</p>
<p>These elements of structure (along with other, like the children demographics) arguably limit how general the PInSoRo dataset is. However, it also enable the free-play sandbox paradigm to retain key properties that makes it a practical and effective scientific tool: because the game builds on simple and universal play mechanics (drawings, pretend play with characters), the paradigm is essentially cross-cultural; because the sandbox is physically bounded and relatively small, it can be easily transported and practically deployed in a range of environments (schools, exhibitions, etc.); because the whole apparatus is well defined and relatively easy to duplicate (it essentially consists in one single touchscreen computer), the free-play sandbox facilitates the replication of studies while preserving ecological validity.</p>
<p>Compared to existing datasets of social interactions (the <italic>Multimodal Dyadic Behavior Dataset</italic>, the <italic>Tower Game</italic> dataset and the <italic>UE-HRI</italic> dataset), PInSoRo is much larger, with more than 45 hours of data, compared to 10.6, 5.6 and 6.9 hours respectively. PInSoRo is fully multi-modal whereas the <italic>Tower Game</italic> dataset does not include verbal interactions, and the <italic>UE-HRI</italic> dataset focuses instead of spoken interactions. Compared to the <italic>Multimodal Dyadic Behavior Dataset</italic>, PInSoRo captures a broader range of social situations, with fully calibrated datastreams, enabling a broad range of automated data processing and machine learning applications. Finally, PInSoRo is also unique for being the first (open) dataset capturing <italic>long sequences</italic> (up to 40 minutes) of <italic>ecologically valid</italic> social interactions amongst children or between children and robots.</p>
</sec>
<sec id="sec026">
<title>Conclusion—Towards the machine learning of social interactions?</title>
<p>We presented in this article the PInSoRo dataset, a large and open dataset of loosely constrained social interactions between children and robots. By relying on prolonged free-play episodes, we captured a rich set of naturally-occurring social interactions taking place between pairs of children or pairs of children and robots. We recorded an extensive set of calibrated and synchronised multimodal datastreams which can be used to mine and analyse the social behaviours of children. As such, this data provides a novel playground for the data-driven investigation and modelling of the social and developmental psychology of children.</p>
<p>The PInSoRo dataset also holds considerable promise for the automatic training of models of social behaviours, including implicit social dynamics (like rhythmic coupling, turn-taking), social attitudes, or engagement interpretation. As such, we foresee that the dataset might play an instrumental role in enabling artificial systems (and in particular, social robots) to recognise, interpret, and possibly, generate, socially congruent signals and behaviours whenever interacting with children. Whether such models can help uncover some of the implicit precursors of social behaviours, and is so, whether the same models, learnt from children data, can as well be used to interpret adult social behaviours, are open—and stimulating—questions that this dataset might contribute to answer.</p>
</sec>
</body>
<back>
<ack>
<p>The authors warmly thank the Plymouth’s BabyLab, Freshlings nursery, Mount Street Primary School and Salisbury Road Primary School for their help with data acquisition. We also want to gratefully acknowledge the annotation work done by Lisa, Scott, Zoe, Rebecca and Sally.</p>
<p>This work has been supported by the EU H2020 Marie Sklodowska-Curie Actions project DoRoThy (grant 657227) and the H2020 L2TOR project (grant 688014).</p>
</ack>
<ref-list>
<title>References</title>
<ref id="pone.0205999.ref001">
<label>1</label>
<mixed-citation publication-type="other" xlink:type="simple">Baxter P, Kennedy J, E S, Lemaignan S, Belpaeme T. From Characterising Three Years of HRI to Methodology and Reporting Recommendations. In: Proceedings of the 2016 ACM/IEEE Human-Robot Interaction Conference (alt.HRI); 2016.</mixed-citation>
</ref>
<ref id="pone.0205999.ref002">
<label>2</label>
<mixed-citation publication-type="book" xlink:type="simple">
<name name-style="western"><surname>Bruner</surname> <given-names>JS</given-names></name>, <name name-style="western"><surname>Jolly</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Sylva</surname> <given-names>K</given-names></name>, editors. <source>Play: Its role in development and evolution</source>. <publisher-name>Penguin</publisher-name>; <year>1976</year>.</mixed-citation>
</ref>
<ref id="pone.0205999.ref003">
<label>3</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Parten</surname> <given-names>MB</given-names></name>. <article-title>Social participation among pre-school children</article-title>. <source>The Journal of Abnormal and Social Psychology</source>. <year>1932</year>;<volume>27</volume>(<issue>3</issue>):<fpage>243</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1037/h0074524" xlink:type="simple">10.1037/h0074524</ext-link></comment></mixed-citation>
</ref>
<ref id="pone.0205999.ref004">
<label>4</label>
<mixed-citation publication-type="book" xlink:type="simple">
<name name-style="western"><surname>Nehaniv</surname> <given-names>CL</given-names></name>, <name name-style="western"><surname>Dautenhahn</surname> <given-names>K</given-names></name>. <source>Imitation and social learning in robots, humans and animals: behavioural, social and communicative dimensions</source>. <publisher-name>Cambridge University Press</publisher-name>; <year>2007</year>.</mixed-citation>
</ref>
<ref id="pone.0205999.ref005">
<label>5</label>
<mixed-citation publication-type="book" xlink:type="simple">
<name name-style="western"><surname>Mohammad</surname> <given-names>Y</given-names></name>, <name name-style="western"><surname>Nishida</surname> <given-names>T</given-names></name>. <chapter-title>Interaction Learning Through Imitation</chapter-title>. In: <source>Data Mining for Social Robotics</source>. <publisher-name>Springer</publisher-name>; <year>2015</year>. p. <fpage>255</fpage>–<lpage>273</lpage>.</mixed-citation>
</ref>
<ref id="pone.0205999.ref006">
<label>6</label>
<mixed-citation publication-type="other" xlink:type="simple">Nagai Y. Learning to comprehend deictic gestures in robots and human infants. In: Proc. of the 14th IEEE Int. Symp. on Robot and Human Interactive Communication. IEEE; 2005. p. 217–222.</mixed-citation>
</ref>
<ref id="pone.0205999.ref007">
<label>7</label>
<mixed-citation publication-type="other" xlink:type="simple">Calinon S, Billard A. Teaching a humanoid robot to recognize and reproduce social cues. In: Proc. of the 15th IEEE Int. Symp. on Robot and Human Interactive Communication. IEEE; 2006. p. 346–351.</mixed-citation>
</ref>
<ref id="pone.0205999.ref008">
<label>8</label>
<mixed-citation publication-type="other" xlink:type="simple">Liu P, Glas DF, Kanda T, Ishiguro H, Hagita N. How to Train Your Robot—Teaching service robots to reproduce human social behavior. In: Proc. of the 23rd IEEE Int. Symp. on Robot and Human Interactive Communication; 2014. p. 961–968.</mixed-citation>
</ref>
<ref id="pone.0205999.ref009">
<label>9</label>
<mixed-citation publication-type="other" xlink:type="simple">Rehg J, Abowd G, Rozga A, Romero M, Clements M, Sclaroff S, et al. Decoding children’s social behavior. In: Proceedings of the IEEE conference on computer vision and pattern recognition; 2013. p. 3414–3421.</mixed-citation>
</ref>
<ref id="pone.0205999.ref010">
<label>10</label>
<mixed-citation publication-type="other" xlink:type="simple">Salter DA, Tamrakar A, Siddiquie B, Amer MR, Divakaran A, Lande B, et al. The tower game dataset: A multimodal dataset for analyzing social interaction predicates. In: Affective Computing and Intelligent Interaction (ACII), 2015 International Conference on. IEEE; 2015. p. 656–662.</mixed-citation>
</ref>
<ref id="pone.0205999.ref011">
<label>11</label>
<mixed-citation publication-type="other" xlink:type="simple">Ben-Youssef A, Clavel C, Essid S, Bilac M, Chamoux M, Lim A. UE-HRI: a new dataset for the study of user engagement in spontaneous human-robot interactions. In: Proceedings of the 19th ACM International Conference on Multimodal Interaction. ACM; 2017. p. 464–472.</mixed-citation>
</ref>
<ref id="pone.0205999.ref012">
<label>12</label>
<mixed-citation publication-type="other" xlink:type="simple">Baxter P, Wood R, Belpaeme T. A touchscreen-based ‘Sandtray’to facilitate, mediate and contextualise human-robot social interaction. In: Human-Robot Interaction (HRI), 2012 7th ACM/IEEE International Conference on. IEEE; 2012. p. 105–106.</mixed-citation>
</ref>
<ref id="pone.0205999.ref013">
<label>13</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Olswang</surname> <given-names>L</given-names></name>, <name name-style="western"><surname>Svensson</surname> <given-names>L</given-names></name>, <name name-style="western"><surname>Coggins</surname> <given-names>T</given-names></name>, <name name-style="western"><surname>Beilinson</surname> <given-names>J</given-names></name>, <name name-style="western"><surname>Donaldson</surname> <given-names>A</given-names></name>. <article-title>Reliability issues and solutions for coding social communication performance in classroom settings</article-title>. <source>Journal of Speech, Language &amp; Hearing Research</source>. <year>2006</year>;<volume>49</volume>(<issue>5</issue>):<fpage>1058</fpage> – <lpage>1071</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1044/1092-4388(2006/075)" xlink:type="simple">10.1044/1092-4388(2006/075)</ext-link></comment></mixed-citation>
</ref>
<ref id="pone.0205999.ref014">
<label>14</label>
<mixed-citation publication-type="other" xlink:type="simple">Eyben F, Weninger F, Gross F, Schuller B. Recent developments in openSMILE, the munich open-source multimedia feature extractor. In: Proceedings of the 21st ACM international conference on Multimedia. May; 2013. p. 835–838. Available from: <ext-link ext-link-type="uri" xlink:href="http://dl.acm.org/citation.cfm?doid=2502081.2502224" xlink:type="simple">http://dl.acm.org/citation.cfm?doid=2502081.2502224</ext-link>.</mixed-citation>
</ref>
<ref id="pone.0205999.ref015">
<label>15</label>
<mixed-citation publication-type="other" xlink:type="simple">Schuller B, Steidl S, Batliner A. The INTERSPEECH 2009 Emotion Challenge. In: Tenth Annual Conference of the International Speech Communication Association; 2009.</mixed-citation>
</ref>
<ref id="pone.0205999.ref016">
<label>16</label>
<mixed-citation publication-type="other" xlink:type="simple">Schuller B, Batliner A, Seppi D, Steidl S, Vogt T, Wagner J, et al. The relevance of feature type for the automatic classification of emotional user states: Low level descriptors and functionals. Proceedings of the Annual Conference of the International Speech Communication Association, INTERSPEECH. 2007;2(101):881–884.</mixed-citation>
</ref>
<ref id="pone.0205999.ref017">
<label>17</label>
<mixed-citation publication-type="other" xlink:type="simple">Kennedy J, Lemaignan S, Montassier C, Lavalade P, Irfan B, Papadopoulos F, et al. Child speech recognition in human-robot interaction: evaluations and recommendations. In: Proceedings of the 2017 ACM/IEEE International Conference on Human-Robot Interaction. ACM; 2017. p. 82–90.</mixed-citation>
</ref>
<ref id="pone.0205999.ref018">
<label>18</label>
<mixed-citation publication-type="book" xlink:type="simple">
<name name-style="western"><surname>Cao</surname> <given-names>Z</given-names></name>, <name name-style="western"><surname>Simon</surname> <given-names>T</given-names></name>, <name name-style="western"><surname>Wei</surname> <given-names>SE</given-names></name>, <name name-style="western"><surname>Sheikh</surname> <given-names>Y</given-names></name>. <source>Realtime Multi-Person 2D Pose Estimation using Part Affinity Fields</source>. In: <publisher-name>CVPR</publisher-name>; <year>2017</year>.</mixed-citation>
</ref>
<ref id="pone.0205999.ref019">
<label>19</label>
<mixed-citation publication-type="other" xlink:type="simple">Baltrušaitis T, Mahmoud M, Robinson P. Cross-dataset learning and person-specific normalisation for automatic action unit detection. In: Automatic Face and Gesture Recognition (FG), 2015 11th IEEE International Conference and Workshops on. vol. 6. IEEE; 2015. p. 1–6.</mixed-citation>
</ref>
<ref id="pone.0205999.ref020">
<label>20</label>
<mixed-citation publication-type="other" xlink:type="simple">Lemaignan S, Garcia F, Jacq A, Dillenbourg P. From Real-time Attention Assessment to “With-me-ness” in Human-Robot Interaction. In: Proceedings of the 2016 ACM/IEEE Human-Robot Interaction Conference; 2016.</mixed-citation>
</ref>
</ref-list>
</back>
</article>