<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Serious Games</journal-id><journal-id journal-id-type="publisher-id">games</journal-id><journal-id journal-id-type="index">15</journal-id><journal-title>JMIR Serious Games</journal-title><abbrev-journal-title>JMIR Serious Games</abbrev-journal-title><issn pub-type="epub">2291-9279</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v14i1e86054</article-id><article-id pub-id-type="doi">10.2196/86054</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>A Serious Game for Soft Skills Assessment in Human Resources: Cross-Sectional Within-Participant Convergent Validity Study</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes" equal-contrib="yes"><name name-style="western"><surname>Boutrouille</surname><given-names>Maxime</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Fichet</surname><given-names>L&#x00E9;o</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Dinet</surname><given-names>J&#x00E9;r&#x00F4;me</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib></contrib-group><aff id="aff1"><institution>Yuzu</institution><addr-line>2 Rue Maurice Barr&#x00E8;s</addr-line><addr-line>Metz</addr-line><country>France</country></aff><aff id="aff2"><institution>Laboratoire de Psychologie et Neurosciences et Chaire industrielle "BEHAVIOUR", Campus Lettres et Sciences Humaines, Universit&#x00E9; de Lorraine</institution><addr-line>Nancy</addr-line><country>France</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Brini</surname><given-names>Stefano</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Dipace</surname><given-names>Anna</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Loh</surname><given-names>Christian S</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Maxime Boutrouille, PhD, Yuzu, 2 Rue Maurice Barr&#x00E8;s, Metz, 57000, France, 33 637401015; <email>mboutrouille@yuzu.hr</email></corresp><fn fn-type="equal" id="equal-contrib1"><label>*</label><p>all authors contributed equally</p></fn></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>19</day><month>8</month><year>2026</year></pub-date><volume>14</volume><elocation-id>e86054</elocation-id><history><date date-type="received"><day>17</day><month>10</month><year>2025</year></date><date date-type="rev-recd"><day>30</day><month>03</month><year>2026</year></date><date date-type="accepted"><day>01</day><month>04</month><year>2026</year></date></history><copyright-statement>&#x00A9; Maxime Boutrouille, L&#x00E9;o Fichet, J&#x00E9;r&#x00F4;me Dinet. Originally published in JMIR Serious Games (<ext-link ext-link-type="uri" xlink:href="https://games.jmir.org">https://games.jmir.org</ext-link>), 19.8.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Serious Games, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://games.jmir.org">https://games.jmir.org</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://games.jmir.org/2026/1/e86054"/><abstract><sec><title>Background</title><p>Soft skills are increasingly assessed in human resources, but commonly used methods (eg, interviews and self-report questionnaires) have well-known limitations. Serious games have been proposed as a complementary assessment format because they can standardize administration, embed assessment in interactive scenarios, and capture behavioral traces. However, evidence for their psychometric validity remains limited and heterogeneous. Establishing convergent validity against well-established reference instruments is a key step in supporting their use as assessment tools.</p></sec><sec><title>Objective</title><p>This study aimed to evaluate the convergent validity of Yuzu, a serious game that assesses (1) active listening via a gamified questionnaire inspired by the Active-Empathic Listening Scale (AELS), (2) decision-making under uncertainty via a gamified adaptation of the Iowa Gambling Task (IGT), and (3) teamwork style via dialogue choices inspired by the SYMLOG (System for the Multiple Level Observation of Groups) model.</p></sec><sec sec-type="methods"><title>Methods</title><p>We conducted a cross-sectional, within-participant convergent validity study in France with 39 adults (n=23 women; mean age 27.79, SD 7.96 y). Participants completed a single laboratory session on a desktop PC with headphones (Yuzu build v2, developed by Yuzu). Participants completed the 3 Yuzu modules and the corresponding reference instruments, administered separately (the AELS, an online IGT via PsyToolkit, and a simplified SYMLOG questionnaire). Primary outcomes were the associations between Yuzu and reference scores for each construct (active listening total score, IGT exploitation-phase net score, and SYMLOG dimension scores). Convergent validity was examined using Spearman correlations (2-sided &#x03B1;=.05). Agreement was additionally examined using Bland-Altman analyses for active listening and equivalence testing using two one-sided tests (TOST) for IGT net scores.</p></sec><sec sec-type="results"><title>Results</title><p>At &#x03B1;=.05, active listening showed strong convergence between Yuzu and the AELS total score (Spearman &#x03C1;=0.890, 95% CI 0.804-0.948; <italic>P</italic>&#x003C;.001), with minimal systematic bias (mean difference of 0.024, 95% CI &#x2212;0.031 to 0.078). Decision-making scores were statistically equivalent across modalities based on TOST. The mean net score difference was 1.35 (90% CI &#x2212;1.22 to 3.93), within the equivalence bounds [&#x2212;5,+5] (TOST lower: <italic>P</italic>=.001; TOST upper: <italic>P</italic>=.01). Teamwork dialogue scores did not converge with the SYMLOG dimensions (dominance: &#x03C1;=&#x2212;0.070, 95% CI &#x2212;0.420 to 0.280; <italic>P</italic>=.69; positivity: &#x03C1;=0.082, 95% CI &#x2212;0.266 to 0.418; <italic>P</italic>=.64; task orientation: &#x03C1;=0.134, 95% CI &#x2212;0.260 to 0.494; <italic>P</italic>=.44), consistent with a ceiling effect toward cooperative choices.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>This study provides convergent validity evidence for 2 complementary assessment modalities embedded in a single serious game, showing that a gamified AELS-inspired module and a gamified IGT adaptation can closely match established reference measures while supporting standardized administration. In contrast, the dialogue-choice teamwork module showed limited sensitivity and no convergence, suggesting that interpersonal profiling in serious games may require more discriminating scenario design and stronger controls for social desirability. Unlike many previous studies that evaluated a single game component, this study provides a module-by-module convergent validity blueprint within a single platform by using matched reference instruments, thereby informing both research and human resources deployment.</p></sec></abstract><kwd-group><kwd>soft skills</kwd><kwd>assessment</kwd><kwd>serious game</kwd><kwd>video game</kwd><kwd>validity</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Game-based assessments and serious games are increasingly used in human resources to evaluate job-relevant competencies, partly because they are expected to improve applicant reactions and limit response distortions while preserving measurement quality [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref4">4</xref>]. However, recent evidence syntheses converge on a more cautious conclusion: empirical support remains uneven, construct validity findings are often inconsistent, and any advantages over conventional methods appear smaller and more contingent on design choices than is sometimes implied in applied discourse [<xref ref-type="bibr" rid="ref4">4</xref>]. This lack of robust and generalizable validity evidence is all the more concerning because these tools are frequently positioned as solutions for assessing soft skills.</p><p>Soft skills are widely treated as important determinants of employability and job performance in contemporary labor markets [<xref ref-type="bibr" rid="ref5">5</xref>]. Technological change has shifted the comparative advantage of human work away from routine execution and toward capacities such as judgment, flexibility, and creativity [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref6">6</xref>]. In project-based and technology-driven settings, individuals must interpret dynamic social contexts, coordinate with others, and adapt their behavior under constraints, making these transversal competencies practically consequential for organizations [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref8">8</xref>]. Although there is no universal consensus on what should be labeled as &#x201C;soft skills&#x201D; [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>], the diversity of terms used in the literature&#x2014;including nontechnical skills [<xref ref-type="bibr" rid="ref11">11</xref>], transversal skills [<xref ref-type="bibr" rid="ref12">12</xref>], adaptive skills [<xref ref-type="bibr" rid="ref13">13</xref>], and sociocognitive skills [<xref ref-type="bibr" rid="ref13">13</xref>]&#x2014;further illustrates the construct&#x2019;s heterogeneity. Therefore, a working definition is needed to guide its operationalization and validation. In this study, we adopt the definition proposed by Haselberger et al [<xref ref-type="bibr" rid="ref14">14</xref>], which is widely cited in this literature [<xref ref-type="bibr" rid="ref15">15</xref>-<xref ref-type="bibr" rid="ref17">17</xref>]: &#x201C;Soft Skills represent a dynamic combination of cognitive and meta-cognitive skills, interpersonal, intellectual and practical skills.&#x201D; From an assessment standpoint, this definition implies that soft skills should ideally be captured through indicators that are meaningful in context rather than only through what individuals report about themselves.</p><p>Several methods are routinely used to assess soft skills in applied settings, but each raises methodological limitations. Interviews and rater-based judgments can be informative; however, they remain sensitive to evaluator variability, bias, and indirect discrimination, and they do not always provide a standardized basis for comparison across candidates [<xref ref-type="bibr" rid="ref18">18</xref>-<xref ref-type="bibr" rid="ref21">21</xref>].</p><p>Self-report questionnaires are commonly used because they are practical and interpretable, and they can capture introspective information [<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref23">23</xref>]. However, self-report measures are vulnerable to systematic response biases and method effects, including anchoring, primacy and recency effects, time pressure, and consistency motivation [<xref ref-type="bibr" rid="ref22">22</xref>]. In selection contexts, socially desirable responding, acquiescent responding, and extreme responding are additional threats to validity [<xref ref-type="bibr" rid="ref24">24</xref>]. As a result, questionnaires may capture self-perceptions or self-presentational strategies rather than performance in work-relevant situations.</p><p>Cognitive tests can also be used as indicators of abilities related to adaptive performance, for example, when cognitive flexibility supports adaptation or decision-making as a component of problem-solving [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref26">26</xref>]. For instance, progressive matrices proposed by Raven have long been used as a benchmark for reasoning [<xref ref-type="bibr" rid="ref25">25</xref>]. However, applicant reactions to cognitive testing can be negative, including perceptions that tests are poorly related to the job. Cognitive testing also raises fairness concerns because it can yield substantial group differences and may induce test anxiety or stereotype threat, potentially undermining both performance and acceptability [<xref ref-type="bibr" rid="ref27">27</xref>-<xref ref-type="bibr" rid="ref31">31</xref>].</p><p>In response to these limitations, serious games have been proposed as a complementary assessment format because they can place candidates in interactive situations and enable observation of in situ behavior [<xref ref-type="bibr" rid="ref1">1</xref>]. They can offer more immersive experiences [<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref33">33</xref>], enhance perceived attractiveness [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref3">3</xref>], and provide access to behavioral traces, such as response sequences, decision trajectories, and reaction times [<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref35">35</xref>]. Some experimental work further suggests that game-based contexts may mitigate certain social biases, for example, by reducing stereotype threat effects on performance [<xref ref-type="bibr" rid="ref36">36</xref>].</p><p>However, the current evidence base indicates that these potential benefits do not translate uniformly into stronger measurement. Validity remains insufficiently documented for many tools [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref28">28</xref>]. More broadly, the validity of game-based assessments appears heterogeneous and dependent on the targeted construct and design features, which limits generalization across tools and contexts [<xref ref-type="bibr" rid="ref4">4</xref>]. Additional challenges include face validity concerns when tasks appear disconnected from the claimed competency, construct contamination when multiple attributes are blended within a single activity, and reduced trust when scoring relies on opaque feature extraction or machine learning [<xref ref-type="bibr" rid="ref28">28</xref>].</p><p>A practical implication is that validity evidence should be established module by module, with explicit reporting of how each construct is operationalized within gameplay and how closely the resulting scores align with well-established reference measures [<xref ref-type="bibr" rid="ref4">4</xref>]. To address this need, we developed Yuzu, a serious game designed to assess soft skills through a multimodal framework. Building on prior feasibility work showing that classical psychometric paradigms can be adapted within Yuzu [<xref ref-type="bibr" rid="ref37">37</xref>], the present study evaluates convergent validity across 3 modules representing distinct measurement modalities: (1) a gamified questionnaire module targeting active-empathic listening, inspired by the Active-Empathic Listening Scale (AELS) [<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref39">39</xref>]; (2) a gamified decision-making task targeting decision-making under uncertainty through an adaptation of the Iowa Gambling Task (IGT) [<xref ref-type="bibr" rid="ref40">40</xref>]; and (3) a dialogue-choice module targeting teamwork style based on constructs from the SYMLOG (System for the Multiple Level Observation of Groups) framework [<xref ref-type="bibr" rid="ref41">41</xref>] and compared with scores from a simplified SYMLOG questionnaire [<xref ref-type="bibr" rid="ref42">42</xref>].</p><p>We used a cross-sectional, within-participant convergent validity design in which the same participants completed each Yuzu module and its corresponding reference measure administered separately. We formulated one primary hypothesis regarding convergent validity across modules. Specifically, we hypothesized that (H1a) scores from Yuzu&#x2019;s active-empathic listening module would be positively correlated with the AELS total score, that (H1b) performance in Yuzu&#x2019;s decision-making module would be equivalent to performance observed in the reference IGT version, and that (H1c) the dimensions derived from Yuzu&#x2019;s teamwork dialogue module would be positively correlated with the corresponding dimensions measured by the SYMLOG questionnaire.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Inclusion and Exclusion</title><p>A total of 39 participants completed the full laboratory session and were included in the analyses. Participants were eligible if they were aged 18 years or older and had sufficient French proficiency to understand the study instructions and complete all study materials. The study had no a priori exclusion criteria beyond these requirements. For each module-level analysis, participants were excluded only if paired data were incomplete for that module (ie, if either the Yuzu module score or the corresponding reference measure score was missing).</p></sec><sec id="s2-2"><title>Participant Characteristics</title><p>The sample comprised 39 participants, including 23 (59.0%) women and 16 (41.0%) men. The mean age was 27.79 (SD 7.96; range 20-54) years. Participants&#x2019; occupational status was as follows: students (n=16, 41.0%), executives or managers (n=14, 35.9%), employees (n=6, 15.4%), retired participants (n=1, 2.6%), entrepreneur (n=1, 2.6%), and job seeker (n=1, 2.6%). Self-reported computer literacy was high on average (mean 7.91, SD 2.11), and familiarity with video games was moderate-to-high (mean 6.67, SD 2.69; <xref ref-type="table" rid="table1">Table 1</xref>).</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Demographic characteristics of participants in the convergent validity study (N=39).</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Variable</td><td align="left" valign="bottom">Values</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="2">Gender, n (%)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Women</td><td align="left" valign="top">23 (59.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Men</td><td align="left" valign="top">16 (41.0)</td></tr><tr><td align="left" valign="top">Age (y), mean (SD), range</td><td align="left" valign="top">27.79 (7.96), 20&#x2010;54</td></tr><tr><td align="left" valign="top" colspan="2">Education and occupational status, n (%)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Students</td><td align="left" valign="top">16 (41.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Executives or managers</td><td align="left" valign="top">14 (35.9)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Employees</td><td align="left" valign="top">6 (15.4)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Retired participants</td><td align="left" valign="top">1 (2.6)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Entrepreneur</td><td align="left" valign="top">1 (2.6)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Job seeker</td><td align="left" valign="top">1 (2.6)</td></tr><tr><td align="left" valign="top">Computer literacy, mean (SD)</td><td align="left" valign="top">7.91 (2.11)</td></tr><tr><td align="left" valign="top">Video game familiarity, mean (SD)</td><td align="left" valign="top">6.67 (2.69)</td></tr></tbody></table></table-wrap></sec><sec id="s2-3"><title>Sampling Procedures</title><sec id="s2-3-1"><title>Recruitment Setting, Location, and Dates</title><p>Participants were recruited in France between November and December 2024 through a study advertisement posted on LinkedIn. Interested individuals contacted the research team and were scheduled for an in-person laboratory session. Data collection took place in controlled laboratory settings at the Centre for Research in Psychology: Cognition, Psychism, and Organizations (CRP-CPO, University of Picardie Jules Verne) and the Lorraine Laboratory of Psychology and Neuroscience of Behavioral Dynamics (2LPN, University of Lorraine). All sessions were conducted in quiet rooms without distractions to ensure standardized testing conditions.</p></sec><sec id="s2-3-2"><title>Sampling Method and Self-Selection</title><p>The study used convenience sampling with self-selection. A public recruitment post was published on LinkedIn, inviting adults to participate in a laboratory-based study. Interested individuals self-referred by contacting the research team and were screened for eligibility based on the inclusion and exclusion criteria. Eligible individuals were then scheduled for an in-person laboratory session.</p></sec></sec><sec id="s2-4"><title>Sample Size, Power, and Precision</title><sec id="s2-4-1"><title>Intended and Achieved Sample Size</title><p>The intended sample size was approximately 40 participants, primarily determined by feasibility constraints for in-laboratory testing and by the objective of detecting at least moderate convergent associations in an early-stage validation study. The achieved sample size was 39 participants for the full session. For the teamwork module, the analytic sample comprised 35 participants because game data for that module were not available for 4 participants.</p></sec><sec id="s2-4-2"><title>Power and Precision Rationale</title><p>With a sample size of 39 participants, the study has approximately 80% power (2-sided &#x03B1;=.05) to detect correlations of <italic>r</italic> or &#x03C1;=0.44 or greater. For the teamwork module (n=35), the corresponding threshold was approximately 0.46. Therefore, the study was not designed to reliably detect small associations. No interim analyses or stopping rules were prespecified.</p></sec></sec><sec id="s2-5"><title>Measures and Covariates</title><sec id="s2-5-1"><title>Primary Measures</title><p>This convergent validity study compared 3 assessment modalities embedded in the serious game Yuzu with corresponding reference instruments administered outside the game. Active-empathic listening was assessed in Yuzu using a questionnaire module based on the AELS [<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref39">39</xref>]. The reference measure was the original AELS questionnaire [<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref39">39</xref>]. Decision-making under uncertainty was assessed in Yuzu using an adaptation of the IGT [<xref ref-type="bibr" rid="ref40">40</xref>]. The reference measure was the original IGT administered online [<xref ref-type="bibr" rid="ref43">43</xref>]. Teamwork-related interpersonal behavior was assessed in Yuzu through structured dialogue choices, operationalized using the SYMLOG framework [<xref ref-type="bibr" rid="ref41">41</xref>]. The reference measure was the simplified SYMLOG questionnaire by Blumberg [<xref ref-type="bibr" rid="ref42">42</xref>].</p></sec><sec id="s2-5-2"><title>Secondary Measures and Covariates</title><p>To evaluate whether performance could be biased by participants&#x2019; relationships with digital technology, measures of computer literacy and technophilia were collected. In addition, participants completed a game experience questionnaire based on the playability concept [<xref ref-type="bibr" rid="ref44">44</xref>] and a demographic questionnaire during the debriefing phase. No other covariates were included in the primary inferential analyses.</p></sec></sec><sec id="s2-6"><title>Data Collection</title><p>All participants completed a single laboratory session lasting approximately 2 hours, including a 10-minute break. Tasks were completed on a desktop PC with a curved monitor and over-ear headphones, using Yuzu build v2 developed by Yuzu (<xref ref-type="fig" rid="figure1">Figure 1</xref>).</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Experimental setup used during the laboratory session.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="games_v14i1e86054_fig01.png"/></fig><p>To control for potential order effects, the administration was counterbalanced: half of the participants completed the Yuzu tasks first and then the reference tests, whereas the other half completed the reference tests first and then Yuzu. The session included (1) a welcome and briefing with an explanation of the objectives and the session outline; (2) completion of the Yuzu tasks and reference tests; and (3) a debriefing interview, a game experience questionnaire, and a demographic questionnaire.</p></sec><sec id="s2-7"><title>Quality of Measurements</title><p>Several procedures were used to enhance measurement quality and standardization. Data were collected in controlled laboratory environments using standardized hardware and with minimal distractions. Instructions and session timing were standardized across study sites. Outcome scoring for the Yuzu modules was automated through embedded trackers to minimize rater-dependent variability. Order effects were addressed through counterbalanced administration of Yuzu and the reference measures.</p></sec><sec id="s2-8"><title>Instrumentation</title><sec id="s2-8-1"><title>Serious Game Platform</title><p>Yuzu is a single-player serious game for soft skills assessment that integrates multiple modalities within a single narrative 3D environment. Players navigate a space station, interact with nonplayable characters, complete tasks, and make decisions under constraints (<xref ref-type="fig" rid="figure2">Figure 2</xref>).</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>In-game screenshots from Yuzu illustrating the 3D environment used in the assessment tasks.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="games_v14i1e86054_fig02.png"/></fig><p>All actions are recorded through embedded trackers, producing behavioral indicators such as response patterns and choice sequences. Yuzu was designed by a multidisciplinary team (work psychology and ergonomics, engineering, and user interface design) in collaboration with the 2LPN research laboratory (University of Lorraine, France).</p></sec><sec id="s2-8-2"><title>Module 1: Active-Empathic Listening (Questionnaire Modality)</title><p>Active listening is commonly assessed using self-report scales because it reflects cognitive and emotional processes that can be difficult to capture through performance tasks. The AELS is a widely used and validated tool for measuring listening quality in interpersonal communication [<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref39">39</xref>]. It distinguishes 3 components: sensing, processing, and responding. In Yuzu, these components served as the conceptual framework for an original in-game questionnaire module. Items were specifically formulated for the serious game, and responses were collected using a 5-point Likert-type scale. The module yields an overall active listening score and 3 subscores corresponding to sensing, processing, and responding. These were compared with the reference AELS questionnaire [<xref ref-type="bibr" rid="ref38">38</xref>]. The interface of the active listening module is shown in <xref ref-type="fig" rid="figure3">Figure 3</xref>.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Screenshot of the Yuzu active listening module, inspired by the Active-Empathic Listening Scale.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="games_v14i1e86054_fig03.png"/></fig><p>Within the narrative, this module appears after the player awakens from a prolonged sleep: the station&#x2019;s physician administers a short questionnaire to verify the participant&#x2019;s condition and adaptation to the situation.</p></sec><sec id="s2-8-3"><title>Module 2: Decision-Making Under Uncertainty (Task Modality)</title><p>We selected the IGT [<xref ref-type="bibr" rid="ref40">40</xref>] as the reference paradigm for decision-making under uncertainty. In the classical IGT, participants complete 100 trials and choose among 4 decks (A, B, C, and D). Decks A and B are disadvantageous over the long term (high immediate gains but long-term losses), whereas decks C and D are advantageous (lower immediate gains but long-term benefits). Expected learning is reflected in an increasing preference for advantageous decks over the course of the task.</p><p>In Yuzu, the core principles of the IGT were retained and transposed into a serious game format in which participants collected meteorite samples using drones across 4 excavation sites. The 100 trials were divided into 10 blocks of 10 choices each. Consistent with prior work [<xref ref-type="bibr" rid="ref45">45</xref>], the first 40 trials (blocks 1-4) were treated as an exploration phase and the remaining 60 trials (blocks 5-10) as an exploitation phase. The interface of the game-based IGT adaptation is shown in <xref ref-type="fig" rid="figure4">Figure 4</xref>.</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Screenshot of the Yuzu decision-making module (an adaptation of the Iowa Gambling Task).</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="games_v14i1e86054_fig04.png"/></fig></sec><sec id="s2-8-4"><title>Module 3: Teamwork (Dialogue-Choice Modality)</title><p>Interpersonal behavior in team settings was operationalized using the SYMLOG framework [<xref ref-type="bibr" rid="ref41">41</xref>], which positions actions along core social dimensions such as dominance, affiliation, and task orientation. The Yuzu teamwork scenario places the player in an emergency situation in which a spacecraft approaches the space station dangerously and rapid decisions must be made. Participants completed 4 sets of dialogue choices during interactions with nonplayer characters. Response options varied in both content and interpersonal tone (eg, more or less dominant or friendly), and some proposals originated from Earth as an external authority, allowing players to either follow or reject directives.</p><p>Each response option was precategorized along 3 axes aligned with SYMLOG: dominance vs submission, friendliness vs hostility, and acceptance vs rejection of norms and tasks. For each participant, mean scores were computed for these dimensions to derive an in-game interpersonal profile, which was compared with the simplified SYMLOG questionnaire [<xref ref-type="bibr" rid="ref42">42</xref>] administered outside the game. The dialogue-choice interface used in the teamwork module is shown in <xref ref-type="fig" rid="figure5">Figure 5</xref>.</p><fig position="float" id="figure5"><label>Figure 5.</label><caption><p>Screenshot of the Yuzu teamwork module (adaptation of the SYMLOG [System for the Multiple Level Observation of Groups] model).</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="games_v14i1e86054_fig05.png"/></fig></sec></sec><sec id="s2-9"><title>Masking</title><p>No masking procedures were implemented. Participants were aware that they were completing both the serious game modules and the corresponding reference measures within the same session. Because primary outcomes were computed algorithmically from response data and in-game trackers, there were no subjective raters whose condition assignment would require masking. Data were analyzed using deidentified datasets.</p></sec><sec id="s2-10"><title>Psychometrics</title><p>The reference tests (AELS [<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref39">39</xref>], IGT [<xref ref-type="bibr" rid="ref40">40</xref>], and simplified SYMLOG [<xref ref-type="bibr" rid="ref42">42</xref>]) have published validation evidence. In the present study, convergent validity was examined by correlating the in-game module scores with scores from these reference questionnaires.</p></sec><sec id="s2-11"><title>Conditions and Design</title><p>This study used a quantitative, cross-sectional, within-participant convergent validity design. No experimental manipulation of conditions was implemented; therefore, the design is best categorized as a nonexperimental observational design. Each participant completed, within the same session, each Yuzu module and its corresponding reference measure (in a counterbalanced order). This study is reported in accordance with the APA (American Psychological Association) <italic>JARS-Quant</italic> (<italic>Journal Article Reporting Standards for Quantitative Research</italic>) guidelines [<xref ref-type="bibr" rid="ref46">46</xref>].</p></sec><sec id="s2-12"><title>Data Diagnostics</title><p>Analyses were conducted on complete paired observations for each module, meaning that participants contributed to an analysis only when both the Yuzu score and the corresponding reference score were available. For the teamwork module, paired game data were unavailable for 4 participants; thus, the teamwork analyses were conducted with 35 participants. Because the primary correlational analyses were nonparametric (Spearman &#x03C1;), these analyses are less sensitive to distributional nonnormality and outliers than Pearson correlations. Data distributions and descriptive statistics were inspected prior to inferential analyses. No data imputation was performed.</p></sec><sec id="s2-13"><title>Analytic Strategy</title><p>All tests used a 2-sided &#x03B1; level of .05. The primary hypothesis (H1) was that each Yuzu module would demonstrate convergent validity with its corresponding reference measure. For the active-empathic listening module, convergent validity was evaluated using Spearman rank correlations between the Yuzu scores and the AELS reference scores. CIs for the correlations were estimated through bootstrap resampling (bias-corrected and accelerated [BCa], 5000 resamples). Absolute agreement was additionally evaluated using Bland-Altman analysis [<xref ref-type="bibr" rid="ref47">47</xref>] (mean bias and 95% limits of agreement). Relative agreement was quantified using the Lin concordance correlation coefficient, computed using the SimplyAgree module in Jamovi [<xref ref-type="bibr" rid="ref48">48</xref>].</p><p>For the decision-making module, IGT performance was summarized using the net score in the exploitation phase (blocks 5-10), defined as (C + D) &#x2212; (A + B). Equivalence between the Yuzu version and the reference IGT was tested using the two one-sided tests (TOST) procedure [<xref ref-type="bibr" rid="ref49">49</xref>], with an a priori equivalence margin of &#x00B1;5 points for the net score. Block-wise descriptive analyses were conducted to visualize learning trajectories across the task.</p><p>For the teamwork module, each in-game dialogue choice was mapped to the 3 SYMLOG dimensions, and mean scores were computed per participant. Convergent validity was evaluated using Spearman rank correlations between the Yuzu-derived dimension scores and simplified SYMLOG questionnaire scores [<xref ref-type="bibr" rid="ref42">42</xref>], with 95% bootstrap CIs (BCa, 5000 resamples). Given the small number of theory-driven module-level comparisons aligned with prespecified hypotheses, <italic>P</italic> values were interpreted alongside effect sizes and CIs. Exploratory analyses included block-wise learning curves for the IGT and inspection of agreement patterns in the active listening module using Bland-Altman plots.</p></sec><sec id="s2-14"><title>Ethical Considerations</title><p>This study was conducted in accordance with French regulations governing noninterventional behavioral research (Loi Jard&#x00E9;, L1121-1-1). The protocol involved no medical procedures, no clinical intervention, and no collection of health-related or sensitive personal data. Under applicable French regulations, submission to a Comit&#x00E9; de Protection des Personnes (CPP) or formal institutional review board (IRB) approval was not required. Consequently, no IRB approval number was assigned. All participants provided written informed consent prior to participation. The consent form specified that participation was voluntary, that withdrawal was possible at any time without justification, and that the data would be used exclusively for research purposes. Research data were deidentified prior to analysis and stored on password-protected servers accessible only to the research team. Participants did not receive monetary compensation. As an acknowledgment of their time, they received a detailed feedback report based on their soft skills. The research participant appearing in <xref ref-type="fig" rid="figure1">Figure 1</xref> provided written informed consent for publication of the photograph.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Participant Flow</title><p>A total of 39 participants were assessed for eligibility, enrolled, and completed the laboratory session. Complete paired data were available for 39 participants in the active-empathic listening and decision-making modules. For the teamwork module, complete paired data were available for 35 participants because the in-game teamwork data were not saved for 4 participants due to a technical issue. <xref ref-type="fig" rid="figure6">Figure 6</xref> presents the participant flow.</p><fig position="float" id="figure6"><label>Figure 6.</label><caption><p>Flowchart of the experimental procedure and assessment sequence.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="games_v14i1e86054_fig06.png"/></fig></sec><sec id="s3-2"><title>Recruitment</title><p>Recruitment and data collection were conducted in France between November 2024 and December 2024. This study used a single-session, cross-sectional design with no follow-up assessments.</p></sec><sec id="s3-3"><title>Statistics and Data Analysis</title><sec id="s3-3-1"><title>Analysis Strategy</title><p>All statistical tests used a 2-sided &#x03B1; level of .05. For equivalence testing, the TOST procedure was conducted at &#x03B1;=.05; therefore, 90% CIs are reported for equivalence tests, whereas 95% CIs are reported elsewhere. Missing data were limited to the teamwork module: 4 of 39 participants (10.3%) had missing in-game teamwork data because the module data were not saved. These cases were excluded from teamwork analyses only, resulting in an analytic sample size of 35 participants. No imputation was performed. Because the missingness was caused by a technical data-saving issue, it was assumed to be unrelated to the underlying teamwork construct; however, this assumption cannot be formally verified.</p><p>Results are reported in the order of the hypotheses stated in the <italic>Introduction</italic> section. Listening and teamwork hypotheses were tested using Spearman rank correlations with bootstrap CIs. Decision-making equivalence was tested using TOST and complemented by a paired 2-tailed <italic>t</italic> test and an effect size estimate.</p></sec><sec id="s3-3-2"><title>Active Empathic Listening Module (H1a)</title><p>Convergent validity was supported for active listening. Yuzu&#x2019;s active listening score showed a strong association with the reference AELS total score (Spearman &#x03C1;=0.890, 95% CI 0.804-0.948; <italic>P</italic>&#x003C;.001; 95% CIs estimated using bootstrap resampling [BCa, 5000 resamples]).</p><p>Agreement between the 2 versions was evaluated using Bland-Altman analysis (<xref ref-type="table" rid="table2">Table 2</xref>; <xref ref-type="fig" rid="figure7">Figure 7</xref>). The mean bias was close to 0, and the limits of agreement indicated no large systematic disagreement between the formats.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Bland-Altman analysis between Yuzu&#x2019;s active listening module and the reference Active-Empathic Listening Scale (N=39).</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Agreement</td><td align="left" valign="bottom">Estimate (95% CI)</td></tr></thead><tbody><tr><td align="left" valign="top">Bias (N=39)</td><td align="left" valign="top">0.0238 (&#x2013;0.0306 to 0.0783)</td></tr><tr><td align="left" valign="top">Lower limit of agreement</td><td align="left" valign="top">&#x2013;0.3056 (&#x2013;0.3995 to &#x2013;0.2117)</td></tr><tr><td align="left" valign="top">Upper limit of agreement</td><td align="left" valign="top">0.3533 (0.2594 to 0.4472)</td></tr></tbody></table></table-wrap><fig position="float" id="figure7"><label>Figure 7.</label><caption><p>Bland-Altman plot between Yuzu&#x2019;s active listening module and the reference Active-Empathic Listening Scale, suggesting the absence of systematic differences (N=39).</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="games_v14i1e86054_fig07.png"/></fig><p>The Lin concordance correlation coefficient (CCC) indicated good-to-excellent agreement between the 2 versions (CCC=0.899, 95% CI 0.817-0.946). Overall, these results support convergent validity and close agreement between Yuzu&#x2019;s listening module and the reference AELS.</p></sec><sec id="s3-3-3"><title>Decision-Making Module (H1b)</title><p>The analyses supported the equivalence between the Yuzu adaptation and the reference IGT during the exploitation phase. The TOST 90% CI for the mean difference fell entirely within the prespecified equivalence interval of &#x2212;5 to +5, and both one-sided tests were significant (<xref ref-type="table" rid="table3">Table 3</xref>; <xref ref-type="fig" rid="figure8">Figure 8</xref>). The paired 2-tailed <italic>t</italic> test was not statistically significant, and the standardized effect size was small (<xref ref-type="table" rid="table3">Table 3</xref>).</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Two one-sided tests (TOST) equivalence and paired <italic>t</italic> tests comparing the reference Iowa Gambling Task with Yuzu&#x2019;s game-based version, confirming equivalence (N=39).</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Statistic</td><td align="left" valign="bottom">Estimate</td><td align="left" valign="bottom"><italic>P</italic> value</td><td align="left" valign="bottom">Equivalence bounds (low, high)</td></tr></thead><tbody><tr><td align="left" valign="top"><italic>t</italic> test (net score)</td><td align="left" valign="top">0.913</td><td align="left" valign="top">.37</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup></td></tr><tr><td align="left" valign="top">TOST lower</td><td align="left" valign="top">3.615</td><td align="left" valign="top">.001</td><td align="left" valign="top">&#x2014;</td></tr><tr><td align="left" valign="top">TOST upper</td><td align="left" valign="top">&#x2013;1.789</td><td align="left" valign="top">.01</td><td align="left" valign="top">&#x2014;</td></tr><tr><td align="left" valign="top">Effect size (raw; 90% CI)</td><td align="left" valign="top">1.352 (&#x2013;1.224 to 3.927)</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2013;5.000, 5.000</td></tr><tr><td align="left" valign="top">Effect size (Hedges <italic>g</italic>[<italic>z</italic>]; 90% CI)</td><td align="left" valign="top">0.210 (&#x2013;0.311 to 0.752)</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2013;0.637, 0.637</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>Not applicable.</p></fn></table-wrap-foot></table-wrap><fig position="float" id="figure8"><label>Figure 8.</label><caption><p>Two one-sided tests for equivalence between Yuzu and the reference Iowa Gambling Task. Net score: the 90% CI for the mean difference lies entirely within the equivalence bounds (&#x00B1;5), supporting equivalence (N=39).</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="games_v14i1e86054_fig08.png"/></fig><p>To describe learning dynamics, <xref ref-type="fig" rid="figure9">Figure 9</xref> shows the evolution of the mean net score across blocks of 10 trials for both modalities. In both conditions, performance increased across blocks, with a similar shift from exploration to exploitation beginning at block 5.</p><fig position="float" id="figure9"><label>Figure 9.</label><caption><p>Evolution of the mean Iowa Gambling Task net score by 10-trial blocks for Yuzu and the reference task. (A) Mean net score across blocks for both modalities. (B) Mean net score by block and modality. Both conditions show a similar upward learning trajectory from exploration (blocks 1&#x2010;4) to exploitation (blocks 5&#x2010;10; N=39).</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="games_v14i1e86054_fig09.png"/></fig></sec><sec id="s3-3-4"><title>Teamwork Module (H1c)</title><p>Convergent validity was not supported for the teamwork module. Yuzu-derived teamwork profiles were clustered toward a highly cooperative style, consistent with a ceiling effect. Spearman rank correlations between Yuzu&#x2019;s 3 dimensions and the corresponding questionnaire dimensions were not statistically significant (<xref ref-type="table" rid="table4">Table 4</xref>).</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Spearman correlations between SYMLOG (System for the Multiple Level Observation of Groups) dimensions derived from Yuzu&#x2019;s teamwork module and the reference SYMLOG questionnaire, rejecting convergence (n=35)<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup>.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Dimension</td><td align="left" valign="bottom">Variable</td><td align="left" valign="bottom">Spearman &#x03C1; (95% CI)</td><td align="left" valign="bottom"><italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top">Dominance</td><td align="left" valign="top">Reference vs Yuzu</td><td align="left" valign="top">&#x2013;0.070 (&#x2013;0.420 to 0.280)</td><td align="left" valign="top">.69</td></tr><tr><td align="left" valign="top">Positivity</td><td align="left" valign="top">Reference vs Yuzu</td><td align="left" valign="top">0.082 (&#x2013;0.266 to 0.418)</td><td align="left" valign="top">.64</td></tr><tr><td align="left" valign="top">Task</td><td align="left" valign="top">Reference vs Yuzu</td><td align="left" valign="top">0.134 (&#x2013;0.260 to 0.494)</td><td align="left" valign="top">.44</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>95% CIs for Spearman correlations were estimated using bootstrap resampling (BCa, 5000 resamples; n=35).</p></fn></table-wrap-foot></table-wrap><p>Descriptive distributions suggested that the questionnaire differentiated participants across the 3 axes, whereas the Yuzu-derived scores were compressed and shifted in a way consistent with limited variance (<xref ref-type="table" rid="table5">Table 5</xref>).</p><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Descriptive statistics for SYMLOG (System for the Multiple Level Observation of Groups) dimensions in the questionnaire vs the Yuzu dialogue module (n=35).</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Dimensions</td><td align="left" valign="bottom">Median (minimum to maximum)</td></tr></thead><tbody><tr><td align="left" valign="top">Paper_dominance&#x2014;<italic>z</italic> score</td><td align="left" valign="top">&#x2013;0.0778 (&#x2013;2.01 to 2.128)</td></tr><tr><td align="left" valign="top">Yuzu_dominance&#x2014;<italic>z</italic> score</td><td align="left" valign="top">&#x2013;0.2395 (&#x2013;1.70 to 1.948)</td></tr><tr><td align="left" valign="top">Paper_positivity&#x2014;<italic>z</italic> score</td><td align="left" valign="top">&#x2013;0.1340 (&#x2013;2.22 to 1.608)</td></tr><tr><td align="left" valign="top">Yuzu_positivity&#x2014;<italic>z</italic> score</td><td align="left" valign="top">0.5657 (&#x2013;2.45 to 0.996)</td></tr><tr><td align="left" valign="top">Paper_task&#x2014;<italic>z</italic> score</td><td align="left" valign="top">&#x2013;0.1764 (&#x2013;2.71 to 2.359)</td></tr><tr><td align="left" valign="top">Yuzu_task&#x2014;<italic>z</italic> score</td><td align="left" valign="top">0.4219 (&#x2013;2.29 to 1.025)</td></tr></tbody></table></table-wrap></sec></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>This study examined the convergent validity of 3 soft skill assessment modalities embedded in the serious game Yuzu by comparing each module with an established reference measure. Regarding the primary hypothesis (H1), the results provided partial support. Convergent validity was strongly supported for the active-empathic listening module (H1a), with a high correlation and close agreement between Yuzu and the AELS reference instrument, and for the decision-making module (H1b), with statistically equivalent results across the gamified and reference versions of the IGT. By contrast, convergent validity was not supported for the teamwork module (H1c), where in-game scores showed restricted variance and no meaningful association with the reference SYMLOG questionnaire. Taken together, these findings suggest that the validity of serious game modules is construct-specific and modality-specific, and cannot be assumed to generalize across the components of a single platform.</p></sec><sec id="s4-2"><title>Interpretation</title><sec id="s4-2-1"><title>Construct Operationalization and Assessment Modality</title><p>Taken together, these results suggest that serious games can support convergent validity for some soft-skill constructs and modalities, but that validity evidence is highly dependent on how the construct is operationalized in gameplay and on the extent to which the measurement modality introduces method-related variance [<xref ref-type="bibr" rid="ref50">50</xref>,<xref ref-type="bibr" rid="ref51">51</xref>]. In practice, modules that closely preserve the structure of a reference instrument (eg, questionnaire-based formats) or a well-defined task paradigm (eg, feedback-based decision tasks) may translate more readily into game-based formats, whereas dialogue-choice approaches to interpersonal profiling may require stronger differentiation in scenario design and response options to generate sufficient variance for construct validation.</p></sec><sec id="s4-2-2"><title>Active-Empathic Listening</title><p>The very high correlation and strong agreement between Yuzu and the AELS indicate that the in-game questionnaire module produced scores that were closely aligned with the reference instrument. One interpretation is that the narrative embedding of the questionnaire did not distort the construct being measured while preserving the conceptual structure of the original scale [<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref39">39</xref>]. However, because both measures rely on self-reported judgments, shared method variance is a plausible contributor to the observed strength of the association [<xref ref-type="bibr" rid="ref51">51</xref>]. In addition, response biases associated with self-report, including impression management and socially desirable responding, may remain present even when the questionnaire is integrated into a game narrative [<xref ref-type="bibr" rid="ref24">24</xref>]. Therefore, the present findings support convergent validity with an established self-report instrument, but they do not yet demonstrate that Yuzu captures listening behavior as it unfolds in real interactions.</p></sec><sec id="s4-2-3"><title>Decision-Making</title><p>Equivalence results indicate that the Yuzu adaptation preserved the decision-making tendencies captured by the reference IGT during the exploitation phase. This finding suggests that the gamified interface and narrative context did not substantially alter the balance between immediate rewards and long-term outcomes that the task is intended to probe [<xref ref-type="bibr" rid="ref40">40</xref>]. The parallel block-wise trajectories are consistent with the interpretation that participants learned from feedback in a comparable manner across both modalities, in line with previous work [<xref ref-type="bibr" rid="ref45">45</xref>].</p><p>This result is noteworthy because it suggests that the structural properties of a well-defined cognitive paradigm can be preserved when transposed into a game narrative, provided that the core contingency structure, here the asymmetric reward schedule across decks, remains intact. From a measurement standpoint, task-based modules of this kind may be more robust to gamification than questionnaire or dialogue-based formats, because scoring relies on behavioral sequences rather than self-report, thereby reducing susceptibility to socially desirable responding [<xref ref-type="bibr" rid="ref24">24</xref>]. Future work should examine whether this equivalence holds across different gamified implementations and population subgroups.</p></sec><sec id="s4-2-4"><title>Teamwork</title><p>The absence of convergent validity for the teamwork module can plausibly be explained by restricted variance and a ceiling effect in the in-game indicators. If most participants select options that are clearly cooperative and norm-conforming, the resulting scores cannot differentiate interpersonal styles and will correlate poorly with questionnaire measures that capture stable individual differences. This aligns with the general expectation that restriction of range attenuates correlations [<xref ref-type="bibr" rid="ref52">52</xref>]. This pattern may also reflect the scenario&#x2019;s incentive structure. Cooperation might have been framed as the most sensible option for narrative success, making alternative responses appear irrational or socially inappropriate. Such tendencies are compatible with the broader literature on socially desirable responding, in which respondents may favor options that maintain a positive self-image or align with perceived expectations [<xref ref-type="bibr" rid="ref24">24</xref>].</p><p>Another possibility is a construct mismatch. The simplified SYMLOG questionnaire assesses broad interpersonal tendencies, whereas the Yuzu dialogue choices may have captured context-specific decision strategies under an emergency narrative. Like situational judgment tests, the dialogue-choice module may primarily capture respondents&#x2019; implicit trait policies or knowledge about which interpersonal behaviors are effective in a given situation, whereas the SYMLOG questionnaire targets broader dispositional teamwork tendencies [<xref ref-type="bibr" rid="ref53">53</xref>]. In that case, the lack of convergence could indicate that the game module measured situational behavior shaped by the scenario rather than the trait-like interpersonal profile assessed by the questionnaire.</p><p>From a measurement perspective, the current dialogue-choice format may also suffer from limited behavioral sampling. With only a small number of choice points, measurement precision is limited, and individual scores become sensitive to idiosyncratic responses to specific prompts. This issue is consistent with discussions of scenario-based assessment formats, where item sampling and option design strongly influence reliability and construct representation [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref54">54</xref>]. Increasing the number of observations, diversifying situations, and designing options that are less transparently &#x201C;good&#x201D; or &#x201C;bad&#x201D; could improve score reliability and increase variance.</p></sec></sec><sec id="s4-3"><title>Implications</title><p>These results suggest that validity evidence for serious game assessment should be developed module by module, with careful attention to modality and construct operationalization. For constructs assessed via in-game questionnaires, embedding within a narrative may preserve measurement properties while potentially improving acceptability. However, future work should include multimethod validation that goes beyond self-report and explicitly addresses method variance [<xref ref-type="bibr" rid="ref51">51</xref>].</p><p>For task-based modules, the decision-making findings support the feasibility of adapting classical paradigms into serious game formats while retaining key psychometric properties. Future research should evaluate the robustness across different devices and contexts of administration, test whether equivalence holds across subgroups, and further examine the psychometric characteristics of task-derived metrics [<xref ref-type="bibr" rid="ref45">45</xref>]. For dialogue-based interpersonal assessment, the present findings indicate that simple &#x201C;choice point&#x201D; implementations may be vulnerable to ceiling effects and restricted variance when scenarios strongly favor cooperation or make socially desirable options transparent [<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref52">52</xref>]. Future iterations should incorporate design strategies to increase behavioral differentiation, for example, by increasing the number of observations, diversifying contexts and interaction partners, introducing credible trade-offs, and reducing the transparency of one clearly superior option. Scenario-based assessment work suggests that option design and sampling of situations are central to capturing construct-relevant variability [<xref ref-type="bibr" rid="ref54">54</xref>].</p><p>From an applied perspective, the present study provides preliminary evidence that a serious game can yield convergent validity for some soft-skill indicators (active listening and decision-making) while highlighting that other constructs (teamwork) require further design and validation work. Organizations considering game-based assessment should treat different modules as distinct measurement instruments and require clear validity evidence for each targeted competency.</p><p>In the longer term, serious games could complement existing assessment batteries by providing standardized behavioral data in interactive contexts and potentially improving acceptability. Candidate reactions to assessment procedures are known to influence perceived fairness and organizational attractiveness. Therefore, it is important to evaluate user perceptions alongside psychometric performance [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref27">27</xref>].</p></sec><sec id="s4-4"><title>Limitations and Suggestions for Future Research</title><p>Several limitations should be acknowledged. First, the relatively small sample size limits the statistical power of the analyses [<xref ref-type="bibr" rid="ref55">55</xref>,<xref ref-type="bibr" rid="ref56">56</xref>]. The relatively homogeneous sample also limits external validity, and replication in more diverse occupational groups is needed [<xref ref-type="bibr" rid="ref57">57</xref>]. It is therefore possible that effects of small magnitude were not detected. A promising avenue for future research would be to increase sample sizes in subsequent empirical studies. Another point concerns the duration of the experimental sessions, which was relatively long in this study. Excessive session length may generate fatigue, reduce participants&#x2019; concentration, and affect engagement. This can particularly influence the quality of responses to questionnaires or the consistency of performance in game-based tasks [<xref ref-type="bibr" rid="ref58">58</xref>]. Future studies could explore shorter and more modular formats, for example, by splitting the session into multiple parts or by offering abbreviated versions of the tests.</p><p>Finally, the limitations observed in the dialogue-choice assessment call for methodological improvements. The low level of correlation with the SYMLOG questionnaire may be explained by a lack of nuance in the response options, a social desirability bias (participants&#x2019; tendency to choose the option perceived as the most socially acceptable), or the fact that in-game dialogues and questionnaires may capture only partially overlapping facets of interpersonal communication [<xref ref-type="bibr" rid="ref59">59</xref>,<xref ref-type="bibr" rid="ref60">60</xref>]. Future studies could diversify scenarios, introduce more contrasting dilemmas, analyze not only the final choice but also response times and hesitation [<xref ref-type="bibr" rid="ref61">61</xref>], and use AI-driven nonplayer characters to increase the credibility and sensitivity of the assessment.</p></sec><sec id="s4-5"><title>Conclusions</title><p>This study provides convergent validity evidence for 2 complementary assessment modalities embedded in a single serious game. Specifically, a gamified AELS-inspired module and a gamified IGT adaptation closely aligned with their corresponding reference measures under standardized administration, whereas the dialogue-choice teamwork module showed limited sensitivity and no convergence with questionnaire dimensions. More broadly, these findings support a modular approach to serious-game validation in human resources. Each module should be treated as a distinct measurement instrument requiring targeted validity evidence. The present work offers a concrete, replicable blueprint for module-by-module validation using matched reference measures and complementary analytic methods, and it highlights design requirements for interpersonal assessment (eg, reducing social desirability cues and increasing behavioral differentiation). Future research should extend this evidence base to predictive validity, subgroup robustness, and more ecologically rich social-interaction mechanics.</p></sec></sec></body><back><ack><p>We thank all participants for their time and involvement in this study. We also acknowledge the contributions of the development team and the Lorraine Laboratory of Psychology and Neuroscience of Behavioral Dynamics (2LPN, University of Lorraine, France) for their support for the scientific design of the assessment modules. We are grateful to the Centre for Research in Psychology: Cognition, Psychism, and Organizations (CRP-CPO, University of Picardie Jules Verne, France) for providing access to its facilities for data collection.</p><p>The authors declare the use of generative AI (GenAI) in the research and writing process. According to the Generative AI Delegation Taxonomy (GAIDeT, 2025), the following tasks were delegated to GenAI tools under full human supervision: proofreading and editing, summarizing text, and translation. The GenAI tool used was ChatGPT (GPT-5). Responsibility for the final manuscript lies entirely with the authors. GenAI tools are not listed as authors and do not bear responsibility for the final outcomes.</p></ack><notes><sec><title>Funding</title><p>The authors declared no financial support was received for this work.</p></sec><sec><title>Data Availability</title><p>Due to data protection regulations and contractual constraints with partner organizations, the datasets generated and analyzed during this study are not publicly available. Data may be made available from the corresponding author upon reasonable request, subject to appropriate data protection agreements.</p></sec></notes><fn-group><fn fn-type="conflict"><p>MB is the Head of Science at Yuzu, and LF is the co-founder of Yuzu. MB contributed to the scientific design and data analysis of the study. All procedures and analyses were conducted according to standard methodological and reporting guidelines. The other authors declare no conflicts of interest.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">2LPN</term><def><p>Lorraine Laboratory of Psychology and Neuroscience of Behavioral Dynamics</p></def></def-item><def-item><term id="abb2">AELS</term><def><p>Active-Empathic Listening Scale</p></def></def-item><def-item><term id="abb3">BCa</term><def><p>bias-corrected and accelerated</p></def></def-item><def-item><term id="abb4">CCC</term><def><p>concordance correlation coefficient</p></def></def-item><def-item><term id="abb5">CPP</term><def><p>Comit&#x00E9; de Protection des Personnes</p></def></def-item><def-item><term id="abb6">CRP-CPO</term><def><p>Centre for Research in Psychology: Cognition, Psychism, and Organizations</p></def></def-item><def-item><term id="abb7">IGT</term><def><p>Iowa Gambling Task</p></def></def-item><def-item><term id="abb8">IRB</term><def><p>institutional review board</p></def></def-item><def-item><term id="abb9">SYMLOG</term><def><p>System for the Multiple Level Observation of Groups</p></def></def-item><def-item><term id="abb10">TOST</term><def><p>two one-sided tests</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>McPherson</surname><given-names>J</given-names> </name><name name-style="western"><surname>Burns</surname><given-names>NR</given-names> </name></person-group><article-title>Assessing the validity of computer-game-like tests of processing speed and working memory</article-title><source>Behav Res Methods</source><year>2008</year><month>11</month><volume>40</volume><issue>4</issue><fpage>969</fpage><lpage>981</lpage><pub-id pub-id-type="doi">10.3758/BRM.40.4.969</pub-id><pub-id pub-id-type="medline">19001388</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Georgiou</surname><given-names>K</given-names> </name><name name-style="western"><surname>Nikolaou</surname><given-names>I</given-names> </name></person-group><article-title>Are applicants in favor of traditional or gamified assessment methods? Exploring applicant reactions towards a gamified selection method</article-title><source>Comput Human Behav</source><year>2020</year><month>08</month><volume>109</volume><fpage>106356</fpage><pub-id pub-id-type="doi">10.1016/j.chb.2020.106356</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mavridis</surname><given-names>A</given-names> </name><name name-style="western"><surname>Tsiatsos</surname><given-names>T</given-names> </name></person-group><article-title>Game&#x2010;based assessment: investigating the impact on test anxiety and exam performance</article-title><source>J Comput Assist Learn</source><year>2017</year><month>04</month><volume>33</volume><issue>2</issue><fpage>137</fpage><lpage>150</lpage><pub-id pub-id-type="doi">10.1111/jcal.12170</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ramos-Villagrasa</surname><given-names>PJ</given-names> </name><name name-style="western"><surname>Fern&#x00E1;ndez-Del-R&#x00ED;o</surname><given-names>E</given-names> </name><name name-style="western"><surname>Castro</surname><given-names>&#x00C1;</given-names> </name></person-group><article-title>Game-related assessments for personnel selection: a systematic review</article-title><source>Front Psychol</source><year>2022</year><volume>13</volume><fpage>952002</fpage><pub-id pub-id-type="doi">10.3389/fpsyg.2022.952002</pub-id><pub-id pub-id-type="medline">36248590</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="report"><person-group person-group-type="author"><name name-style="western"><surname>Deming</surname><given-names>DJ</given-names> </name></person-group><article-title>The value of soft skills in the labor market</article-title><year>2017</year><access-date>2026-07-28</access-date><publisher-name>National Bureau of Economic Research (NBER)</publisher-name><fpage>7</fpage><lpage>11</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://www.nber.org/sites/default/files/2019-08/2017number4.pdf">https://www.nber.org/sites/default/files/2019-08/2017number4.pdf</ext-link></comment></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Autor</surname><given-names>DH</given-names> </name></person-group><article-title>Why are there still so many jobs? The history and future of workplace automation</article-title><source>J Econ Perspect</source><year>2015</year><month>08</month><day>1</day><volume>29</volume><issue>3</issue><fpage>3</fpage><lpage>30</lpage><pub-id pub-id-type="doi">10.1257/jep.29.3.3</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Azim</surname><given-names>S</given-names> </name><name name-style="western"><surname>Gale</surname><given-names>A</given-names> </name><name name-style="western"><surname>Lawlor&#x2010;Wright</surname><given-names>T</given-names> </name><name name-style="western"><surname>Kirkham</surname><given-names>R</given-names> </name><name name-style="western"><surname>Khan</surname><given-names>A</given-names> </name><name name-style="western"><surname>Alam</surname><given-names>M</given-names> </name></person-group><article-title>The importance of soft skills in complex projects</article-title><source>Int J Manag Proj Bus</source><year>2010</year><month>06</month><day>22</day><volume>3</volume><issue>3</issue><fpage>387</fpage><lpage>401</lpage><pub-id pub-id-type="doi">10.1108/17538371011056048</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Burbekova</surname><given-names>S</given-names> </name></person-group><article-title>Soft skills as the most in-demand skills of future IT specialists</article-title><source>2021 IEEE International Conference on Smart Information Systems and Technologies (SIST)</source><year>2021</year><publisher-name>IEEE</publisher-name><fpage>1</fpage><lpage>5</lpage><pub-id pub-id-type="doi">10.1109/SIST50301.2021.9465935</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pol&#x00E1;kov&#x00E1;</surname><given-names>M</given-names> </name><name name-style="western"><surname>Suleimanov&#x00E1;</surname><given-names>JH</given-names> </name><name name-style="western"><surname>Madz&#x00ED;k</surname><given-names>P</given-names> </name><name name-style="western"><surname>Copu&#x0161;</surname><given-names>L</given-names> </name><name name-style="western"><surname>Moln&#x00E1;rov&#x00E1;</surname><given-names>I</given-names> </name><name name-style="western"><surname>Polednov&#x00E1;</surname><given-names>J</given-names> </name></person-group><article-title>Soft skills and their importance in the labour market under the conditions of Industry 5.0</article-title><source>Heliyon</source><year>2023</year><month>08</month><volume>9</volume><issue>8</issue><fpage>e18670</fpage><pub-id pub-id-type="doi">10.1016/j.heliyon.2023.e18670</pub-id><pub-id pub-id-type="medline">37593611</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Matteson</surname><given-names>ML</given-names> </name><name name-style="western"><surname>Anderson</surname><given-names>L</given-names> </name><name name-style="western"><surname>Boyden</surname><given-names>C</given-names> </name></person-group><article-title>&#x201C;Soft Skills&#x201D;: a phrase in search of meaning</article-title><source>Portal Libr Acad</source><year>2016</year><month>01</month><volume>16</volume><issue>1</issue><fpage>71</fpage><lpage>88</lpage><pub-id pub-id-type="doi">10.1353/pla.2016.0009</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Nasir</surname><given-names>ANM</given-names> </name><name name-style="western"><surname>Ali</surname><given-names>DF</given-names> </name><name name-style="western"><surname>Noordin</surname><given-names>MKB</given-names> </name><name name-style="western"><surname>Nordin</surname><given-names>MSB</given-names> </name></person-group><article-title>Technical skills and non-technical skills: predefinition concept</article-title><access-date>2026-08-04</access-date><conf-name>IETEC&#x2019;11 Conference</conf-name><conf-date>Jan 16-19, 2011</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.researchgate.net/publication/259782791_Technical_skills_and_non-technical_skills_predefinition_concept">https://www.researchgate.net/publication/259782791_Technical_skills_and_non-technical_skills_predefinition_concept</ext-link></comment></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="report"><person-group person-group-type="author"><name name-style="western"><surname>Venckut&#x0117;</surname><given-names>M</given-names> </name><name name-style="western"><surname>Berg Mulvik</surname><given-names>I</given-names> </name><name name-style="western"><surname>Lucas</surname><given-names>B</given-names> </name></person-group><article-title>Creativity&#x2014;a transversal skill for lifelong learning: an overview of existing concepts and practices</article-title><year>2020</year><access-date>2026-07-28</access-date><publisher-name>Publications Office of the European Union</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://publications.jrc.ec.europa.eu/repository/handle/JRC122016">https://publications.jrc.ec.europa.eu/repository/handle/JRC122016</ext-link></comment></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Lamri</surname><given-names>J</given-names> </name><name name-style="western"><surname>Barabel</surname><given-names>M</given-names> </name><name name-style="western"><surname>Lubart</surname><given-names>T</given-names> </name><name name-style="western"><surname>Meier</surname><given-names>O</given-names> </name></person-group><article-title>Chapitre 1. g&#x00E9;n&#x00E9;ralit&#x00E9;s sur les soft skills</article-title><source>Le D&#x00E9;fi Des Soft Skills: Comment Les D&#x00E9;velopper Au XXIe Si&#x00E8;cle? [Book in French]</source><year>2022</year><access-date>2026-07-28</access-date><publisher-name>Dunod</publisher-name><fpage>23</fpage><lpage>42</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://shs.cairn.info/le-defi-des-soft-skills--9782100830923-page-23?lang=fr">https://shs.cairn.info/le-defi-des-soft-skills--9782100830923-page-23?lang=fr</ext-link></comment></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="report"><person-group person-group-type="author"><name name-style="western"><surname>Haselberger</surname><given-names>D</given-names> </name><name name-style="western"><surname>Oberheumer</surname><given-names>P</given-names> </name><name name-style="western"><surname>Perez</surname><given-names>E</given-names> </name><name name-style="western"><surname>Cinque</surname><given-names>M</given-names> </name><name name-style="western"><surname>Capasso</surname><given-names>D</given-names> </name></person-group><article-title>Mediating soft skills at higher education institutions: guidelines for the design of learning situations supporting soft skills achievement</article-title><year>2012</year><access-date>2026-07-28</access-date><publisher-name>Education and Culture DG, Lifelong Learning Programme, European Union</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://docs.wixstatic.com/ugd/67267c_df6eeb2f47664754a4085f3bdf4bc7bb.pdf">https://docs.wixstatic.com/ugd/67267c_df6eeb2f47664754a4085f3bdf4bc7bb.pdf</ext-link></comment></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Succi</surname><given-names>C</given-names> </name><name name-style="western"><surname>Wieandt</surname><given-names>M</given-names> </name></person-group><article-title>Walk the talk: soft skills&#x2019; assessment of graduates</article-title><source>Eur J Manag Bus Econ</source><year>2019</year><month>07</month><day>16</day><volume>28</volume><issue>2</issue><fpage>114</fpage><lpage>125</lpage><pub-id pub-id-type="doi">10.1108/EJMBE-01-2019-0011</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Andr&#x00E9;s</surname><given-names>JC</given-names> </name><name name-style="western"><surname>Aguado</surname><given-names>D</given-names> </name><name name-style="western"><surname>Garc&#x00ED;a-Izquierdo</surname><given-names>AL</given-names> </name></person-group><article-title>Big Four LinkedIn dimensions: signals of soft skills?</article-title><source>J Work Organ Psychol</source><year>2023</year><month>08</month><day>9</day><volume>39</volume><issue>2</issue><fpage>75</fpage><lpage>88</lpage><pub-id pub-id-type="doi">10.5093/jwop2023a9</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Werner</surname><given-names>K</given-names> </name><name name-style="western"><surname>Junek</surname><given-names>O</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>C</given-names> </name></person-group><article-title>Event management skills in the post-COVID-19 world: insights from China, Germany, and Australia</article-title><source>Event Manag</source><year>2022</year><month>05</month><day>18</day><volume>26</volume><issue>4</issue><fpage>867</fpage><lpage>882</lpage><pub-id pub-id-type="doi">10.3727/152599521X16288665119558</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Aamodt</surname><given-names>MG</given-names> </name><name name-style="western"><surname>Custer</surname><given-names>H</given-names> </name></person-group><article-title>Who can best catch a liar? A meta-analysis of individual differences in detecting deception</article-title><source>Forensic Exam</source><year>2006</year><access-date>2026-07-28</access-date><volume>15</volume><issue>1</issue><fpage>6</fpage><lpage>11</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://psycnet.apa.org/record/2006-02487-001">https://psycnet.apa.org/record/2006-02487-001</ext-link></comment></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Galos</surname><given-names>DR</given-names> </name><name name-style="western"><surname>Coppock</surname><given-names>A</given-names> </name></person-group><article-title>Gender composition predicts gender bias: a meta-reanalysis of hiring discrimination audit experiments</article-title><source>Sci Adv</source><year>2023</year><month>05</month><day>5</day><volume>9</volume><issue>18</issue><fpage>eade7979</fpage><pub-id pub-id-type="doi">10.1126/sciadv.ade7979</pub-id><pub-id pub-id-type="medline">37146136</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Batinovic</surname><given-names>L</given-names> </name><name name-style="western"><surname>Howe</surname><given-names>M</given-names> </name><name name-style="western"><surname>Sinclair</surname><given-names>S</given-names> </name><name name-style="western"><surname>Carlsson</surname><given-names>R</given-names> </name></person-group><article-title>Ageism in hiring: a systematic review and meta-analysis of age discrimination</article-title><source>Collabra Psychol</source><year>2023</year><month>08</month><day>11</day><volume>9</volume><issue>1</issue><fpage>82194</fpage><pub-id pub-id-type="doi">10.1525/collabra.82194</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Johnson</surname><given-names>SK</given-names> </name><name name-style="western"><surname>Podratz</surname><given-names>KE</given-names> </name><name name-style="western"><surname>Dipboye</surname><given-names>RL</given-names> </name><name name-style="western"><surname>Gibbons</surname><given-names>E</given-names> </name></person-group><article-title>Physical attractiveness biases in ratings of employment suitability: tracking down the &#x201C;beauty is beastly&#x201D; effect</article-title><source>J Soc Psychol</source><year>2010</year><volume>150</volume><issue>3</issue><fpage>301</fpage><lpage>318</lpage><pub-id pub-id-type="doi">10.1080/00224540903365414</pub-id><pub-id pub-id-type="medline">20575336</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Paulhus</surname><given-names>DL</given-names> </name><name name-style="western"><surname>Vazire</surname><given-names>S</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Robins</surname><given-names>RW</given-names> </name><name name-style="western"><surname>Fraley</surname><given-names>RC</given-names> </name><name name-style="western"><surname>Krueger</surname><given-names>RF</given-names></name></person-group><article-title>The self-report method</article-title><source>Handbook of Research Methods in Personality Psychology</source><year>2007</year><publisher-name>The Guilford Press</publisher-name><fpage>224</fpage><lpage>239</lpage><pub-id pub-id-type="other">9781593851118</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Lucas</surname><given-names>RE</given-names> </name><name name-style="western"><surname>Baird</surname><given-names>BM</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Eid</surname><given-names>M</given-names> </name><name name-style="western"><surname>Diener</surname><given-names>E</given-names> </name></person-group><article-title>Global self-assessment</article-title><source>Handbook of Multimethod Measurement in Psychology</source><year>2006</year><publisher-name>American Psychological Association</publisher-name><fpage>29</fpage><lpage>42</lpage><pub-id pub-id-type="doi">10.1037/11383-003</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Paulhus</surname><given-names>DL</given-names> </name><name name-style="western"><surname>Reid</surname><given-names>DB</given-names> </name></person-group><article-title>Enhancement and denial in socially desirable responding</article-title><source>J Pers Soc Psychol</source><year>1991</year><volume>60</volume><issue>2</issue><fpage>307</fpage><lpage>317</lpage><pub-id pub-id-type="doi">10.1037/0022-3514.60.2.307</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Raven</surname><given-names>J</given-names> </name></person-group><article-title>The Raven&#x2019;s progressive matrices: change and stability over culture and time</article-title><source>Cogn Psychol</source><year>2000</year><month>08</month><volume>41</volume><issue>1</issue><fpage>1</fpage><lpage>48</lpage><pub-id pub-id-type="doi">10.1006/cogp.1999.0735</pub-id><pub-id pub-id-type="medline">10945921</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kuncel</surname><given-names>NR</given-names> </name><name name-style="western"><surname>Ones</surname><given-names>DS</given-names> </name><name name-style="western"><surname>Sackett</surname><given-names>PR</given-names> </name></person-group><article-title>Individual differences as predictors of work, educational, and broad life outcomes</article-title><source>Pers Individ Dif</source><year>2010</year><month>09</month><volume>49</volume><issue>4</issue><fpage>331</fpage><lpage>336</lpage><pub-id pub-id-type="doi">10.1016/j.paid.2010.03.042</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hausknecht</surname><given-names>JP</given-names> </name><name name-style="western"><surname>Day</surname><given-names>DV</given-names> </name><name name-style="western"><surname>Thomas</surname><given-names>SC</given-names> </name></person-group><article-title>Applicant reactions to selection procedures: an updated model and meta-analysis</article-title><source>Pers Psychol</source><year>2004</year><month>09</month><volume>57</volume><issue>3</issue><fpage>639</fpage><lpage>683</lpage><pub-id pub-id-type="doi">10.1111/j.1744-6570.2004.00003.x</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Leutner</surname><given-names>F</given-names> </name><name name-style="western"><surname>Codreanu</surname><given-names>SC</given-names> </name><name name-style="western"><surname>Brink</surname><given-names>S</given-names> </name><name name-style="western"><surname>Bitsakis</surname><given-names>T</given-names> </name></person-group><article-title>Game based assessments of cognitive ability in recruitment: validity, fairness and test-taking experience</article-title><source>Front Psychol</source><year>2022</year><volume>13</volume><fpage>942662</fpage><pub-id pub-id-type="doi">10.3389/fpsyg.2022.942662</pub-id><pub-id pub-id-type="medline">36743642</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hunter</surname><given-names>JE</given-names> </name><name name-style="western"><surname>Hunter</surname><given-names>RF</given-names> </name></person-group><article-title>Validity and utility of alternative predictors of job performance</article-title><source>Psychol Bull</source><year>1984</year><volume>96</volume><issue>1</issue><fpage>72</fpage><lpage>98</lpage><pub-id pub-id-type="doi">10.1037/0033-2909.96.1.72</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ployhart</surname><given-names>RE</given-names> </name><name name-style="western"><surname>Holtz</surname><given-names>BC</given-names> </name></person-group><article-title>The diversity-validity dilemma: strategies for reducing racioethnic and sex subgroup differences and adverse impact in selection</article-title><source>Pers Psychol</source><year>2008</year><volume>61</volume><issue>1</issue><fpage>153</fpage><lpage>172</lpage><pub-id pub-id-type="doi">10.1111/j.1744-6570.2008.00109.x</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Smither</surname><given-names>JW</given-names> </name><name name-style="western"><surname>Reilly</surname><given-names>RR</given-names> </name><name name-style="western"><surname>Millsap</surname><given-names>RE</given-names> </name><name name-style="western"><surname>At&#x0026;t</surname><given-names>KP</given-names> </name><name name-style="western"><surname>Stoffey</surname><given-names>RW</given-names> </name></person-group><article-title>Applicant reactions to selection procedures</article-title><source>Pers Psychol</source><year>1993</year><month>03</month><volume>46</volume><issue>1</issue><fpage>49</fpage><lpage>76</lpage><pub-id pub-id-type="doi">10.1111/j.1744-6570.1993.tb00867.x</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jennett</surname><given-names>C</given-names> </name><name name-style="western"><surname>Cox</surname><given-names>AL</given-names> </name><name name-style="western"><surname>Cairns</surname><given-names>P</given-names> </name><etal/></person-group><article-title>Measuring and defining the experience of immersion in games</article-title><source>Int J Hum Comput Stud</source><year>2008</year><volume>66</volume><issue>9</issue><fpage>641</fpage><lpage>661</lpage><pub-id pub-id-type="doi">10.1016/j.ijhcs.2008.04.004</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Weibel</surname><given-names>D</given-names> </name><name name-style="western"><surname>Wissmath</surname><given-names>B</given-names> </name><name name-style="western"><surname>Habegger</surname><given-names>S</given-names> </name><name name-style="western"><surname>Steiner</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Groner</surname><given-names>R</given-names> </name></person-group><article-title>Playing online games against computer- vs. human-controlled opponents: effects on presence, flow, and enjoyment</article-title><source>Comput Human Behav</source><year>2008</year><month>09</month><volume>24</volume><issue>5</issue><fpage>2274</fpage><lpage>2291</lpage><pub-id pub-id-type="doi">10.1016/j.chb.2007.11.002</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Altomari</surname><given-names>L</given-names> </name><name name-style="western"><surname>Altomari</surname><given-names>N</given-names> </name><name name-style="western"><surname>Iazzolino</surname><given-names>G</given-names> </name></person-group><article-title>Gamification and soft skills assessment in the development of a serious game: design and feasibility pilot study</article-title><source>JMIR Serious Games</source><year>2023</year><month>07</month><day>26</day><volume>11</volume><fpage>e45436</fpage><pub-id pub-id-type="doi">10.2196/45436</pub-id><pub-id pub-id-type="medline">37494078</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Pouezevara</surname><given-names>S</given-names> </name><name name-style="western"><surname>Powers</surname><given-names>S</given-names> </name><name name-style="western"><surname>Moore</surname><given-names>G</given-names> </name><name name-style="western"><surname>Strigel</surname><given-names>C</given-names> </name><name name-style="western"><surname>McKnight</surname><given-names>K</given-names> </name></person-group><article-title>Assessing soft skills in youth through digital games</article-title><conf-name>12th annual International Conference of Education, Research and Innovation</conf-name><conf-date>Nov 11-13, 2019</conf-date><conf-loc>Seville, Spain</conf-loc><fpage>3057</fpage><lpage>3066</lpage><pub-id pub-id-type="doi">10.21125/iceri.2019.0774</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Albuquerque</surname><given-names>J</given-names> </name><name name-style="western"><surname>Bittencourt</surname><given-names>II</given-names> </name><name name-style="western"><surname>Coelho</surname><given-names>JAPM</given-names> </name><name name-style="western"><surname>Silva</surname><given-names>AP</given-names> </name></person-group><article-title>Does gender stereotype threat in gamified educational environments cause anxiety? An experimental study</article-title><source>Comput Educ</source><year>2017</year><month>12</month><volume>115</volume><fpage>161</fpage><lpage>170</lpage><pub-id pub-id-type="doi">10.1016/j.compedu.2017.08.005</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Dinet</surname><given-names>J</given-names> </name><name name-style="western"><surname>Kitajima</surname><given-names>M</given-names> </name><name name-style="western"><surname>Fichet</surname><given-names>L</given-names> </name><name name-style="western"><surname>Paquet</surname><given-names>C</given-names> </name><name name-style="western"><surname>Coursac</surname><given-names>V</given-names> </name></person-group><article-title>A gamified sorting test to assess cognitive flexibility in personnel selection: a pilot study</article-title><access-date>2026-07-28</access-date><conf-name>COGNITIVE 2023: The Fifteenth International Conference on Advanced COGNITIVE Technologies and Applications</conf-name><conf-date>Jun 26-30, 2023</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.researchgate.net/profile/Muneo-Kitajima/publication/371955981_A_Gamified_Sorting_Test_to_Assess_Cognitive_Flexibility_in_Personnel_Selection_A_Pilot_Study/links/649e6706c41fb852dd40f53c/A-Gamified-Sorting-Test-to-Assess-Cognitive-Flexibility-in-Personnel-Selection-A-Pilot-Study.pdf">https://www.researchgate.net/profile/Muneo-Kitajima/publication/371955981_A_Gamified_Sorting_Test_to_Assess_Cognitive_Flexibility_in_Personnel_Selection_A_Pilot_Study/links/649e6706c41fb852dd40f53c/A-Gamified-Sorting-Test-to-Assess-Cognitive-Flexibility-in-Personnel-Selection-A-Pilot-Study.pdf</ext-link></comment></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bodie</surname><given-names>GD</given-names> </name></person-group><article-title>The Active-Empathic Listening Scale (AELS): conceptualization and evidence of validity within the interpersonal domain</article-title><source>Commun Q</source><year>2011</year><month>07</month><volume>59</volume><issue>3</issue><fpage>277</fpage><lpage>295</lpage><pub-id pub-id-type="doi">10.1080/01463373.2011.583495</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Drollinger</surname><given-names>T</given-names> </name><name name-style="western"><surname>Comer</surname><given-names>LB</given-names> </name><name name-style="western"><surname>Warrington</surname><given-names>PT</given-names> </name></person-group><article-title>Development and validation of the Active Empathetic Listening Scale</article-title><source>Psychol Mark</source><year>2006</year><month>02</month><volume>23</volume><issue>2</issue><fpage>161</fpage><lpage>180</lpage><pub-id pub-id-type="doi">10.1002/mar.20105</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Buelow</surname><given-names>MT</given-names> </name><name name-style="western"><surname>Suhr</surname><given-names>JA</given-names> </name></person-group><article-title>Construct validity of the Iowa Gambling Task</article-title><source>Neuropsychol Rev</source><year>2009</year><month>03</month><volume>19</volume><issue>1</issue><fpage>102</fpage><lpage>114</lpage><pub-id pub-id-type="doi">10.1007/s11065-009-9083-4</pub-id><pub-id pub-id-type="medline">19194801</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bales</surname><given-names>RF</given-names> </name><name name-style="western"><surname>Couch</surname><given-names>AS</given-names> </name></person-group><article-title>The value profile: a factor analytic study of value statements</article-title><source>Sociol Inq</source><year>1969</year><month>01</month><volume>39</volume><issue>1</issue><fpage>3</fpage><lpage>17</lpage><pub-id pub-id-type="doi">10.1111/j.1475-682X.1969.tb00934.x</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Blumberg</surname><given-names>HH</given-names> </name></person-group><article-title>A simplified version of the SYMLOG&#x00AE; trait rating form</article-title><source>Psychol Rep</source><year>2006</year><month>08</month><volume>99</volume><issue>1</issue><fpage>46</fpage><lpage>50</lpage><pub-id pub-id-type="doi">10.2466/pr0.99.1.46-50</pub-id><pub-id pub-id-type="medline">17037449</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="web"><article-title>Iowa Gambling Task</article-title><source>PsyToolkit</source><year>2025</year><month>12</month><day>8</day><access-date>2026-01-21</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.psytoolkit.org/experiment-library/igt.html">https://www.psytoolkit.org/experiment-library/igt.html</ext-link></comment></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>S&#x00E1;nchez</surname><given-names>JLG</given-names> </name><name name-style="western"><surname>Vela</surname><given-names>FLG</given-names> </name><name name-style="western"><surname>Simarro</surname><given-names>FM</given-names> </name><name name-style="western"><surname>Padilla-Zea</surname><given-names>N</given-names> </name></person-group><article-title>Playability: analysing user experience in video games</article-title><source>Behav Inf Technol</source><year>2012</year><month>10</month><volume>31</volume><issue>10</issue><fpage>1033</fpage><lpage>1054</lpage><pub-id pub-id-type="doi">10.1080/0144929X.2012.710648</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Schmitz</surname><given-names>F</given-names> </name><name name-style="western"><surname>Kunina-Habenicht</surname><given-names>O</given-names> </name><name name-style="western"><surname>Hildebrandt</surname><given-names>A</given-names> </name><name name-style="western"><surname>Oberauer</surname><given-names>K</given-names> </name><name name-style="western"><surname>Wilhelm</surname><given-names>O</given-names> </name></person-group><article-title>Psychometrics of the Iowa and Berlin gambling tasks: unresolved issues with reliability and validity for risk taking</article-title><source>Assessment</source><year>2020</year><month>03</month><volume>27</volume><issue>2</issue><fpage>232</fpage><lpage>245</lpage><pub-id pub-id-type="doi">10.1177/1073191117750470</pub-id><pub-id pub-id-type="medline">29310459</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Appelbaum</surname><given-names>M</given-names> </name><name name-style="western"><surname>Cooper</surname><given-names>H</given-names> </name><name name-style="western"><surname>Kline</surname><given-names>RB</given-names> </name><name name-style="western"><surname>Mayo-Wilson</surname><given-names>E</given-names> </name><name name-style="western"><surname>Nezu</surname><given-names>AM</given-names> </name><name name-style="western"><surname>Rao</surname><given-names>SM</given-names> </name></person-group><article-title>Journal Article Reporting Standards for Quantitative Research in Psychology: the APA Publications and Communications Board task force report</article-title><source>Am Psychol</source><year>2018</year><month>01</month><volume>73</volume><issue>1</issue><fpage>3</fpage><lpage>25</lpage><pub-id pub-id-type="doi">10.1037/amp0000191</pub-id><pub-id pub-id-type="medline">29345484</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Martin Bland</surname><given-names>JM</given-names> </name><name name-style="western"><surname>Altman</surname><given-names>DG</given-names> </name></person-group><article-title>Statistical methods for assessing agreement between two methods of clinical measurement</article-title><source>The Lancet</source><year>1986</year><month>02</month><volume>1</volume><issue>8476</issue><fpage>307</fpage><lpage>310</lpage><pub-id pub-id-type="doi">10.1016/S0140-6736(86)90837-8</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Caldwell</surname><given-names>AR</given-names> </name></person-group><article-title>SimplyAgree: an R package and jamovi module for simplifying agreement and reliability analyses</article-title><source>J Open Source Softw</source><year>2022</year><volume>7</volume><issue>71</issue><fpage>4148</fpage><pub-id pub-id-type="doi">10.21105/joss.04148</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Schuirmann</surname><given-names>DJ</given-names> </name></person-group><article-title>A comparison of the two one-sided tests procedure and the power approach for assessing the equivalence of average bioavailability</article-title><source>J Pharmacokinet Biopharm</source><year>1987</year><month>12</month><volume>15</volume><issue>6</issue><fpage>657</fpage><lpage>680</lpage><pub-id pub-id-type="doi">10.1007/BF01068419</pub-id><pub-id pub-id-type="medline">3450848</pub-id></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Campbell</surname><given-names>DT</given-names> </name><name name-style="western"><surname>Fiske</surname><given-names>DW</given-names> </name></person-group><article-title>Convergent and discriminant validation by the multitrait-multimethod matrix</article-title><source>Psychol Bull</source><year>1959</year><month>03</month><volume>56</volume><issue>2</issue><fpage>81</fpage><lpage>105</lpage><pub-id pub-id-type="medline">13634291</pub-id></nlm-citation></ref><ref id="ref51"><label>51</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Podsakoff</surname><given-names>PM</given-names> </name><name name-style="western"><surname>MacKenzie</surname><given-names>SB</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>JY</given-names> </name><name name-style="western"><surname>Podsakoff</surname><given-names>NP</given-names> </name></person-group><article-title>Common method biases in behavioral research: a critical review of the literature and recommended remedies</article-title><source>J Appl Psychol</source><year>2003</year><month>10</month><volume>88</volume><issue>5</issue><fpage>879</fpage><lpage>903</lpage><pub-id pub-id-type="doi">10.1037/0021-9010.88.5.879</pub-id><pub-id pub-id-type="medline">14516251</pub-id></nlm-citation></ref><ref id="ref52"><label>52</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Thorndike</surname><given-names>RL</given-names> </name></person-group><source>Personnel Selection: Test and Measurement Techniques</source><year>1949</year><access-date>2026-07-28</access-date><publisher-name>Wiley</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://psycnet.apa.org/record/1949-05074-000">https://psycnet.apa.org/record/1949-05074-000</ext-link></comment></nlm-citation></ref><ref id="ref53"><label>53</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Motowidlo</surname><given-names>SJ</given-names> </name><name name-style="western"><surname>Hooper</surname><given-names>AC</given-names> </name><name name-style="western"><surname>Jackson</surname><given-names>HL</given-names> </name></person-group><article-title>Implicit policies about relations between personality traits and behavioral effectiveness in situational judgment items</article-title><source>J Appl Psychol</source><year>2006</year><month>07</month><volume>91</volume><issue>4</issue><fpage>749</fpage><lpage>761</lpage><pub-id pub-id-type="doi">10.1037/0021-9010.91.4.749</pub-id><pub-id pub-id-type="medline">16834503</pub-id></nlm-citation></ref><ref id="ref54"><label>54</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kepes</surname><given-names>S</given-names> </name><name name-style="western"><surname>Keener</surname><given-names>SK</given-names> </name><name name-style="western"><surname>Lievens</surname><given-names>F</given-names> </name><name name-style="western"><surname>McDaniel</surname><given-names>MA</given-names> </name></person-group><article-title>An integrative, systematic review of the situational judgment test literature</article-title><source>J Manage</source><year>2025</year><month>07</month><volume>51</volume><issue>6</issue><fpage>2278</fpage><lpage>2319</lpage><pub-id pub-id-type="doi">10.1177/01492063241288545</pub-id></nlm-citation></ref><ref id="ref55"><label>55</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cohen</surname><given-names>J</given-names> </name></person-group><article-title>Statistical power analysis</article-title><source>Curr Dir Psychol Sci</source><year>1992</year><month>06</month><volume>1</volume><issue>3</issue><fpage>98</fpage><lpage>101</lpage><pub-id pub-id-type="doi">10.1111/1467-8721.ep10768783</pub-id></nlm-citation></ref><ref id="ref56"><label>56</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sch&#x00F6;nbrodt</surname><given-names>FD</given-names> </name><name name-style="western"><surname>Perugini</surname><given-names>M</given-names> </name></person-group><article-title>At what sample size do correlations stabilize?</article-title><source>J Res Pers</source><year>2013</year><month>10</month><volume>47</volume><issue>5</issue><fpage>609</fpage><lpage>612</lpage><pub-id pub-id-type="doi">10.1016/j.jrp.2013.05.009</pub-id></nlm-citation></ref><ref id="ref57"><label>57</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Henrich</surname><given-names>J</given-names> </name><name name-style="western"><surname>Heine</surname><given-names>SJ</given-names> </name><name name-style="western"><surname>Norenzayan</surname><given-names>A</given-names> </name></person-group><article-title>The weirdest people in the world?</article-title><source>Behav Brain Sci</source><year>2010</year><month>06</month><volume>33</volume><issue>2-3</issue><fpage>61</fpage><lpage>83</lpage><pub-id pub-id-type="doi">10.1017/S0140525X0999152X</pub-id><pub-id pub-id-type="medline">20550733</pub-id></nlm-citation></ref><ref id="ref58"><label>58</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Galesic</surname><given-names>M</given-names> </name><name name-style="western"><surname>Bosnjak</surname><given-names>M</given-names> </name></person-group><article-title>Effects of questionnaire length on participation and indicators of response quality in a web survey</article-title><source>Public Opin Q</source><year>2009</year><month>06</month><day>1</day><volume>73</volume><issue>2</issue><fpage>349</fpage><lpage>360</lpage><pub-id pub-id-type="doi">10.1093/poq/nfp031</pub-id></nlm-citation></ref><ref id="ref59"><label>59</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Whetzel</surname><given-names>DL</given-names> </name><name name-style="western"><surname>McDaniel</surname><given-names>MA</given-names> </name></person-group><article-title>Situational judgment tests: an overview of current research</article-title><source>Hum Resour Manage Rev</source><year>2009</year><month>09</month><volume>19</volume><issue>3</issue><fpage>188</fpage><lpage>202</lpage><pub-id pub-id-type="doi">10.1016/j.hrmr.2009.03.007</pub-id></nlm-citation></ref><ref id="ref60"><label>60</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Whetzel</surname><given-names>D</given-names> </name><name name-style="western"><surname>Sullivan</surname><given-names>T</given-names> </name><name name-style="western"><surname>McCloy</surname><given-names>RA</given-names> </name></person-group><article-title>Situational judgment tests: an overview of development practices and psychometric characteristics</article-title><source>Pers Assess Decis</source><year>2020</year><volume>6</volume><issue>1</issue><fpage>1</fpage><pub-id pub-id-type="doi">10.25035/pad.2020.01.001</pub-id></nlm-citation></ref><ref id="ref61"><label>61</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Scrams</surname><given-names>DJ</given-names> </name><name name-style="western"><surname>Schnipke</surname><given-names>DL</given-names> </name></person-group><article-title>Making use of response times in standardized tests: are accuracy and speed measuring the same thing?</article-title><access-date>2026-08-07</access-date><conf-name>The Annual Meeting of the American Educational Research Association</conf-name><conf-date>Mar 24-28, 1997</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://files.eric.ed.gov/fulltext/ED409357.pdf">https://files.eric.ed.gov/fulltext/ED409357.pdf</ext-link></comment></nlm-citation></ref></ref-list></back></article>