<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e100559</article-id><article-id pub-id-type="doi">10.2196/100559</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Evaluation of the Square Eyes Model as a Screening Tool for Identifying Digital Technologies in Wearable Camera Images Among Children: Laboratory Study</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Rasmussen</surname><given-names>Charlotte Lund</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Sanders</surname><given-names>Taren</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Howie</surname><given-names>Erin Kaye</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Beynon</surname><given-names>Amber</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Hendry</surname><given-names>Danica</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Zabatiero</surname><given-names>Juliana</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Campbell</surname><given-names>Amity</given-names></name><degrees>PhD, Prof Dr</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Straker</surname><given-names>Leon</given-names></name><degrees>Prof Dr</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib></contrib-group><aff id="aff1"><institution>School of Allied Health, Curtin University</institution><addr-line>Building 401, Kent Street</addr-line><addr-line>Perth</addr-line><addr-line>Western Australia</addr-line><country>Australia</country></aff><aff id="aff2"><institution>ARC Centre of Excellence for the Digital Child</institution><addr-line>Brisbane</addr-line><country>Australia</country></aff><aff id="aff3"><institution>Institute for Positive Psychology and Education, Australian Catholic University</institution><addr-line>North Sydney</addr-line><addr-line>Queensland</addr-line><country>Australia</country></aff><aff id="aff4"><institution>Department of Health, Human Performance and Recreation, University of Arkansas</institution><addr-line>Fayetteville</addr-line><addr-line>AR</addr-line><country>United States</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Balcarras</surname><given-names>Matthew</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Moyano</surname><given-names>Adri&#x00E1;n Marco</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Meredith-Jones</surname><given-names>Kim</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Charlotte Lund Rasmussen, PhD, School of Allied Health, Curtin University, Building 401, Kent Street, Perth, Western Australia, 6102, Australia, 61 (0) 8 92661771; <email>charlotte.rasmussen@curtin.edu.au</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>14</day><month>9</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e100559</elocation-id><history><date date-type="received"><day>07</day><month>05</month><year>2026</year></date><date date-type="rev-recd"><day>21</day><month>08</month><year>2026</year></date><date date-type="accepted"><day>26</day><month>08</month><year>2026</year></date></history><copyright-statement>&#x00A9; Charlotte Lund Rasmussen, Taren Sanders, Erin Kaye Howie, Amber Beynon, Danica Hendry, Juliana Zabatiero, Amity Campbell, Leon Straker. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 14.9.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e100559"/><abstract><sec><title>Background</title><p>Accurate measurements of children&#x2019;s digital technology use are essential for understanding its potential implications on health and well-being. Wearable cameras can provide such measurements, but image coding is a high burden for researchers. Machine learning&#x2013;based object-recognition models have the potential to reduce this burden by identifying images containing technology.</p></sec><sec><title>Objective</title><p>This study aims to evaluate the performance of an object recognition model, the Square Eyes model, as a screening tool for identifying technologies in wearable camera images among children for further human review, as well as to examine the potential influence of face-blurring methods on the model&#x2019;s performance.</p></sec><sec sec-type="methods"><title>Methods</title><p>This study used data collected on 48 children (aged 3&#x2010;14 y) during an approximately 1-hour laboratory session. The children performed various technology-related tasks while wearing a camera. A total of 221,226 images were coded by humans and processed through the Square Eyes model. The performance of the Square Eyes model as a screening tool was evaluated by (1) assessing agreement between the model and human coding; (2) evaluating the N-back algorithm, an algorithm embedded in the model aimed to flag images requiring human review; and (3) examining the potential influence of facial-blurring on model performance.</p></sec><sec sec-type="results"><title>Results</title><p>Humans detected technology in 92,745 (41.9%) images, and the Square Eyes model detected technologies with an overall accuracy of 78.0%. When considering specific technologies, agreement between the model and human coders was the highest for <italic>Television</italic> (n=19,148, 54.3%) and <italic>Laptop</italic> (n=8492, 44.5%) and lowest for smaller devices such as <italic>Smartphone</italic> (n=2600, 31.3%) and <italic>Tablet</italic> (n=3685, 25.1%). The model&#x2019;s N-back algorithm effectively flagged images that required further human review, with only 7144 (3.2%) images that were not flagged for screening containing a human-coded technology. An explorative analysis indicated that using a square face-blurring with border could have reduced the model&#x2019;s ability to accurately detect technologies.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>The Square Eyes model demonstrated overall satisfying accuracy in detecting technologies and successfully flagged images that required further review by humans. These findings suggest that the model could be used as an effective screening tool for reducing the burden of human coding. However, the model could be improved to more accurately detect smaller devices, and the form of facial blurring in images should be considered.</p></sec></abstract><kwd-group><kwd>children</kwd><kwd>wearable camera</kwd><kwd>machine learning</kwd><kwd>object recognition</kwd><kwd>image processing</kwd><kwd>screens</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Children&#x2019;s exposure to digital technologies has significantly increased over the last few decades, and the age of first exposure has decreased [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref3">3</xref>]. Along with the increase in daily exposure, there has also been a rapid expansion in the range of screen-based technology devices available to children, including both stationary (eg, television and desktop computers) and portable devices (eg, tablets and smartphones) [<xref ref-type="bibr" rid="ref2">2</xref>]. Accordingly, the potential impacts of technology on children&#x2019;s health and well-being are increasingly being debated and investigated [<xref ref-type="bibr" rid="ref4">4</xref>]. Some studies have found excessive screen time to be associated with negative health effects such as obesity, sleep disturbances, and attention issues [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref6">6</xref>]. Other studies have reported positive associations between technology used for educational or social purposes and children&#x2019;s learning capacities and social interactions [<xref ref-type="bibr" rid="ref6">6</xref>-<xref ref-type="bibr" rid="ref8">8</xref>]. Thus, the potential impacts of screen-based technology, in particular, on children&#x2019;s health and well-being are complex and influenced not only by exposure time but also by the content and context of use [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref9">9</xref>]. Accurate measurements of technology use are therefore essential for a comprehensive understanding of the topic.</p><p>Children&#x2019;s technology use has commonly been measured using parent- or self-reported measures such as questionnaires, diaries, or interviews [<xref ref-type="bibr" rid="ref10">10</xref>]. Advantages of these methods include low cost and participant burden. However, data obtained from these methods can be inaccurate due to social desirability and recall biases [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref11">11</xref>]. Objective methods have the potential to overcome these limitations and accurately capture how and when children use technology. Wearable cameras are increasingly being used within health behavior research, given the ability to capture recordings of the wearer&#x2019;s surroundings [<xref ref-type="bibr" rid="ref12">12</xref>]. Wearable cameras are small devices typically affixed to clothing that can capture audio-visual recordings from a first-person perspective and have been used to investigate children&#x2019;s and adolescents&#x2019; diets [<xref ref-type="bibr" rid="ref13">13</xref>], exposure to food [<xref ref-type="bibr" rid="ref14">14</xref>] and alcohol marketing [<xref ref-type="bibr" rid="ref15">15</xref>], smoking [<xref ref-type="bibr" rid="ref16">16</xref>], and travel to school [<xref ref-type="bibr" rid="ref17">17</xref>]. Moreover, wearable cameras have been used to capture exposure, content, and context of technology use among children [<xref ref-type="bibr" rid="ref18">18</xref>-<xref ref-type="bibr" rid="ref20">20</xref>] and adolescents [<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref22">22</xref>]. These findings demonstrate the potential of wearable cameras to provide objective and useful assessments of technology use.</p><p>Although wearable cameras have the potential to offer high-quality data on children&#x2019;s technology use, there are challenges for researchers to consider. The manual coding of the data, the typical gold standard of processing, is very time-consuming and thus results in high researcher burden [<xref ref-type="bibr" rid="ref10">10</xref>]. Work has been done to reduce this burden and improve interrater reliability between coders by developing a standardized protocol for annotating technology use among children based on wearable camera images [<xref ref-type="bibr" rid="ref23">23</xref>]. Another option could be to use machine learning&#x2013;based object-recognition methods, which is a subfield of computer vision that focuses on detecting, analyzing and classifying specific objects within images and videos [<xref ref-type="bibr" rid="ref24">24</xref>]. Object recognition has been successfully implemented in several research areas, such as medical image analysis and security surveillance, and could significantly reduce the extensive researcher burden and potential errors associated with manual coding of images [<xref ref-type="bibr" rid="ref24">24</xref>]. However, understanding the feasibility and validity of object recognition methods for identifying technologies in images from children wearing cameras is important for confidence in using this approach.</p><p>The accuracy of using object recognition methods for detecting technologies in wearable camera images is increasingly being investigated. One experimental study used an object recognition model to detect laptop, phone, and television screens on 1029 wearable camera images collected among adults and found a mean accuracy of 82% [<xref ref-type="bibr" rid="ref25">25</xref>]. Another study used an object recognition model to identify screen-based technologies from wearable camera images collected from 5 adults over a 7-day period (n=30,661 images) [<xref ref-type="bibr" rid="ref26">26</xref>]. Based on a subset of manually annotated images (n=2024), the model showed a mean accuracy of 85% for capturing phone, laptop, tablet, desktop computer, and television screens. Finally, one study used an object recognition model to identify technologies among 1800 images captured from 30 children aged between 3 and 5 years, who wore a camera over 2 days [<xref ref-type="bibr" rid="ref19">19</xref>]. The authors found that the model could detect computers, smartphones, and televisions with a mean accuracy of 93%. Taken together, these findings support the feasibility of using object recognition models to identify technologies in wearable camera images with good accuracy.</p><p>However, there are several limitations of object recognition models. One limitation of such models is the limited ability to capture content and contextual aspects of technology use. Therefore, a hybrid approach that uses both the capacity of machine learning to identify images with technologies and the capacity of humans to code a range of aspects of technology use could together provide a more efficient and accurate data processing approach. The Square Eyes model was developed with this purpose in mind [<xref ref-type="bibr" rid="ref27">27</xref>]. Specifically, this model was developed as a screening tool to identify the presence of technologies in wearable camera images and flag images that were likely to contain technologies and thus require further human review. The Square Eyes model is currently being used in the KidVision project [<xref ref-type="bibr" rid="ref28">28</xref>] but has not been evaluated on datasets that were not used for model development. However, evaluating model performance on independent datasets is essential to determine generalizability and robustness.</p><p>Another challenge with the use of wearable cameras is ethical concerns, including data confidentiality and security, the intrusive nature of the device, and the risk of unintentionally recording sensitive images of participants and third parties [<xref ref-type="bibr" rid="ref12">12</xref>]. These issues may be even more pronounced in studies with vulnerable populations such as children. An ethical framework has been developed by Kelly et al [<xref ref-type="bibr" rid="ref29">29</xref>], which details these ethical issues and provides a guideline to ensure the protection of participants and third parties when using wearable cameras [<xref ref-type="bibr" rid="ref29">29</xref>]. One of the recommendations is to obscure faces and identifying features, commonly achieved through image blurring techniques [<xref ref-type="bibr" rid="ref29">29</xref>]. While this approach can alleviate some ethical concerns, a potential consequence is reduced image quality, which in turn may influence the performance of object recognition models [<xref ref-type="bibr" rid="ref30">30</xref>]. However, the potential impact of face blurring on the performance of a model for identifying technologies is unknown.</p><p>The primary aim of this study was to evaluate the performance of the Square Eyes model as a screening tool for identifying technologies in wearable camera images among children for further review by humans. Additionally, the potential influence of face-blurring methods on the model&#x2019;s ability to detect technologies was examined.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Design</title><p>This study was based on data collected in a laboratory-based setting while children performed a variety of tasks with and without technology.</p></sec><sec id="s2-2"><title>Ethical Considerations</title><p>The study gained ethical approval from the Curtin University Human Ethics Committee (HRE2022-0157). Written informed consent was obtained from caregivers and age-appropriate assent (written or verbal) was obtained from children.</p></sec><sec id="s2-3"><title>Participants</title><p>Children with typical development aged 3 to 14 years were recruited through university networks, word of mouth, flyers, and social media posts. Children with parent-reported psychological or physical clinical diagnoses that could affect their ability to follow instructions or perform tasks were excluded.</p><p>Recruitment was sex- and age-stratified for balanced representation across girls and boys and the following three age brackets: aged 3 to 5 years to represent a preschool population, aged 6 to 10 years to represent a primary school population, and aged 11 to 14 years to represent secondary school or early adolescent population. Age categories were defined to reflect transitions at school leading to different day-to-day structures in their environment [<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref32">32</xref>] and acknowledge technology engagement differences across childhood [<xref ref-type="bibr" rid="ref33">33</xref>,<xref ref-type="bibr" rid="ref34">34</xref>].</p></sec><sec id="s2-4"><title>Data Collection</title><p>Data were collected between July 2022 and October 2022. Children attended a single approximately 1-hour data collection session at a Curtin University laboratory with their caregivers either alone (n=10) or with a friend or sibling (n=38). At each data collection session, 2 to 3 researchers with experience working with children were present. The laboratory was set up with different activities aimed at capturing different tasks with and without technology. Caregivers provided informed consent and completed a brief sociodemographic questionnaire. Depending on their age, children were then asked for verbal or written assent and subsequently fitted with the wearable camera.</p><p>One Brinno TLC120 Automated Wearable Camera operating at a 1-second frequency was fitted to the child&#x2019;s chest using an adjustable chest-mounted harness. The Brinno camera weighs 101 g, is 60&#x00D7;60&#x00D7;35 mm in size, captures a 112&#x00B0; field of view. The Brinno camera was used to capture technology exposure during the data collection session and has previously been used in adolescent [<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref22">22</xref>] and adult populations [<xref ref-type="bibr" rid="ref35">35</xref>].</p><p>After becoming familiar with the laboratory and equipment, children performed a range of tasks for 2 to 3 minutes each. Tasks were generally completed in a standard order, but adjustments were made to maintain participant enjoyment. Participants engaged in tasks such as watching shows on a television or laptop, gaming, reading, drawing, and using a smartphone for taking photographs. Children received specific instructions for the postures or movements required for each task, allowing the Brinno camera to capture images from various positions and movement patterns. Table S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> provides a full overview of the tasks completed by the children and the corresponding technological devices.</p></sec><sec id="s2-5"><title>Data Processing and Coding</title><sec id="s2-5-1"><title>Wearable Camera Images</title><p>The Brinno camera automatically generated time-lapse video files (.avi), which were subsequently downloaded and converted into individual images (.jpg) using the open-source software FFmpeg (version 4.3). The images were coded for technologies by a member of the research team, as well as being processed through the Square Eyes model. All images used for the primary analysis had faces blurred by a square with a border applied to the faces within the images using a custom-made Python program.</p></sec><sec id="s2-5-2"><title>Human Coding</title><p>A member of the research team viewed images for each participant and recorded the corresponding technology code for each image in a spreadsheet alongside the image number (see <xref ref-type="table" rid="table1">Table 1</xref> for an overview of technology codes). The human coder completed several hours of structured training using a study coding protocol prior to commencing annotation. Moreover, coding for the first 3 participants was reviewed by an experienced researcher, and feedback was provided before the remaining images were coded. Images were date- and time-stamped. Moreover, images were coded in chronological order. The coding protocol applied was as follows. First, the coder would assess if an image was suitable for coding. Images that were deemed uncodable due to poor image quality, such as motion blur, poor lighting, or being obscured were excluded. Next, for codable images, the coder would determine the presence of any visible technology. If no technology was identified, the image was classified as &#x201C;no digital technology.&#x201D; When digital technology was both visible and had its screen turned on, it was classified as &#x201C;Technology 1,&#x201D; referencing the technology associated with the primary task being performed by the child. In instances where multiple technologies appeared in the same image, additional devices were labeled as &#x201C;Technology 2&#x201D; or &#x201C;Technology 3,&#x201D; allowing up to 3 distinct technologies to be recorded per image.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Technology codes and their definition used by the human coders and the Square Eyes model, respectively.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Code comparison number</td><td align="left" valign="bottom" colspan="2">Human coding</td><td align="left" valign="bottom" colspan="2">Square Eyes model</td></tr><tr><td align="left" valign="bottom"/><td align="left" valign="top">Technology code</td><td align="left" valign="top">Definition</td><td align="left" valign="top">Technology code</td><td align="left" valign="top">Definition</td></tr></thead><tbody><tr><td align="char" char="." valign="top">1</td><td align="left" valign="top">Desktop computer</td><td align="left" valign="top">A computer that fits on a desk but is not easily moved from place to place. Has a monitor, keyboard, mouse, and tower.</td><td align="left" valign="top">Desktop computer</td><td align="left" valign="top">A computer that fits on a desk but is not easily moved from place to place. Has a monitor, keyboard, mouse, and tower. Refers to the physical computer box rather than the monitor.</td></tr><tr><td align="char" char="." valign="top">2</td><td align="left" valign="top">No matching code</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup></td><td align="left" valign="top">Computer monitor</td><td align="left" valign="top">The monitor of a computer that fits on a desk but is not easily moved from place to place.</td></tr><tr><td align="char" char="." valign="top">3</td><td align="left" valign="top">No matching code</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">Computer mouse</td><td align="left" valign="top">The mouse of a computer that fits on a desk but is not easily moved from place to place.</td></tr><tr><td align="char" char="." valign="top">4</td><td align="left" valign="top">No matching code</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">Computer keyboard</td><td align="left" valign="top">The keyboard of a computer that fits on a desk but is not easily moved from place to place.</td></tr><tr><td align="char" char="." valign="top">5</td><td align="left" valign="top">Laptop computer</td><td align="left" valign="top">A computer that is small enough to be carried around easily and is flat when closed. Indicated by an inbuilt keyboard.</td><td align="left" valign="top">Laptop computer</td><td align="left" valign="top">A computer that is battery operated and has an integrated screen. Indicated by an inbuilt keyboard.</td></tr><tr><td align="char" char="." valign="top">6</td><td align="left" valign="top">Smartphone</td><td align="left" valign="top">A handheld device that can be used as a small computer, connect to the internet, and run applications.</td><td align="left" valign="top">Mobile phone</td><td align="left" valign="top">A portable, handheld wireless device that enables voice calls, text messaging, internet access, and runs applications.</td></tr><tr><td align="char" char="." valign="top">7</td><td align="left" valign="top">No matching code</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">iPod</td><td align="left" valign="top">A small electronic device for playing and storing digital audio and video files.</td></tr><tr><td align="char" char="." valign="top">8</td><td align="left" valign="top">Tablet</td><td align="left" valign="top">A small, flat computer that is controlled by touching the screen with one&#x2019;s finger or a special pen. Does not require a keyboard or mouse. Includes e-readers.</td><td align="left" valign="top">Tablet</td><td align="left" valign="top">A small, flat computer that is controlled by touching the screen with one&#x2019;s finger or a special pen. Does not require a keyboard or mouse. Includes e-readers.</td></tr><tr><td align="char" char="." valign="top">9</td><td align="left" valign="top">Television</td><td align="left" valign="top">A device shaped like a box or rectangle with a screen that receives electrical signals and changes them into moving images. Can stand alone or be mounted to a wall. Can be a smart TV (ie, internet connected)</td><td align="left" valign="top">Television</td><td align="left" valign="top">A device shaped like a box or rectangle with a screen that receives electrical signals and changes them into moving images. Can stand alone or be mounted to a wall.</td></tr><tr><td align="char" char="." valign="top">10</td><td align="left" valign="top">No matching code</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">Remote</td><td align="left" valign="top">A handheld, portable device used to operate machinery, appliances, or digital equipment from a distance.</td></tr><tr><td align="char" char="." valign="top">11</td><td align="left" valign="top">Handheld game console</td><td align="left" valign="top">Portable, self-contained devices that have a built-in screen, game controls, and speakers.</td><td align="left" valign="top">Handheld game console</td><td align="left" valign="top">Portable, self-contained devices that have a built-in screen, game controls, and speakers.</td></tr><tr><td align="char" char="." valign="top">12</td><td align="left" valign="top">Combined controller</td><td align="left" valign="top">When the child has the Nintendo Switch controller which is controlled by 2 hands (ie, it has both the red and the blue parts together or the black controller in their hand).</td><td align="left" valign="top">No matching code</td><td align="left" valign="top">&#x2014;</td></tr><tr><td align="char" char="." valign="top">13</td><td align="left" valign="top">Single controller</td><td align="left" valign="top">When the child has the controller separated and/or in separate hands.</td><td align="left" valign="top">No matching code</td><td align="left" valign="top">&#x2014;</td></tr><tr><td align="char" char="." valign="top">14</td><td align="left" valign="top">Smartwatch</td><td align="left" valign="top">A watch that has an electronic screen with features of a smartphone or a computer. Includes fitness trackers, such as a FitBit.</td><td align="left" valign="top">No matching code</td><td align="left" valign="top">&#x2014;</td></tr><tr><td align="char" char="." valign="top">15</td><td align="left" valign="top">Uncodable: blurry</td><td align="left" valign="top">Any image or set of images where the image quality is so poor due to being blurred that the coder is unable to confidently determine what is occurring in all aspects of the image.</td><td align="left" valign="top">No matching code</td><td align="left" valign="top">&#x2014;</td></tr><tr><td align="char" char="." valign="top">16</td><td align="left" valign="top">Uncodable: obscured images</td><td align="left" valign="top">Any image or set of images where the image quality is so poor due to lighting that the coder is unable to confidently determine what is occurring in all aspects of the image. Poor lighting can include images that are too dark or too overexposed to accurately determine anything.</td><td align="left" valign="top">No matching code</td><td align="left" valign="top">&#x2014;</td></tr><tr><td align="char" char="." valign="top">17</td><td align="left" valign="top">Uncodable: poor lighting</td><td align="left" valign="top">Images that are completely black or fully blocked by something and cannot be coded as having an active screen-based media device according to the coding rules. Includes completely blacked images.</td><td align="left" valign="top">No matching code</td><td align="left" valign="top">&#x2014;</td></tr><tr><td align="char" char="." valign="top">18</td><td align="left" valign="top">Unsure</td><td align="left" valign="top">Coder is uncertain what is in the image.</td><td align="left" valign="top">No matching code</td><td align="left" valign="top"/></tr><tr><td align="char" char="." valign="top">19</td><td align="left" valign="top">No digital technology</td><td align="left" valign="top">No digital technology is visible (or active for room video).</td><td align="left" valign="top">No matching code</td><td align="left" valign="top"/></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>Not applicable.</p></fn></table-wrap-foot></table-wrap><p>A subset of the images from 3 participants (one from each age-group, n=12,024) were repeat coded by a second researcher, and interrater reliability was determined via assessing percentage agreement and Cohen &#x03BA; [<xref ref-type="bibr" rid="ref36">36</xref>]. Specifically, agreement in all coding options (n=14) for the &#x201C;Technology 1&#x201D; category was evaluated. Across all category codes and age-groups, very good interrater reliability was observed, with an average percentage agreement of 92.5% (SD 2.2%, range: 90.3%&#x2010;94.7%) and average &#x03BA; of 0.88 (SD 0.03, range: 0.86&#x2010;0.92).</p></sec><sec id="s2-5-3"><title>The Square Eyes Model</title><p>The Square Eyes model is a computer vision model developed to detect technologies in images captured from wearable cameras [<xref ref-type="bibr" rid="ref27">27</xref>]. The model is a fine-tuning of the YOLO-v8 model developed by Ultralytics [<xref ref-type="bibr" rid="ref37">37</xref>], which builds on the original YOLO architecture [<xref ref-type="bibr" rid="ref38">38</xref>]. Specifically, the model was fine-tuned using a mix of wearable camera data and publicly available images of screen-based technologies to detect the 11 technology categories shown in <xref ref-type="table" rid="table1">Table 1</xref>. When making predictions, the Square Eyes model first removes overlapping duplicate detections using nonmaximum suppression. The remaining detections are then ordered by confidence score. The highest-confidence detection is assigned to &#x201C;Device 1&#x201D; and the second highest to &#x201C;Device 2,&#x201D; with further detections disregarded. For the purposes of visual review, bounding boxes are overlaid on the images based on the coordinates predicted by the model for each detection, indicating the estimated location and size of the detected device in the image.</p><p>The primary aim of the Square Eyes model was to minimize the burden on researchers to code images collected in free-living settings. In general, most images captured in free-living settings do not contain technologies and therefore do not need to be coded, but identifying which images do require coding is time consuming. To minimize false negatives from the model detection, the Square Eyes model is complemented by an N-back algorithm which flags images that are likely to contain a screen-based technology and requiring humans to review. The rationale for this algorithm was that the Square Eyes model was expected to miss some technologies due to, for example, motion blur and odd viewing angles. If a technology is present in 1 image, it is more likely to be present in adjacent images. Thus, the N-back algorithm flags an image as needing human review if it contains a technology or if it is within N chronologically adjacent images of an image that contains a technology. For example, if &#x201C;n=2&#x201D; and image 10 is identified as containing a technology, images number 8, 9, 10, 11, and 12 would be flagged for human review. Images where the model does not detect a technology or is not located within N frames of an image with model-detected technology are not flagged for human review. In the following, the term &#x201C;detected&#x201D; will be used when the Square Eyes model detected a technology on an image either as Device 1 or Device 2. The term &#x201C;flagged&#x201D; will be used to describe images identified by the N-back algorithm, which always include the &#x201C;detected&#x201D; images.</p></sec></sec><sec id="s2-6"><title>Data Analysis</title><sec id="s2-6-1"><title>Statistical Approach and Model Evaluation</title><p>Basic descriptive statistics were used to describe the study sample and the distribution of technologies coded by humans and detected by the Square Eyes model, respectively.</p><p>The performance of the Square Eyes model as a screening tool was evaluated by investigating (1) agreement between the model and human coders for technology detection, (2) evaluation of the N-back algorithm for image flagging, and (3) examination of the potential influence of image blurring on model performance. Each analytical step is described in detail below.</p></sec><sec id="s2-6-2"><title>Evaluation of Technology Detection by the Square Eyes Model</title><p>To evaluate the overall performance of the Square Eyes model as a screening tool for detecting the presence of any technology in all images, agreement between the model and human coding was assessed by calculating the number of true positives, false positives, true negatives, and false negatives using human coding as the reference standard. True positives were defined as images where both the model and human coders detected a technology; false positives were defined as images where only the model detected a technology; true negatives were defined as images where neither the model nor the human coders detected a technology; and false negatives were defined as images where only the humans detected a technology. Moreover, sensitivity (true positives / (true positives + false negatives)), specificity (true negatives / (true negatives + false positives)), and overall accuracy ((true positives + true negatives) / (true positives + true negatives + false positives + false negatives)) were calculated.</p><p>We further evaluated the Square Eyes model performance in detecting different technology types by assessing agreement between model detection and human coding separately for each technology category. To enable this comparison, a 6-category variable (&#x201C;any technology combination&#x201D;) was created based on the presence of specific technologies identified by either the model or human coders. Specifically, this was a nonexclusive variable where each category described when a technology was either detected alone or in combination with another technology. For example, the <italic>Any television</italic> category included images where a television was detected by the model (as Device 1 or Device 2) and/or coded by humans (as Technology 1, 2, or 3), regardless of whether other technologies were also present. This resulted in the following categories: <italic>Any computer, Any gaming, Any laptop, Any smartphone, Any tablet,</italic> and <italic>Any television</italic>. Table S2 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref> provides an overview of how technology codes used by the Square Eyes model and the human coders were matched to create the &#x201C;any technology combination&#x201D; variable. Agreement between the model and human coding was then evaluated separately for each technology category. Percent agreement between the human coders and the Square Eyes model, and prevalence-adjusted bias-adjusted kappa (PABAK) were calculated separately for each category [<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref39">39</xref>]. The PABAK was calculated to account for the potential influence of imbalanced prevalence of some technology types on agreement estimates. The PABAK values were interpreted using the Landis and Koch [<xref ref-type="bibr" rid="ref40">40</xref>] classification scale as below chance agreement &#x003C;0.00, slight agreement 0.00&#x2010;0.20, fair agreement 0.21&#x2010;0.40, moderate agreement 0.41&#x2010;0.60, substantial agreement 0.61&#x2010;0.80, and almost perfect agreement 0.81&#x2010;1.00. Of note, negative PABAK values are more likely to arise in situations with substantial class imbalance and high true negative frequency. In this context, it is encouraged to interpret the PABAK values alongside category-specific percent agreement as the negative values may reflect limited sensitivity for infrequent categories rather than systematic disagreement [<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref39">39</xref>].</p></sec><sec id="s2-6-3"><title>Evaluation of Image Flagging by the Square Eyes Model With the N-Back Algorithm</title><p>To enable evaluation of the N-back algorithm, the number of images flagged and not flagged for human review when &#x201C;n=2&#x201D; was quantified. The frequency and prorata proportion of technology types were calculated among images that were not flagged for human review, but where humans had coded the presence of a technology. This was done to identify which technologies were most likely to be overlooked by the N-back algorithm. The performance of the N-Back algorithm was further evaluated by calculating sensitivity, specificity, and the proportion of images flagged for human review across <italic>N</italic> values ranging from 0 to 10. These metrics were used to investigate the expected trade-off between technology flagging performance and human coding burden as the buffer size (<italic>N</italic>) increased.</p></sec><sec id="s2-6-4"><title>Examination of the Influence of Image Blurring on Technology Detection by the Square Eyes Model</title><p>An explorative assessment was conducted to examine if the Square Eyes model performed differently on images in which faces had been blurred compared to images where faces had not been blurred (ie, nonblurred images) or were blurred in a different manner. For this assessment, images from a single participant were examined to identify discrepancies in technology detection attributable to the blurring process. The blur was originally applied as a square with a border, which may resemble a digital screen. Thus, to explore whether the detection was driven by the presence of the blur border or the shape of the blur, alternative blur styles were tested&#x2014;specifically, a square blur without a border and a circular blur without a border.</p></sec></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Participants and Coding of Images</title><p>Among the 48 included participants, 21 were boys and 27 were girls (overall mean age 8.5, SD 3.3 y; boys: mean age 8.2, SD 3.2 y; girls: mean age 8.7, SD 3.5 y). A total of 221,226 images were coded by humans and processed through the Square Eyes model.</p><p>Humans detected technology in 92,745 (41.9%) images, of which <italic>Television</italic> (n=40,231; 43.4%) and <italic>Laptop</italic> (n=12,717; 13.7%) were the most frequently coded as Technology 1 (Table S3 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>). The Square Eyes model detected technology in 101,075 (45.7%) images, with <italic>Television</italic> (n=23,590; 23.3%) and <italic>Computer monitor</italic> (n=17,766; 17.6%) being the most commonly detected as Device 1 (Table S4 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>).</p></sec><sec id="s3-2"><title>Evaluation of Technology Detection by the Square Eyes Model</title><p>Both humans and the Square Eyes model detected a technology in 72,470 images (ie, true positive rate=32.8%) and no technology in 99,876 images (ie, true negative rate=45.1%). A technology was detected by humans only in 20,275 images (ie, false negative rate=9.2%) and by the Square Eyes model only in 28,605 images (ie, false positive rate=12.9%). Thus, overall sensitivity was 78.1%, specificity was 77.7%, and overall accuracy for detecting technologies was 78.0%.</p><p><xref ref-type="table" rid="table2">Table 2</xref> shows the frequency distribution of the 6 &#x201C;any technology combination&#x201D; categories for images where both the humans and the Square Eyes model had detected a technology (n=74,470). The most frequently coded technologies based on the human coding were <italic>Any television</italic> (n=32,662) and <italic>Any laptop</italic> (n=17,880). The Square Eyes model most frequently detected <italic>Any television</italic> (n=21,762) and <italic>Any computer</italic> (n=34,530). Overall, there was a below chance agreement between the human coding and the model. Mean percentage agreement was 32.8% (14.0%&#x2010;54.3%) and the PABAK score ranged between &#x2013;0.719 and 0.086. The highest level of agreement was observed for <italic>Any television</italic> (n=19,148, 54.3%; PABAK=0.086), while the lowest level of agreement was found for <italic>Any gaming</italic> (n=1013, 14.0%; PABAK=&#x2212;0.719).</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Frequency and agreement of &#x201C;any technology combination&#x201C; based on human coding and Square Eyes model detection.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Technology</td><td align="left" valign="bottom">Human coding, n</td><td align="left" valign="bottom">Square Eyes model, n</td><td align="left" valign="bottom">Total N<sup>*<xref ref-type="table-fn" rid="table2fn1">a</xref></sup></td><td align="left" valign="bottom">Agreement, n (%)</td><td align="left" valign="bottom">PABAK<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup></td></tr></thead><tbody><tr><td align="left" valign="top">Any computer</td><td align="left" valign="top">9989</td><td align="left" valign="top">34,530</td><td align="left" valign="top">34,908</td><td align="left" valign="top">9611 (27.5)</td><td align="left" valign="top">&#x2212;0.449</td></tr><tr><td align="left" valign="top">Any gaming</td><td align="left" valign="top">6911</td><td align="left" valign="top">1320</td><td align="left" valign="top">7218</td><td align="left" valign="top">1013 (14.0)</td><td align="left" valign="top">&#x2212;0.719</td></tr><tr><td align="left" valign="top">Any laptop</td><td align="left" valign="top">17,880</td><td align="left" valign="top">9703</td><td align="left" valign="top">19,091</td><td align="left" valign="top">8492 (44.5)</td><td align="left" valign="top">&#x2212;0.110</td></tr><tr><td align="left" valign="top">Any smartphone</td><td align="left" valign="top">5346</td><td align="left" valign="top">5567</td><td align="left" valign="top">8313</td><td align="left" valign="top">2600 (31.3)</td><td align="left" valign="top">&#x2212;0.374</td></tr><tr><td align="left" valign="top">Any tablet</td><td align="left" valign="top">6419</td><td align="left" valign="top">11,932</td><td align="left" valign="top">14,666</td><td align="left" valign="top">3685 (25.1)</td><td align="left" valign="top">&#x2212;0.497</td></tr><tr><td align="left" valign="top">Any television</td><td align="left" valign="top">32,662</td><td align="left" valign="top">21,762</td><td align="left" valign="top">35,276</td><td align="left" valign="top">19,148 (54.3)</td><td align="left" valign="top">0.086</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>Total N* coded by either by humans and/or detected by the model.</p></fn><fn id="table2fn2"><p><sup>b</sup>PABAK: prevalence-adjusted bias-adjusted kappa. Frequency and PABAK calculations based on images with both human coding and the model detecting a technology (N=74,470).</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-3"><title>Evaluation of Image Flagging by the Square Eyes Model With N-Back Algorithm</title><p>The N-back algorithm (with n=2) flagged 148,546 (67.2%) images that would require human review and 72,680 (32.8%) images were not flagged.</p><p>Among the flagged images, a technology was coded by humans only in 13,131 (8.8%) of the flagged images, and a technology was detected by the model only in 28,605 (19.3%) images. Neither the humans nor the model detected a technology in 34,340 (23.1%) of the flagged images. In 72,470 (48.8%) of the flagged images, both humans and the model detected a technology. When considering the images not flagged for human review, 7144 (9.8%) images had a technology coded by humans only, while 63,536 (90.2%) images did not have a technology coded by humans. None of the images had a technology detected by the model only.</p><p><xref ref-type="table" rid="table3">Table 3</xref> shows the proportion of human-coded technologies (based on Technology 1) that were not flagged for human review by the N-back algorithm. Among the images that were not flagged, the most frequently missed technologies (based on prorata percentages) were <italic>Smartwatch</italic> (n=115, 46.7%), <italic>Single controller</italic> (n=62, 19.8%), and <italic>Tablet</italic> (n=1518, 15.1%).</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Proportion of human-coded technologies on images not flagged for review by the N-back algorithm.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Technology 1 (human-coded)</td><td align="left" valign="bottom">Total images, n</td><td align="left" valign="bottom">Nonflagged images, n (%)</td></tr></thead><tbody><tr><td align="left" valign="top">Combined controller</td><td align="left" valign="top">3675</td><td align="left" valign="top">157 (4.3)</td></tr><tr><td align="left" valign="top">Desktop</td><td align="left" valign="top">10,714</td><td align="left" valign="top">363 (3.4)</td></tr><tr><td align="left" valign="top">Handheld gaming</td><td align="left" valign="top">7678</td><td align="left" valign="top">146 (1.9)</td></tr><tr><td align="left" valign="top">Laptop</td><td align="left" valign="top">12,527</td><td align="left" valign="top">978 (7.8)</td></tr><tr><td align="left" valign="top">Single controller</td><td align="left" valign="top">313</td><td align="left" valign="top">62 (19.8)</td></tr><tr><td align="left" valign="top">Smartphone</td><td align="left" valign="top">6668</td><td align="left" valign="top">458 (6.9)</td></tr><tr><td align="left" valign="top">Smartwatch</td><td align="left" valign="top">246</td><td align="left" valign="top">115 (46.7)</td></tr><tr><td align="left" valign="top">Tablet</td><td align="left" valign="top">10,066</td><td align="left" valign="top">1518 (15.1)</td></tr><tr><td align="left" valign="top">Television</td><td align="left" valign="top">39,145</td><td align="left" valign="top">3346 (8.5)</td></tr></tbody></table></table-wrap><p><xref ref-type="fig" rid="figure1">Figure 1</xref> illustrates the trade-off between sensitivity, specificity, and the proportion of images flagged by the N-back algorithm as the N-back buffer size increases. Specifically, the figure illustrates that as the N-back buffer size increases, sensitivity increases, and the proportion of images flagged for human review increases while specificity decreases.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Effect of N-back buffer size on sensitivity, specificity, and images flagged for human review.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e100559_fig01.png"/></fig></sec><sec id="s3-4"><title>Influence of Face Blurring on Technology Detection by the Square Eyes Model</title><p>Detection performance by the Square Eyes model on blurred and nonblurred versions of 5667 images from a single participant was examined. The model detected technology on slightly more blurred images than on their nonblurred counterparts. Specifically, the model detected a technology on 173 blurred images, which was not detected when the images were nonblurred. In contrast, 60 images were classified as having a technology when nonblurred but not when they were blurred.</p><p><xref ref-type="fig" rid="figure2">Figure 2</xref> represents an example where the model detected a technology only in the blurred version. The bounding box for the model predictions is shown as a blue square, illustrating that the model detected a <italic>Mobile phone</italic>. <xref ref-type="fig" rid="figure3">Figures 3</xref> and <xref ref-type="fig" rid="figure4">4</xref> show examples of the alternative blur styles. When the blur was applied either as a square without border (<xref ref-type="fig" rid="figure3">Figure 3</xref>) or in a circular shape without border (<xref ref-type="fig" rid="figure4">Figure 4</xref>), the model no longer detected technologies in these images. Thus, this suggests that the specific characteristics of the original blur effect influenced the model&#x2019;s misclassification rather than the presence of blurring itself.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Example case of the Square Eyes model detecting a face blur as technology.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e100559_fig02.png"/></fig><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Example case of alternative face blur (square shape without border).</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e100559_fig03.png"/></fig><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Example case of alternative face blur (circular shape).</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e100559_fig04.png"/></fig></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Overall Summary</title><p>This study aimed to evaluate the performance of the Square Eyes model as a screening tool for identifying technologies in wearable camera images among children for further review by humans. The Square Eyes model demonstrated good agreement with human coders in detecting technologies with an overall accuracy of 78.0% and a false negative rate of only 9.2%. However, when considering specific technologies, lower levels of accuracy were found for some devices (eg, <italic>Smartphone</italic> and <italic>Computer</italic>). The model&#x2019;s N-back algorithm effectively flagged images that required further human review. Specifically, of the complete sample of 221,226 images, only 7144 were not flagged for screening and contained a human-coded technology, thereby reducing the false negative rate from 9.2% to 3.2%. Our explorative examination of the potential influence of the image blurring method on the model performance revealed that using a square blurring with border could have reduced the model&#x2019;s ability to accurately detect technologies.</p></sec><sec id="s4-2"><title>Overall Model Performance in Detecting Technologies</title><p>When assessing the Square Eyes model&#x2019;s ability to detect technologies, it should be noted that the model was intended to be used as a screening tool for reducing researchers&#x2019; burden of manually annotating wearable camera images rather than as a replacement for human coding. Nevertheless, the model showed good agreement with human coding, with sensitivity of 78.1%, specificity of 77.7%, and overall accuracy of 78.0%. Thus, the model failed to detect a technology on 9.2% (n=20,275) of all images. Given that each image was taken at a 1-second frequency, this corresponds to approximately 338 minutes of missed technology exposure out of a total of 1546 minutes of technology exposure as identified by humans. Moreover, differences were observed in the performance for detecting specific technologies. We found that the best agreement between the Square Eyes model and human coders was for <italic>Television</italic> (54.3%, n=19,148) and <italic>Laptop</italic> (44.5%, n=8492). In contrast, lower levels of agreement were found for <italic>Smartphone</italic> (31.3%, n=2600), <italic>Computer</italic> (27.5%, n=9611), <italic>Tablet</italic> (25.1%, n=3685), and <italic>Gaming devices</italic> (14.0%, n=1013). Thus, a relatively high proportion of handheld devices was not identified by the model. This finding aligns with known limitations of object recognition models that tend to show lower accuracy in detecting smaller objects compared to larger objects as these are often partially occluded and with limited discriminative visual features [<xref ref-type="bibr" rid="ref38">38</xref>]. The poorer performance for detecting <italic>Computer</italic> by the Square Eyes model may have been a result of overlap with related categories being present in the image (eg, <italic>Laptop</italic>, <italic>Monitor</italic>, or <italic>Keyboard</italic>), which complicates correct classification. The poor detection of particularly handheld devices should be acknowledged as a key limitation of using the Square Eyes model as a standalone tool given that such devices are commonly used among children and adolescents [<xref ref-type="bibr" rid="ref41">41</xref>].</p><p>The observed accuracies of the Square Eyes model are generally lower than what has been reported in studies evaluating the use of similar object recognition models for identifying technologies in wearable camera images. One prior study based on 1029 images from an adult sample reported detection accuracies ranging between 81% and 84% for phones, laptops, and televisions [<xref ref-type="bibr" rid="ref25">25</xref>]. Another adult-based study found a mean accuracy of 85% for identifying phones, laptops, tablets, desktops, computers, and television screens based on 215 images [<xref ref-type="bibr" rid="ref26">26</xref>]. A study based on 1800 images from cameras worn by children aged 3 to 5 years observed a mean accuracy of 93% for identifying computers, smartphones, and televisions [<xref ref-type="bibr" rid="ref19">19</xref>].</p><p>Several methodological factors may explain the lower accuracy of the Square Eyes model found in the present study. First, we evaluated the accuracy of the Square Eyes model based on an independent set of images. In contrast, in all the previous studies, model performance was evaluated based on a subset of images drawn from a sample used for training the models. While this approach is commonly used for model evaluation, it is likely that performance estimates were increased due to similarity between the training and testing datasets. Moreover, the subsets were substantially smaller (n=215&#x2010;1800 images) than the number of images used for evaluating the model in the current study (n=221,226). The smaller samples may have resulted in higher performance estimates as a result of issues such as overfitting [<xref ref-type="bibr" rid="ref42">42</xref>]. The studies by Hou et al [<xref ref-type="bibr" rid="ref19">19</xref>] and Li et al [<xref ref-type="bibr" rid="ref25">25</xref>] excluded images with multiple screens in the same image, whereas multiple devices were present on the images used in the current study. We elected to keep multiple screens as this is more likely what will be found in free-living environments. However, the presence of multiple screens will increase the challenge of object recognition for the model, particularly when screens partially occlude one another. The studies by Li et al [<xref ref-type="bibr" rid="ref25">25</xref>] and Su et al [<xref ref-type="bibr" rid="ref26">26</xref>] were based on adult samples. It can be expected that images from cameras worn by adults are less influenced by motion blur than those from cameras worn by children who typically have more rapid movements than adults. Moreover, the lower height of children may influence the ability of wearable cameras to capture stationary technologies, such as televisions, which may be placed above the children&#x2019;s eye line. Finally, low-quality images affected by motion blur were excluded prior to analysis in the child-based study by Hou et al [<xref ref-type="bibr" rid="ref19">19</xref>], which likely improved model performance. Taken together, differences in training&#x2013;testing dataset overlap, sample size, image complexity, and image quality likely contribute to the lower agreement observed in the present study.</p></sec><sec id="s4-3"><title>The Potential for Reducing Researcher Burden</title><p>The model&#x2019;s N-back algorithm effectively excluded 32.8% (n=72,680) of the images from requiring manual annotation, indicating a substantial reduction in workload. Importantly, of the images not flagged for human review, 90.2% (n=63,536) did not contain a human-coded technology, suggesting that most images were correctly identified as not requiring human review. Across the images not flagged for human review but containing a human-coded technology (n=7144, 3.2%), these images primarily contained smaller devices (eg, smartwatches and smaller gaming devices). Thus, while the algorithm reduced human coding burden by approximately one-third, some smaller technologies were more frequently missed. Nevertheless, our findings suggest that the Square Eyes model could be used as an efficient tool for reducing human coding burden.</p><p>To our knowledge, no prior study has reported an evaluation of the ability of object recognition models to reduce overall researcher burden for processing wearable camera images. Decreasing reliance on intensive manual coding may contribute to greater research workforce sustainability by reducing a highly time-consuming task. Thus, we encourage future research to explore the use of similar hybrid approaches. Such approaches could facilitate the use of object recognition models to identify images likely to contain technologies, followed by human coders to confirm the presence of technologies. Moreover, human coding would remain essential for obtaining information on the content and context of technology use. Additionally, engaging with parents and children in study designs may further facilitate the inclusion of larger and more diverse study populations by supporting feasibility and acceptability of wearable camera research.</p><p>An additional consideration for future research relates to the rapid evolution of wearable camera devices, which include higher quality images and continuous video recording. The Square Eyes model was trained and developed for still-image object recognition and thus, we encourage future work to test the feasibility of model development for video-based analysis. This would likely involve further development of the current YOLOv8-based framework to incorporate temporal information across continuous frames and to optimize performance for small objects such as smartphones and handheld devices. Future advances in AI are likely to enhance model performance and should be evaluated as advanced models become available, building on the knowledge from this study.</p></sec><sec id="s4-4"><title>Potential Impact of Face Blurring</title><p>The primary purpose of obscuring faces using a blurring technique is to ensure third-party privacy [<xref ref-type="bibr" rid="ref29">29</xref>]. However, a potential consequence may be reduced model performance [<xref ref-type="bibr" rid="ref30">30</xref>]. In the current study, we observed that the use of a square-bordered blur was sometimes mistakenly identified as a technology by the Square Eyes model. In contrast, this did not appear to be the case when using either a nonbordered square or nonbordered circular face blur. Thus, the performance of the Square Eyes model for detecting technologies reported above may have been reduced by the facial blur technique applied to the images in this study. To our knowledge, the potential influence of facial blur techniques on the performance of object recognition models used for technology detection in wearable camera images has not been previously examined. However, our findings contradict those of a study assessing the effect of face blurring on the performance of a deep-learning model used for human pose estimation [<xref ref-type="bibr" rid="ref43">43</xref>]. The authors concluded that face blurring had minimal impact on model performance when the model was trained using face-blurred images. However, this model was designed for whole-body posture recognition, not for the detection of technologies. As ensuring privacy is important in wearable camera research, we encourage future work to evaluate how different face-blurring techniques may influence the performance of machine learning models used for detecting technologies. This may involve training models directly on face-blurred images and comparing different blurring features (eg, circular vs square) across diverse real-world conditions to support the development of standardized ethical pipelines for analyzing wearable camera images.</p></sec><sec id="s4-5"><title>Strengths and Limitations</title><p>A strength of this study was the large data sample of images captured by wearable cameras worn by children aged 3 to 14 years. Few studies have evaluated the use of wearable cameras and machine learning models for technology identification among children, and thus our study substantially adds to an underinvestigated research area. Moreover, images captured among children are likely to have lower visual quality due to increased motion blur compared with images captured among adults. Thus, evaluating the Square Eyes model using a visually challenging dataset provided a rigorous test of model robustness. This was further supported by using images that were not part of the dataset used for training the model. Specifically, models often demonstrate reduced performance when applied to datasets from other samples and settings. Our results, therefore, support the external validity of the Square Eyes model. Nevertheless, some limitations to the dataset should be acknowledged. This study was based on images obtained in a laboratory, which limits generalizability to free-living settings. Specifically, in free-living settings, the quality of images obtained with wearable cameras is likely influenced by substantial variability in lighting, postures, and movements, and thus image quality. Moreover, patterns of technology use in a laboratory setting may not reflect real-world settings where children are likely to engage with multiple technologies across varied physical environments and social settings. Such factors are likely to reduce detection performance of object recognition models. Thus, the findings should be interpreted as a proof-of-concept under controlled conditions rather than estimates of real-world applicability. Given the high burden of human coding, only one coder was used. Although coder training and initial review procedures were implemented to ensure high-quality coding, reliability was not assessed across a larger subset of the dataset. Therefore, we cannot assume that the human-coded reference was free from misclassifications. The potential influence of the facial blurring technique on model performance was evaluated using a small sample of images from a single participant. This analysis was explorative, and it was beyond the scope of the study to estimate performance differences between facial blur methods for the whole sample. Therefore, given the limited sample size and lack of systematic comparison, these findings regarding the impact of facial blur methods should be interpreted with caution.</p></sec><sec id="s4-6"><title>Conclusions</title><p>We found that the Square Eyes model could detect technologies with overall satisfying accuracy and successfully flag images that required further review by humans. These findings suggest that the model can be used as an efficient screening tool to reduce human coding burden for researchers processing wearable camera images of children. However, the model could be improved for accurately detecting smaller devices, such as smartphones. Moreover, we found that the method used for facial blurring might influence model performance. We encourage further research to evaluate the use of similar hybrid approaches that use machine learning models to reduce human coding burden for both technology identification and other important aspects of technology use, such as context and content.</p></sec></sec></body><back><ack><p>The authors would like to thank the participating children and their caregivers for their time in participating in this study and acknowledge the valuable contributions of the Healthy Digital Child Parent Advisory Group and the Expert Advisory Group made to the study design and interpretation. No AI tool was used at any stage of preparation of this manuscript.</p></ack><notes><sec><title>Funding</title><p>This study was partly funded by the Australian Research Council through the ARC Centre of Excellence for the Digital Child, grant number CE200100022, and the Curtin School of Allied Health 2022 Teaching and Research Grant.</p></sec><sec><title>Data Availability</title><p>The dataset analyzed during this study is available from the corresponding author upon reasonable request.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: LS (lead), DH (equal), AC (equal)</p><p>Data curation: CLR</p><p>Formal analysis: CLR (lead), TS (supporting), AB (supporting)</p><p>Funding acquisition: LS (lead)</p><p>Investigation: DH (lead)</p><p>Methodology: CLR (lead), TS (equal), AB (supporting)</p><p>Project administration: LS (lead)</p><p>Resources: LS (lead)</p><p>Software: TS (lead)</p><p>Supervision: LS (lead), AC (supporting), JZ (supporting)</p><p>Validation: CLR (lead), TS (supporting)</p><p>Visualization: CLR (lead), TS (equal)</p><p>Writing &#x2013; original draft: CLR (lead), TS (supporting), EKH (supporting), LS (supporting), AC (supporting)</p><p>Writing &#x2013; review &#x0026; editing: CLR (lead), TS (supporting), EKH (supporting), AB (supporting), DH (supporting), JZ (supporting), AC (supporting), LS (supporting)</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">PABAK</term><def><p>prevalence-adjusted bias-adjusted kappa</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Thomas</surname><given-names>G</given-names> </name><name name-style="western"><surname>Bennie</surname><given-names>JA</given-names> </name><name name-style="western"><surname>De Cocker</surname><given-names>K</given-names> </name><name name-style="western"><surname>Castro</surname><given-names>O</given-names> </name><name name-style="western"><surname>Biddle</surname><given-names>SJH</given-names> </name></person-group><article-title>A descriptive epidemiology of screen-based devices by children and adolescents: a scoping review of 130 surveillance studies since 2000</article-title><source>Child Ind Res</source><year>2020</year><month>06</month><volume>13</volume><issue>3</issue><fpage>935</fpage><lpage>950</lpage><pub-id pub-id-type="doi">10.1007/s12187-019-09663-1</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mannell</surname><given-names>K</given-names> </name><name name-style="western"><surname>Bloul</surname><given-names>S</given-names> </name><name name-style="western"><surname>Sefton-Green</surname><given-names>J</given-names> </name><name name-style="western"><surname>Willcox</surname><given-names>M</given-names> </name></person-group><article-title>Digital media and technology use by families with infants, toddlers, and young children: a scoping review and call for forward momentum</article-title><source>J Child Media</source><year>2024</year><month>10</month><volume>18</volume><issue>4</issue><fpage>605</fpage><lpage>628</lpage><pub-id pub-id-type="doi">10.1080/17482798.2024.2394939</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="report"><person-group person-group-type="author"><name name-style="western"><surname>Mann</surname><given-names>SC</given-names> </name><name name-style="western"><surname>Lenhart</surname><given-names>A</given-names> </name><name name-style="western"><surname>Robb</surname><given-names>MB</given-names> </name></person-group><article-title>The common sense census: media use by kids age zero to eight</article-title><year>2025</year><access-date>2026-09-03</access-date><publisher-name>Common Sense Media</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://www.commonsensemedia.org/research/the-2025-common-sense-census-media-use-by-kids-zero-to-eight">https://www.commonsensemedia.org/research/the-2025-common-sense-census-media-use-by-kids-zero-to-eight</ext-link></comment></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>C</given-names> </name><name name-style="western"><surname>Cheng</surname><given-names>G</given-names> </name><name name-style="western"><surname>Sha</surname><given-names>T</given-names> </name><name name-style="western"><surname>Cheng</surname><given-names>W</given-names> </name><name name-style="western"><surname>Yan</surname><given-names>Y</given-names> </name></person-group><article-title>The relationships between screen use and health indicators among infants, toddlers, and preschoolers: a meta-analysis and systematic review</article-title><source>Int J Environ Res Public Health</source><year>2020</year><month>10</month><day>7</day><volume>17</volume><issue>19</issue><fpage>7324</fpage><pub-id pub-id-type="doi">10.3390/ijerph17197324</pub-id><pub-id pub-id-type="medline">33036443</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Saunders</surname><given-names>TJ</given-names> </name><name name-style="western"><surname>Vallance</surname><given-names>JK</given-names> </name></person-group><article-title>Screen time and health indicators among children and youth: current evidence, limitations and future directions</article-title><source>Appl Health Econ Health Policy</source><year>2017</year><month>06</month><volume>15</volume><issue>3</issue><fpage>323</fpage><lpage>331</lpage><pub-id pub-id-type="doi">10.1007/s40258-016-0289-3</pub-id><pub-id pub-id-type="medline">27798796</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kostyrka-Allchorne</surname><given-names>K</given-names> </name><name name-style="western"><surname>Cooper</surname><given-names>NR</given-names> </name><name name-style="western"><surname>Simpson</surname><given-names>A</given-names> </name></person-group><article-title>The relationship between television exposure and children&#x2019;s cognition and behaviour: a systematic review</article-title><source>Dev Rev</source><year>2017</year><month>06</month><volume>44</volume><fpage>19</fpage><lpage>58</lpage><pub-id pub-id-type="doi">10.1016/j.dr.2016.12.002</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fitzpatrick</surname><given-names>C</given-names> </name><name name-style="western"><surname>Binet</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Cristini</surname><given-names>E</given-names> </name><name name-style="western"><surname>Almeida</surname><given-names>ML</given-names> </name><name name-style="western"><surname>B&#x00E9;gin</surname><given-names>M</given-names> </name><name name-style="western"><surname>Frizzo</surname><given-names>GB</given-names> </name></person-group><article-title>Reducing harm and promoting positive media use strategies: new perspectives in understanding the impact of preschooler media use on health and development</article-title><source>Psicol Refl Cr&#x00ED;t</source><year>2023</year><month>08</month><day>9</day><volume>36</volume><issue>1</issue><pub-id pub-id-type="doi">10.1186/s41155-023-00262-2</pub-id><pub-id pub-id-type="medline">37553485</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sanders</surname><given-names>T</given-names> </name><name name-style="western"><surname>Noetel</surname><given-names>M</given-names> </name><name name-style="western"><surname>Parker</surname><given-names>P</given-names> </name><etal/></person-group><article-title>An umbrella review of the benefits and risks associated with youths&#x2019; interactions with electronic screens</article-title><source>Nat Hum Behav</source><year>2024</year><month>01</month><volume>8</volume><issue>1</issue><fpage>82</fpage><lpage>99</lpage><pub-id pub-id-type="doi">10.1038/s41562-023-01712-8</pub-id><pub-id pub-id-type="medline">37957284</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mallawaarachchi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Burley</surname><given-names>J</given-names> </name><name name-style="western"><surname>Mavilidi</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Early childhood screen use contexts and cognitive and psychosocial outcomes: a systematic review and meta-analysis</article-title><source>JAMA Pediatr</source><year>2024</year><month>10</month><day>1</day><volume>178</volume><issue>10</issue><fpage>1017</fpage><lpage>1026</lpage><pub-id pub-id-type="doi">10.1001/jamapediatrics.2024.2620</pub-id><pub-id pub-id-type="medline">39102255</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Beynon</surname><given-names>A</given-names> </name><name name-style="western"><surname>Hendry</surname><given-names>D</given-names> </name><name name-style="western"><surname>Lund Rasmussen</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Measurement method options to investigate digital screen technology use by children and adolescents: a narrative review</article-title><source>Children (Basel)</source><year>2024</year><month>06</month><day>21</day><volume>11</volume><issue>7</issue><fpage>754</fpage><pub-id pub-id-type="doi">10.3390/children11070754</pub-id><pub-id pub-id-type="medline">39062204</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Parry</surname><given-names>DA</given-names> </name><name name-style="western"><surname>Davidson</surname><given-names>BI</given-names> </name><name name-style="western"><surname>Sewall</surname><given-names>CJR</given-names> </name><name name-style="western"><surname>Fisher</surname><given-names>JT</given-names> </name><name name-style="western"><surname>Mieczkowski</surname><given-names>H</given-names> </name><name name-style="western"><surname>Quintana</surname><given-names>DS</given-names> </name></person-group><article-title>A systematic review and meta-analysis of discrepancies between logged and self-reported digital media use</article-title><source>Nat Hum Behav</source><year>2021</year><month>11</month><volume>5</volume><issue>11</issue><fpage>1535</fpage><lpage>1547</lpage><pub-id pub-id-type="doi">10.1038/s41562-021-01117-5</pub-id><pub-id pub-id-type="medline">34002052</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Doherty</surname><given-names>AR</given-names> </name><name name-style="western"><surname>Hodges</surname><given-names>SE</given-names> </name><name name-style="western"><surname>King</surname><given-names>AC</given-names> </name><etal/></person-group><article-title>Wearable cameras in health: the state of the art and future possibilities</article-title><source>Am J Prev Med</source><year>2013</year><month>03</month><volume>44</volume><issue>3</issue><fpage>320</fpage><lpage>323</lpage><pub-id pub-id-type="doi">10.1016/j.amepre.2012.11.008</pub-id><pub-id pub-id-type="medline">23415132</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Beltran</surname><given-names>A</given-names> </name><name name-style="western"><surname>Dadabhoy</surname><given-names>H</given-names> </name><name name-style="western"><surname>Ryan</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Dietary assessment with a wearable camera among children: feasibility and intercoder reliability</article-title><source>J Acad Nutr Diet</source><year>2018</year><month>11</month><volume>118</volume><issue>11</issue><fpage>2144</fpage><lpage>2153</lpage><pub-id pub-id-type="doi">10.1016/j.jand.2018.05.013</pub-id><pub-id pub-id-type="medline">30115556</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Signal</surname><given-names>LN</given-names> </name><name name-style="western"><surname>Stanley</surname><given-names>J</given-names> </name><name name-style="western"><surname>Smith</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Children&#x2019;s everyday exposure to food marketing: an objective analysis using wearable cameras</article-title><source>Int J Behav Nutr Phys Act</source><year>2017</year><month>10</month><day>8</day><volume>14</volume><issue>1</issue><fpage>137</fpage><pub-id pub-id-type="doi">10.1186/s12966-017-0570-3</pub-id><pub-id pub-id-type="medline">28988542</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chambers</surname><given-names>T</given-names> </name><name name-style="western"><surname>Pearson</surname><given-names>AL</given-names> </name><name name-style="western"><surname>Stanley</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Children&#x2019;s exposure to alcohol marketing within supermarkets: an objective analysis using GPS technology and wearable cameras</article-title><source>Health Place</source><year>2017</year><month>07</month><volume>46</volume><fpage>274</fpage><lpage>280</lpage><pub-id pub-id-type="doi">10.1016/j.healthplace.2017.06.003</pub-id><pub-id pub-id-type="medline">28672147</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gurtner</surname><given-names>M</given-names> </name><name name-style="western"><surname>Gage</surname><given-names>R</given-names> </name><name name-style="western"><surname>Thomson</surname><given-names>G</given-names> </name><etal/></person-group><article-title>Are children smoke-free at home? Using wearable cameras to study children&#x2019;s exposure to smoking and smoking paraphernalia in private spaces</article-title><source>Child Care Health Dev</source><year>2019</year><month>03</month><volume>45</volume><issue>2</issue><fpage>306</fpage><lpage>309</lpage><pub-id pub-id-type="doi">10.1111/cch.12631</pub-id><pub-id pub-id-type="medline">30548456</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gage</surname><given-names>R</given-names> </name><name name-style="western"><surname>Gurtner</surname><given-names>M</given-names> </name><name name-style="western"><surname>Keall</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Fun, food and friends: a wearable camera analysis of children&#x2019;s school journeys</article-title><source>J Transp Health</source><year>2023</year><month>05</month><volume>30</volume><fpage>101604</fpage><pub-id pub-id-type="doi">10.1016/j.jth.2023.101604</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>KX</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>YX</given-names> </name><name name-style="western"><surname>Yan</surname><given-names>SY</given-names> </name><name name-style="western"><surname>Hebestreit</surname><given-names>A</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>HJ</given-names> </name></person-group><article-title>Wearable camera-based objective screen time and its combined associations with dietary and physical activity factors in relation to childhood obesity</article-title><source>Nutrients</source><year>2025</year><volume>17</volume><issue>18</issue><fpage>2990</fpage><pub-id pub-id-type="doi">10.3390/nu17182990</pub-id><pub-id pub-id-type="medline">41010518</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Hou</surname><given-names>X</given-names> </name><name name-style="western"><surname>Shen</surname><given-names>S</given-names> </name><name name-style="western"><surname>Li</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Enhancing screen time identification in children with a multi-view vision language model and screen time tracker</article-title><source>arXiv</source><comment>Preprint posted online on  Oct 2, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2410.01966</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lowe</surname><given-names>BM</given-names> </name><name name-style="western"><surname>Smith</surname><given-names>M</given-names> </name><name name-style="western"><surname>Jaine</surname><given-names>R</given-names> </name><name name-style="western"><surname>Stanley</surname><given-names>J</given-names> </name><name name-style="western"><surname>Gage</surname><given-names>R</given-names> </name><name name-style="western"><surname>Signal</surname><given-names>L</given-names> </name></person-group><article-title>Watching the watchers: assessing the nature and extent of children&#x2019;s screen time using wearable cameras</article-title><source>N Z Med J</source><year>2023</year><month>07</month><day>7</day><volume>136</volume><issue>1578</issue><fpage>12</fpage><lpage>31</lpage><pub-id pub-id-type="doi">10.26635/6965.6023</pub-id><pub-id pub-id-type="medline">37414074</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Thomas</surname><given-names>G</given-names> </name><name name-style="western"><surname>Bennie</surname><given-names>JA</given-names> </name><name name-style="western"><surname>De Cocker</surname><given-names>K</given-names> </name><name name-style="western"><surname>Dwi Andriyani</surname><given-names>F</given-names> </name><name name-style="western"><surname>Booker</surname><given-names>B</given-names> </name><name name-style="western"><surname>Biddle</surname><given-names>SJH</given-names> </name></person-group><article-title>Using wearable cameras to categorize the type and context of screen-based behaviors among adolescents: observational study</article-title><source>JMIR Pediatr Parent</source><year>2022</year><month>03</month><day>21</day><volume>5</volume><issue>1</issue><fpage>e28208</fpage><pub-id pub-id-type="doi">10.2196/28208</pub-id><pub-id pub-id-type="medline">35311672</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Smith</surname><given-names>C</given-names> </name><name name-style="western"><surname>Galland</surname><given-names>BC</given-names> </name><name name-style="western"><surname>de Bruin</surname><given-names>WE</given-names> </name><name name-style="western"><surname>Taylor</surname><given-names>RW</given-names> </name></person-group><article-title>Feasibility of automated cameras to measure screen use in adolescents</article-title><source>Am J Prev Med</source><year>2019</year><month>09</month><volume>57</volume><issue>3</issue><fpage>417</fpage><lpage>424</lpage><pub-id pub-id-type="doi">10.1016/j.amepre.2019.04.012</pub-id><pub-id pub-id-type="medline">31377085</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Brosnan</surname><given-names>BJ</given-names> </name><name name-style="western"><surname>Wickham</surname><given-names>SR</given-names> </name><name name-style="western"><surname>Meredith-Jones</surname><given-names>KA</given-names> </name><name name-style="western"><surname>Galland</surname><given-names>BC</given-names> </name><name name-style="western"><surname>Haszard</surname><given-names>JJ</given-names> </name><name name-style="western"><surname>Taylor</surname><given-names>RW</given-names> </name></person-group><article-title>Development of a protocol for objectively measuring digital device use in youth</article-title><source>Am J Prev Med</source><year>2023</year><month>11</month><volume>65</volume><issue>5</issue><fpage>923</fpage><lpage>931</lpage><pub-id pub-id-type="doi">10.1016/j.amepre.2023.05.004</pub-id><pub-id pub-id-type="medline">37156402</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Guleria</surname><given-names>A</given-names> </name><name name-style="western"><surname>Varshney</surname><given-names>K</given-names> </name><name name-style="western"><surname>Jindal</surname><given-names>S</given-names> </name></person-group><article-title>A systematic review: object detection</article-title><source>AI Soc</source><year>2025</year><month>12</month><volume>40</volume><issue>8</issue><fpage>6511</fpage><lpage>6528</lpage><pub-id pub-id-type="doi">10.1007/s00146-025-02372-0</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>X</given-names> </name><name name-style="western"><surname>Holiday</surname><given-names>S</given-names> </name><name name-style="western"><surname>Cribbet</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Non-invasive screen exposure time assessment using wearable sensor and object detection</article-title><conf-name>2022 44th Annual International Conference of the IEEE Engineering in Medicine &#x0026; Biology Society (EMBC)</conf-name><conf-date>Jul 11-15, 2022</conf-date><pub-id pub-id-type="doi">10.1109/EMBC48229.2022.9871903</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Su</surname><given-names>N</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>J</given-names> </name><name name-style="western"><surname>Li</surname><given-names>W</given-names> </name><name name-style="western"><surname>Long</surname><given-names>Y</given-names> </name></person-group><article-title>Assessing personal screen exposure with ever-changing contexts using wearable cameras and computer vision</article-title><source>Build Environ</source><year>2024</year><month>08</month><volume>261</volume><fpage>111720</fpage><pub-id pub-id-type="doi">10.1016/j.buildenv.2024.111720</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Sanders</surname><given-names>T</given-names> </name></person-group><article-title>Motivation-and-behaviour/squareeyes</article-title><source>GitHub</source><year>2025</year><access-date>2026-09-07</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/Motivation-and-Behaviour/SquareEyes">https://github.com/Motivation-and-Behaviour/SquareEyes</ext-link></comment></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="web"><article-title>DP200101912 &#x2014; Australian Catholic University</article-title><source>Australian Research Council</source><year>2020</year><access-date>2026-09-03</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://dataportal.arc.gov.au/NCGP/Web/Grant/Grant/DP200101912">https://dataportal.arc.gov.au/NCGP/Web/Grant/Grant/DP200101912</ext-link></comment></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kelly</surname><given-names>P</given-names> </name><name name-style="western"><surname>Marshall</surname><given-names>SJ</given-names> </name><name name-style="western"><surname>Badland</surname><given-names>H</given-names> </name><etal/></person-group><article-title>An ethical framework for automated, wearable cameras in health behavior research</article-title><source>Am J Prev Med</source><year>2013</year><month>03</month><volume>44</volume><issue>3</issue><fpage>314</fpage><lpage>319</lpage><pub-id pub-id-type="doi">10.1016/j.amepre.2012.11.006</pub-id><pub-id pub-id-type="medline">23415131</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>L&#x00E9;bl</surname><given-names>M</given-names> </name><name name-style="western"><surname>&#x0160;roubek</surname><given-names>F</given-names> </name><name name-style="western"><surname>Flusser</surname><given-names>J</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Gade</surname><given-names>R</given-names> </name><name name-style="western"><surname>Felsberg</surname><given-names>M</given-names> </name><name name-style="western"><surname>K&#x00E4;m&#x00E4;r&#x00E4;inen</surname><given-names>JK</given-names> </name></person-group><article-title>Impact of image blur on classification and augmentation of deep convolutional networks</article-title><source>Image Analysis: 22nd Scandinavian Conference, SCIA 2023, Sirkka, Finland, April 18&#x2013;21, 2023, Proceedings, Part II</source><year>2023</year><publisher-name>Springer</publisher-name><fpage>108</fpage><lpage>117</lpage><pub-id pub-id-type="doi">10.1007/978-3-031-31438-4_8</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Carson</surname><given-names>V</given-names> </name><name name-style="western"><surname>Salmon</surname><given-names>J</given-names> </name><name name-style="western"><surname>Crawford</surname><given-names>D</given-names> </name><name name-style="western"><surname>Hinkley</surname><given-names>T</given-names> </name><name name-style="western"><surname>Hesketh</surname><given-names>KD</given-names> </name></person-group><article-title>Longitudinal levels and bouts of objectively measured sedentary time among young Australian children in the HAPPY study</article-title><source>J Sci Med Sport</source><year>2016</year><month>03</month><volume>19</volume><issue>3</issue><fpage>232</fpage><lpage>236</lpage><pub-id pub-id-type="doi">10.1016/j.jsams.2015.01.009</pub-id><pub-id pub-id-type="medline">25683731</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pearson</surname><given-names>N</given-names> </name><name name-style="western"><surname>Haycraft</surname><given-names>E</given-names> </name><name name-style="western"><surname>Johnston</surname><given-names>JP</given-names> </name><name name-style="western"><surname>Atkin</surname><given-names>AJ</given-names> </name></person-group><article-title>Sedentary behaviour across the primary-secondary school transition: a systematic review</article-title><source>Prev Med</source><year>2017</year><month>01</month><volume>94</volume><issue>40-7</issue><fpage>40</fpage><lpage>47</lpage><pub-id pub-id-type="doi">10.1016/j.ypmed.2016.11.010</pub-id><pub-id pub-id-type="medline">27856338</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Marks</surname><given-names>J</given-names> </name><name name-style="western"><surname>Barnett</surname><given-names>LM</given-names> </name><name name-style="western"><surname>Strugnell</surname><given-names>C</given-names> </name><name name-style="western"><surname>Allender</surname><given-names>S</given-names> </name></person-group><article-title>Changing from primary to secondary school highlights opportunities for school environment interventions aiming to increase physical activity and reduce sedentary behaviour: a longitudinal cohort study</article-title><source>Int J Behav Nutr Phys Act</source><year>2015</year><month>05</month><day>8</day><volume>12</volume><issue>59</issue><fpage>59</fpage><pub-id pub-id-type="doi">10.1186/s12966-015-0218-0</pub-id><pub-id pub-id-type="medline">25952318</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>McArthur</surname><given-names>BA</given-names> </name><name name-style="western"><surname>Browne</surname><given-names>D</given-names> </name><name name-style="western"><surname>Tough</surname><given-names>S</given-names> </name><name name-style="western"><surname>Madigan</surname><given-names>S</given-names> </name></person-group><article-title>Trajectories of screen use during early childhood: predictors and associated behavior and learning outcomes</article-title><source>Comput Human Behav</source><year>2020</year><month>12</month><volume>113</volume><fpage>106501</fpage><pub-id pub-id-type="doi">10.1016/j.chb.2020.106501</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kerr</surname><given-names>J</given-names> </name><name name-style="western"><surname>Marshall</surname><given-names>SJ</given-names> </name><name name-style="western"><surname>Godbole</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Using the SenseCam to improve classifications of sedentary behavior in free-living settings</article-title><source>Am J Prev Med</source><year>2013</year><month>03</month><volume>44</volume><issue>3</issue><fpage>290</fpage><lpage>296</lpage><pub-id pub-id-type="doi">10.1016/j.amepre.2012.11.004</pub-id><pub-id pub-id-type="medline">23415127</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sim</surname><given-names>J</given-names> </name><name name-style="western"><surname>Wright</surname><given-names>CC</given-names> </name></person-group><article-title>The kappa statistic in reliability studies: use, interpretation, and sample size requirements</article-title><source>Phys Ther</source><year>2005</year><month>03</month><volume>85</volume><issue>3</issue><fpage>257</fpage><lpage>268</lpage><pub-id pub-id-type="doi">10.1093/ptj/85.3.257</pub-id><pub-id pub-id-type="medline">15733050</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Jocher</surname><given-names>G</given-names> </name><name name-style="western"><surname>Chaurasia</surname><given-names>A</given-names> </name><name name-style="western"><surname>Jing</surname><given-names>Q</given-names> </name></person-group><article-title>Ultralytics yolov8. 8.0.0 ed</article-title><source>Ultralytics</source><year>2023</year><access-date>2026-09-03</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://docs.ultralytics.com/models/yolov8">https://docs.ultralytics.com/models/yolov8</ext-link></comment></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Redmon</surname><given-names>J</given-names> </name><name name-style="western"><surname>Divvala</surname><given-names>S</given-names> </name><name name-style="western"><surname>Girshick</surname><given-names>R</given-names> </name><name name-style="western"><surname>Farhadi</surname><given-names>A</given-names> </name></person-group><article-title>You only look once: unified, real-time object detection</article-title><conf-name>2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name><conf-date>Jun 27-30, 2016</conf-date><pub-id pub-id-type="doi">10.1109/CVPR.2016.91</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Byrt</surname><given-names>T</given-names> </name><name name-style="western"><surname>Bishop</surname><given-names>J</given-names> </name><name name-style="western"><surname>Carlin</surname><given-names>JB</given-names> </name></person-group><article-title>Bias, prevalence and kappa</article-title><source>J Clin Epidemiol</source><year>1993</year><month>05</month><volume>46</volume><issue>5</issue><fpage>423</fpage><lpage>429</lpage><pub-id pub-id-type="doi">10.1016/0895-4356(93)90018-v</pub-id><pub-id pub-id-type="medline">8501467</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Landis</surname><given-names>JR</given-names> </name><name name-style="western"><surname>Koch</surname><given-names>GG</given-names> </name></person-group><article-title>The measurement of observer agreement for categorical data</article-title><source>Biometrics</source><year>1977</year><month>03</month><volume>33</volume><issue>1</issue><fpage>159</fpage><lpage>174</lpage><pub-id pub-id-type="doi">10.2307/2529310</pub-id><pub-id pub-id-type="medline">843571</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Straker</surname><given-names>L</given-names> </name><name name-style="western"><surname>Harris</surname><given-names>C</given-names> </name><name name-style="western"><surname>Joosten</surname><given-names>J</given-names> </name><name name-style="western"><surname>Howie</surname><given-names>EK</given-names> </name></person-group><article-title>Mobile technology dominates school children&#x2019;s IT use in an advantaged school community and is associated with musculoskeletal and visual symptoms</article-title><source>Ergonomics</source><year>2018</year><month>05</month><volume>61</volume><issue>5</issue><fpage>658</fpage><lpage>669</lpage><pub-id pub-id-type="doi">10.1080/00140139.2017.1401671</pub-id><pub-id pub-id-type="medline">29103354</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rajput</surname><given-names>D</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>WJ</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>CC</given-names> </name></person-group><article-title>Evaluation of a decided sample size in machine learning applications</article-title><source>BMC Bioinformatics</source><year>2023</year><month>02</month><day>14</day><volume>24</volume><issue>1</issue><fpage>48</fpage><pub-id pub-id-type="doi">10.1186/s12859-023-05156-9</pub-id><pub-id pub-id-type="medline">36788550</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jiang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Skalli</surname><given-names>W</given-names> </name><name name-style="western"><surname>Siadat</surname><given-names>A</given-names> </name><name name-style="western"><surname>Gajny</surname><given-names>L</given-names> </name></person-group><article-title>Effect of face blurring on human pose estimation: ensuring subject privacy for medical and occupational health applications</article-title><source>Sensors (Basel)</source><year>2022</year><month>12</month><day>1</day><volume>22</volume><issue>23</issue><fpage>9376</fpage><pub-id pub-id-type="doi">10.3390/s22239376</pub-id><pub-id pub-id-type="medline">36502076</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Overview of tasks completed by children during the data collection session.</p><media xlink:href="jmir_v28i1e100559_app1.docx" xlink:title="DOCX File, 23 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Overview of matching between technology codes used by the human coders and the Square Eyes model.</p><media xlink:href="jmir_v28i1e100559_app2.docx" xlink:title="DOCX File, 21 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>Frequency of technologies as coded by humans and detected by the Square Eyes model.</p><media xlink:href="jmir_v28i1e100559_app3.docx" xlink:title="DOCX File, 24 KB"/></supplementary-material></app-group></back></article>