<?xml version="1.0" encoding="UTF-8"?><article xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="pmc-domain-id">1660</journal-id><journal-id journal-id-type="pmc-domain">sensors</journal-id><journal-title-group><journal-title>Sensors (Basel, Switzerland)</journal-title><abbrev-journal-title>Sensors (Basel)</abbrev-journal-title></journal-title-group><publisher><publisher-name>Multidisciplinary Digital Publishing Institute (MDPI)</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="pmcid">PMC11548613</article-id><article-id pub-id-type="pmcaid">11548613</article-id><article-id pub-id-type="pmcaiid">11548613</article-id><article-id pub-id-type="pmid">39517671</article-id><article-id pub-id-type="doi">10.3390/s24216774</article-id><title-group><article-title>Enhancing Grapevine Node Detection to Support Pruning Automation: Leveraging State-of-the-Art YOLO Detection Models for 2D Image Analysis</article-title></title-group><contrib-group content-type="author"><contrib><name name-style="western"><surname>Oliveira</surname><given-names initials="F">Francisco</given-names></name><role>Conceptualization, Methodology, Software, Investigation, Data curation, Writing – original draft, Visualization</role><xref ref-type="aff" rid="af1-sensors-24-06774">1</xref><xref ref-type="aff" rid="af2-sensors-24-06774">2</xref><xref rid="c1-sensors-24-06774" ref-type="author-notes">*</xref></contrib><contrib><name name-style="western"><surname>da Silva</surname><given-names initials="DQ">Daniel Queirós</given-names></name><role>Conceptualization, Methodology, Software, Investigation, Data curation, Writing – review &amp; editing</role><xref ref-type="aff" rid="af2-sensors-24-06774">2</xref></contrib><contrib><name name-style="western"><surname>Filipe</surname><given-names initials="V">Vítor</given-names></name><role>Conceptualization, Validation, Writing – review &amp; editing</role><xref ref-type="aff" rid="af1-sensors-24-06774">1</xref><xref ref-type="aff" rid="af2-sensors-24-06774">2</xref></contrib><contrib><name name-style="western"><surname>Pinho</surname><given-names initials="TM">Tatiana Martins</given-names></name><role>Validation, Writing – review &amp; editing, Supervision</role><xref ref-type="aff" rid="af2-sensors-24-06774">2</xref></contrib><contrib><name name-style="western"><surname>Cunha</surname><given-names initials="M">Mário</given-names></name><role>Validation, Writing – review &amp; editing</role><xref ref-type="aff" rid="af2-sensors-24-06774">2</xref><xref ref-type="aff" rid="af3-sensors-24-06774">3</xref></contrib><contrib><name name-style="western"><surname>Cunha</surname><given-names initials="JB">José Boaventura</given-names></name><role>Validation, Writing – review &amp; editing, Supervision</role><xref ref-type="aff" rid="af1-sensors-24-06774">1</xref><xref ref-type="aff" rid="af2-sensors-24-06774">2</xref></contrib><contrib><name name-style="western"><surname>dos Santos</surname><given-names initials="FN">Filipe Neves</given-names></name><role>Validation, Writing – review &amp; editing, Supervision, Project administration, Funding acquisition</role><xref ref-type="aff" rid="af2-sensors-24-06774">2</xref></contrib></contrib-group><contrib-group content-type="editor"><contrib><name name-style="western"><surname>Yoon</surname><given-names initials="SC">Seung-Chul</given-names></name><role>Academic Editor</role></contrib><contrib><name name-style="western"><surname>Wang</surname><given-names initials="W">Wei</given-names></name><role>Academic Editor</role></contrib><contrib><name name-style="western"><surname>Chu</surname><given-names initials="X">Xuan</given-names></name><role>Academic Editor</role></contrib><contrib><name name-style="western"><surname>Sun</surname><given-names initials="H">Hongwei</given-names></name><role>Academic Editor</role></contrib></contrib-group><aff id="af1-sensors-24-06774"><label>1</label>School of Science and Technology, University of Trás-os-Montes and Alto Douro (UTAD), 5000-801 Vila Real, Portugal; vfilipe@utad.pt (V.F.); jboavent@utad.pt (J.B.C.)</aff><aff id="af2-sensors-24-06774"><label>2</label>INESC Technology and Science (INESC TEC), 4200-465 Porto, Portugal; daniel.q.silva@inesctec.pt (D.Q.d.S.); tatiana.m.pinho@inesctec.pt (T.M.P.); mccunha@fc.up.pt (M.C.); filipe.n.santos@inesctec.pt (F.N.d.S.)</aff><aff id="af3-sensors-24-06774"><label>3</label>Faculty of Sciences, University of Porto (FCUP), 4169-007 Porto, Portugal</aff><author-notes><fn id="c1-sensors-24-06774"><label>*</label><p>Correspondence: <email>francisco.a.oliveira@inesctec.pt</email></p></fn></author-notes><pub-date><day>22</day><month>10</month><year>2024</year></pub-date><volume>24</volume><issue>21</issue><fpage>6774</fpage><page-range>6774</page-range><pub-history><event event-type="pmc-release"><date><day>9</day><month>11</month><year>2024</year></date></event></pub-history><permissions><copyright-statement>© 2024 by the authors.</copyright-statement><license><license-p>Licensee MDPI, Basel, Switzerland. This article is an open access article distributed under the terms and conditions of the Creative Commons Attribution (CC BY) license (<ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://creativecommons.org/licenses/by/4.0/" ext-link-type="uri">https://creativecommons.org/licenses/by/4.0/</ext-link>).</license-p></license></permissions><self-uri xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="sensors-24-06774.pdf" content-type="pmc-pdf"><?cloudpmc-path af01/11548613/ebb9697bdda7/sensors-24-06774.pdf?><?cloudpmc-bucket app?><?size 41795344?></self-uri><abstract id="abstract1"><title>Abstract</title><p>Automating pruning tasks entails overcoming several challenges, encompassing not only robotic manipulation but also environment perception and detection. To achieve efficient pruning, robotic systems must accurately identify the correct cutting points. A possible method to define these points is to choose the cutting location based on the number of nodes present on the targeted cane. For this purpose, in grapevine pruning, it is required to correctly identify the nodes present on the primary canes of the grapevines. In this paper, a novel method of node detection in grapevines is proposed with four distinct state-of-the-art versions of the YOLO detection model: YOLOv7, YOLOv8, YOLOv9 and YOLOv10. These models were trained on a public dataset with images containing artificial backgrounds and afterwards validated on different cultivars of grapevines from two distinct Portuguese viticulture regions with cluttered backgrounds. This allowed us to evaluate the robustness of the algorithms on the detection of nodes in diverse environments, compare the performance of the YOLO models used, as well as create a publicly available dataset of grapevines obtained in Portuguese vineyards for node detection. Overall, all used models were capable of achieving correct node detection in images of grapevines from the three distinct datasets. Considering the trade-off between accuracy and inference speed, the YOLOv7 model demonstrated to be the most robust in detecting nodes in 2D images of grapevines, achieving F1-Score values between 70% and 86.5% with inference times of around 89 ms for an input size of 1280 × 1280 px. Considering these results, this work contributes with an efficient approach for real-time node detection for further implementation on an autonomous robotic pruning system.</p><sec id="kwd-group1" sec-type="kwd-group" disp-level="2"><p><bold>Keywords:</bold> deep learning, precision agriculture, pruning, robotic systems, YOLO</p></sec></abstract><custom-meta-group><custom-meta><meta-name>status</meta-name><meta-value>released</meta-value></custom-meta><custom-meta><meta-name>display-pdf</meta-name><meta-value>yes</meta-value></custom-meta><custom-meta><meta-name>is-olf</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>is-manuscript</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>is-preprint</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>is-journal-matter</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>is-scanned</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>is-retracted</meta-name><meta-value>no</meta-value></custom-meta></custom-meta-group></article-meta><notes notes-type="article-notes"><sec id="historyarticle-meta1" sec-type="history" disp-level="2"><p>Received 2024 Aug 26; Revised 2024 Oct 3; Accepted 2024 Oct 18; Collection date 2024 Nov.</p></sec></notes></front><body><sec id="sec1-sensors-24-06774" disp-level="1"><title>1. Introduction</title><p>Pruning is a labour-intensive and costly agricultural task, which is mainly dependable on manual labour [<xref rid="B1-sensors-24-06774" ref-type="bibr">1</xref>] and requires approximately between 70 and 90 h of work per ha according to the training system [<xref rid="B2-sensors-24-06774" ref-type="bibr">2</xref>]. However, this task is highly relevant for certain species of plants, such as grapevines. Beyond mere aesthetics, pruning plays a crucial role in shaping plant health and growth by trimming branches [<xref rid="B3-sensors-24-06774" ref-type="bibr">3</xref>]. Performing pruning on grapevines promotes the growth of healthier and stronger shoots, allowing for adjustments in crop-load while maintaining the vine’s balance [<xref rid="B4-sensors-24-06774" ref-type="bibr">4</xref>]. This task is essentially performed in the winter, when the grapevines are dormant and leafless [<xref rid="B5-sensors-24-06774" ref-type="bibr">5</xref>]. Due to the difficulty in having enough labour available to perform seasonal agricultural tasks, such as pruning, research on automating these tasks is increasingly becoming a topic of interest [<xref rid="B6-sensors-24-06774" ref-type="bibr">6</xref>]. In the automation field, robotic pruning is a process that usually involves selective pruning of specific branches, similar to the method used manually [<xref rid="B7-sensors-24-06774" ref-type="bibr">7</xref>]. To effectively deploy a robotic system capable of autonomously conducting pruning tasks, the system must possess comprehensive information on task completion procedures. This entails more than just providing the robot with visual guidance to navigate to the cutting point on the branch; it also requires endowing it with the ability to accurately select the appropriate location for the cut. In pruning, there are several rules and methods that can be followed to complete the procedure. Despite the pruning method employed, it is crucial to consider the canes’ nodes location when making the cutting decisions, since it is from those nodes that buds will grow new branches. Consequently, the number of buds left after pruning will influence the grapevine’s yield [<xref rid="B8-sensors-24-06774" ref-type="bibr">8</xref>].</p><sec id="sec1dot1-sensors-24-06774" disp-level="2"><title>1.1. Related Work</title><p>Recently, there have been a few works focusing on algorithms both for branch and node detection to support agricultural task decisions, such as for pruning. Cuevas-Velasquez et al. [<xref rid="B9-sensors-24-06774" ref-type="bibr">9</xref>] developed a cutting location algorithm focusing on completing pruning tasks on rose bushes. The authors’ approach was based on stem detection, making use of a Convolutional Neural Network (CNN) to segment the branches from the background of the image. The cutting point location was then selected according to the segmented stem’s height. Marset et al. [<xref rid="B10-sensors-24-06774" ref-type="bibr">10</xref>] proposed the use of computer vision algorithms to detect buds in grapevine branches. For this purpose, the authors implemented a Fully Convolutional Network (FCN) with a MobileNet, which segmented pixels in the image corresponding to buds. Gentilhomme et al. [<xref rid="B11-sensors-24-06774" ref-type="bibr">11</xref>] proposed a novel method for grapevine structure estimation to assist in pruning activities. The authors analysed the grapevines’ structure considering the similarities with the problem of human pose estimation, with the plant’s nodes as the landmarks and the branches as the spatial dependencies. Gentilhomme et al.’s method [<xref rid="B11-sensors-24-06774" ref-type="bibr">11</xref>] consists essentially of a Stacked Hourglass Network (SHG), which outputs both the nodes and vector fields of the branches out of two-dimensional (2D) images of the grapevines. Afterwards, these features are associated according to the points’ coordinates, resulting in a complete vine structure.</p></sec><sec id="sec1dot2-sensors-24-06774" disp-level="2"><title>1.2. Contributions</title><p>This paper aims to present efficient light-computing deep learning models for node detection on 2D images of grapevines. Despite the fact that recent works have been able to use 2D images to compute cutting locations for pruning tasks, they present constraints. In the work of Cuevas-Velasquez et al. [<xref rid="B9-sensors-24-06774" ref-type="bibr">9</xref>], the cutting decision was based on the stems’ heights, not providing data on the detection of small objects such as nodes. The authors acknowledged that, in future work, the pruning criteria should be updated to consider more phenotypic traits. Meanwhile, in the research of Marset et al. [<xref rid="B10-sensors-24-06774" ref-type="bibr">10</xref>], although a detection algorithm was implemented for bud detection in grapevines, the images used were close-ups of the branch, not allowing for visibility over the whole grapevine. The work of Gentilhomme et al. [<xref rid="B11-sensors-24-06774" ref-type="bibr">11</xref>] successfully detected nodes in grapevines. However, the proposed network was trained with images containing an artificial background, and the authors considered that the algorithm was not robust enough to successfully perform on environments where the artificial background was removed from behind the grapevines.</p><p>This work aims to surpass the constraints present in the previous approaches for node detection by providing a solution that guarantees the following:</p><list list-type="bullet"><list-item><p>Nodes are detected with the further objective of being considered for pruning rules;</p></list-item><list-item><p>Node detection is achieved by the algorithm even when visualizing the entire grapevine;</p></list-item><list-item><p>Elimination of artificial background requirement for the implementation of the detection model ensures acceptable accuracy, practicality, and versatility in real-world vineyard environments, where natural backgrounds vary widely;</p></list-item><list-item><p>Further implementation on a real-time system is possible, not producing inference times that compromise the pruning task execution.</p></list-item></list><p>In order to achieve these requirements, this paper studies the adaptability of four distinct <italic>You Only Look Once</italic> (YOLO) [<xref rid="B12-sensors-24-06774" ref-type="bibr">12</xref>] models (YOLOv7 [<xref rid="B13-sensors-24-06774" ref-type="bibr">13</xref>], YOLOv8 [<xref rid="B14-sensors-24-06774" ref-type="bibr">14</xref>], YOLOv9 [<xref rid="B15-sensors-24-06774" ref-type="bibr">15</xref>] and YOLOv10 [<xref rid="B16-sensors-24-06774" ref-type="bibr">16</xref>]) on performing node detection on grapevines. These models were trained with a public grapevine dataset containing images with artificial backgrounds and were further tested on detecting nodes of grapevines with distinct configurations on natural environments without artificial backgrounds, as detailed in <xref rid="sec2dot1-sensors-24-06774" ref-type="sec">Section 2.1</xref>.</p><p>Furthermore, this allowed us to understand the requirements to correctly detect the grapevine’s nodes on two distinct and highly relevant viticulture regions in Portugal, in which the grapevines’ configurations differ significantly, as well as the backgrounds created by the region’s landscapes and climate [<xref rid="B17-sensors-24-06774" ref-type="bibr">17</xref>]. The work developed also resulted on the creation of a publicly available dataset of Portuguese grapevines for node detection [<xref rid="B18-sensors-24-06774" ref-type="bibr">18</xref>], contributing as a first approach towards the development of a perception module for an autonomous robotic system capable of performing pruning tasks with equivalent or higher performance than the same type of task performed by human labour.</p></sec></sec><sec id="sec2-sensors-24-06774" disp-level="1"><title>2. Materials and Methods</title><sec id="sec2dot1-sensors-24-06774" disp-level="2"><title>2.1. Datasets</title><p>In order to train the models, we used the publicly available 3D2cut Single Guyot dataset [<xref rid="B11-sensors-24-06774" ref-type="bibr">11</xref>], whose respective authors created to develop a method for pruning task assistance. This choice was made mainly due to three reasons: the images captured contained the entire grapevine on each frame, making it possible to visually distinguish each constituent and to simultaneously perform the detection of the same plant on all the canes; a single grapevine was captured per frame in most of the images, avoiding the presence of additional unwanted information; the dataset was collected with an artificial background behind the grapevines, which resulted in a more intuitive annotation procedure, validation of the bounding boxes’ location and distinction of the different constituents of the plant.</p><p>New annotations to the existing dataset have been performed to consider only the nodes on the primary canes of the grapevine in the centre of each image, which are the main detection goals since pruning will be conducted on these primary canes by considering the number of nodes present as a criterion. This step was performed to guarantee that it would achieve lightweight models for node detection by removing non-relevant information on the detection framework, such as nodes in areas of the grapevine not corresponding to primary canes. These annotations were performed using the Computer Vision Annotation Tool (CVAT) (CVAT. <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://www.cvat.ai/" ext-link-type="uri">https://www.cvat.ai/</ext-link> (accessed on 17 October 2024)) platform, with bounding boxes surrounding the portions of the canes containing the nodes, as seen in <xref rid="sensors-24-06774-f001" ref-type="fig">Figure 1</xref>.</p><fig id="sensors-24-06774-f001" position="float"><?disp-level 3?><label>Figure 1</label><caption><p>Manually annotated image from the 3D2cut Single Guyot dataset [<xref rid="B11-sensors-24-06774" ref-type="bibr">11</xref>] containing only the nodes on the primary canes of the grapevine.</p></caption><alternatives><graphic xmlns:xlink="http://www.w3.org/1999/xlink" content-type="image" xlink:href="sensors-24-06774-g001.jpg"><?cloudpmc-path blobs/af01/11548613/e39f470765e5/sensors-24-06774-g001.jpg?><?cloudpmc-bucket cdn?><?image-server-status LOAD_COMPLETED?><?original-height 1631?><?original-width 1971?><?scaled-height 652?><?scaled-width 788?></graphic><graphic xmlns:xlink="http://www.w3.org/1999/xlink" content-type="thumb" xlink:href="sensors-24-06774-g001.gif"><?cloudpmc-path blobs/af01/11548613/249b3e2171ec/sensors-24-06774-g001.gif?><?cloudpmc-bucket cdn?></graphic></alternatives></fig><p>To comprehensively assess the resilience of the deep learning models proposed in this study to varying grapevine conditions, particularly those with cluttered backgrounds, it became imperative to curate new datasets distinct from the 3D2cut Single Guyot dataset [<xref rid="B11-sensors-24-06774" ref-type="bibr">11</xref>], mainly containing images without artificial backgrounds. Thus, images from two distinct Portuguese viticulture regions were used to improve this evaluation.</p><p>The first set of images was collected at <italic>Centro de Estudos Vitivinícolas do Dão</italic>, located in the <italic>Dão</italic> region (40°31′31.5″ N 7°51′22.7″ W), while the second dataset was obtained from <italic>Quinta do Seixo</italic> in the <italic>Douro</italic> region (41°10′05.4″ N 7°33′17.5″ W). The performance of the trained YOLO models on images from these two locations was considered relevant to this study because the grapevine configuration is different from the 3D2cut Single Guyot dataset [<xref rid="B11-sensors-24-06774" ref-type="bibr">11</xref>]. Additionally, these images were captured without using artificial backgrounds on regions with distinct background environment information due to natural light sources and landscape, which might have influenced the outcome of the detection. Both datasets were acquired during the winter months, between December and January. The <italic>Dão</italic> dataset was captured on a clear afternoon with abundant light, whereas the <italic>Douro</italic> dataset was captured on a cloudy afternoon, where, despite sufficient lighting, the sunlight was less intense when compared to the <italic>Dão</italic> location.</p><p>For both datasets, two camera perspective variations in relation to the grapevine were employed for further performance comparison. The first was a horizontal perspective, which allowed the background of the image to have more objects in sight, such as adjacent vineyard rows and hills. The second was a bottom-to-top perspective using the sky as a more uniform background of the image, as illustrated in <xref rid="sensors-24-06774-f002" ref-type="fig">Figure 2</xref>. The images were acquired using a digital OM System camera (OM System. <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://explore.omsystem.com/us/en/" ext-link-type="uri">https://explore.omsystem.com/us/en/</ext-link> (accessed on 17 October 2024)) at a distance of approximately 80 cm of the captured grapevine with a resolution of 4608 × 3456 pixels (px).</p><fig id="sensors-24-06774-f002" position="float"><?disp-level 3?><label>Figure 2</label><caption><p>Examples of different perspectives of the captured images.</p></caption><alternatives><graphic xmlns:xlink="http://www.w3.org/1999/xlink" content-type="image" xlink:href="sensors-24-06774-g002.jpg"><?cloudpmc-path blobs/af01/11548613/73e72bdf994d/sensors-24-06774-g002.jpg?><?cloudpmc-bucket cdn?><?image-server-status LOAD_COMPLETED?><?original-height 2183?><?original-width 2658?><?scaled-height 623?><?scaled-width 759?></graphic><graphic xmlns:xlink="http://www.w3.org/1999/xlink" content-type="thumb" xlink:href="sensors-24-06774-g002.gif"><?cloudpmc-path blobs/af01/11548613/56f1c021eec5/sensors-24-06774-g002.gif?><?cloudpmc-bucket cdn?></graphic></alternatives></fig></sec><sec id="sec2dot2-sensors-24-06774" disp-level="2"><title>2.2. Deep Learning Models</title><p>YOLO deep learning models have been widely used in real-time object detection and applied in various fields, including robotics [<xref rid="B19-sensors-24-06774" ref-type="bibr">19</xref>]. These models are single-stage objective detection algorithms, granting them high speed combined with precision [<xref rid="B20-sensors-24-06774" ref-type="bibr">20</xref>]. Due to the detection performance reported by their respective authors on the Microsoft Common Objects in Context (MS COCO) dataset [<xref rid="B21-sensors-24-06774" ref-type="bibr">21</xref>], the following state-of-the-art YOLO versions were selected for this work.</p><p>YOLOv7 was published in 2022 and at the time surpassed the previously developed object detectors in terms of speed and accuracy. This accuracy was improved without incrementing the inference speed at the cost of increased training time [<xref rid="B22-sensors-24-06774" ref-type="bibr">22</xref>]. The authors also designed a version of this model to be run in edge devices—the YOLOv7-tiny [<xref rid="B13-sensors-24-06774" ref-type="bibr">13</xref>]—and since the goal of this work was to study detection algorithms capable of real-time inference on an autonomous robotic system, this version was chosen as one of the models for this purpose.</p><p>The YOLOv8 is one of the most recent object detection models, released in 2023. It has a backbone similar to the one used on YOLOv5 [<xref rid="B23-sensors-24-06774" ref-type="bibr">23</xref>] with modifications that allow for improvements on detection accuracy [<xref rid="B22-sensors-24-06774" ref-type="bibr">22</xref>]. Moreover, the YOLOv8 uses Distribution Focal Loss (DFL) [<xref rid="B24-sensors-24-06774" ref-type="bibr">24</xref>] and Complete Intersection over Union (CIoU) [<xref rid="B25-sensors-24-06774" ref-type="bibr">25</xref>] loss functions for the bounding-box loss, which is expected to contribute to improving the object detection performance in smaller objects [<xref rid="B22-sensors-24-06774" ref-type="bibr">22</xref>], as it is relevant for node detection scenarios due to the size of the nodes when compared to the full image of a grapevine. In order to guarantee that the model used could compute light and was capable of being implemented on a real-time object detection system, the model selected was the YOLOv8 Small (YOLOv8s), which is the second-smallest scaled version of the model provided by Ultralytics (Ultralytics. <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://www.ultralytics.com/" ext-link-type="uri">https://www.ultralytics.com/</ext-link> (accessed on 17 October 2024)). Furthermore, previous works have already developed on bud detection based on this version of the YOLOv8 model. Xie et al. [<xref rid="B20-sensors-24-06774" ref-type="bibr">20</xref>] implemented the YOLOv8s to detect buds on tea trees, achieving a Mean Average Precision (mAP) of 88.27% and an inference time of 37.1 ms.</p><p>In 2024, the YOLOv9 was published by the same authors of the YOLOv7. The authors designed a novel lightweight network architecture called the Generalized Efficient Layer Aggregation Network (GELAN), which, when combined with the concept of Programmable Gradient Information (PGI), also developed by the same authors, allowed it to surpass the performance of the existing object detectors. The PGI framework on the GELAN architecture helped to reduce information bottleneck during the feedforward process, which resulted in more retained complete information when compared to architectures such as PlainNet, ResNet and CSPNet [<xref rid="B15-sensors-24-06774" ref-type="bibr">15</xref>]. In order to compare the performance of this novel YOLO model in grapevine node detection with the previously presented ones, we used the light-computing YOLOv9 Small (YOLOv9-S).</p><p>Also in 2024, YOLOv10 was announced with a novel approach for real-time object detection, eliminating the use of non-maximum suppression (NMS). The model is based on the YOLOv8 from Ultralytics, which the authors have chosen due to the efficient balance achieved between latency and accuracy. Despite being based on the YOLOv8, according to the authors, the YOLOv10 achieves higher average precision (AP) while requiring less parameters and fewer calculations and also achieving lower latencies. The YOLOv10 was developed containing dual-label assignments, which, during the training phases, recurs to two heads to optimize the model. However, during inference, only one head is maintained (one-to-one) to make predictions, avoiding using NMS in post-processing. Similarly to the previous ones, the model used was the small variant: the YOLOv10 Small (YOLOv10-S).</p></sec><sec id="sec2dot3-sensors-24-06774" disp-level="2"><title>2.3. Training Configuration and Augmentations</title><p>The four distinct YOLO models presented were trained on a total of 13,539 images of grapevines, whereby 759 of the images were from the 3D2cut Single Guyot dataset [<xref rid="B11-sensors-24-06774" ref-type="bibr">11</xref>] and the remaining 12,780 images were from augmentations of these original images.</p><p>The 3D2cut Single Guyot dataset [<xref rid="B11-sensors-24-06774" ref-type="bibr">11</xref>] contained varied images of grapevines with significant differences between vines not only on their main structure, but also on cane orientation and size. However, this dataset is not a complete representation of possible environments and scenarios in a real-life application, lacking variations in the image capture perspective as well as external elements that might be present in the field. To address the lack of variability in this dataset, the following augmentations presented in <xref rid="sensors-24-06774-t001" ref-type="table">Table 1</xref> were implemented. These augmentations used to improve the training image set presented geometric modifications to the existing dataset (image flip, scaling, and angle change), and also inserted some external elements that can be present in real-life applications on the field, such as rain, mud, fog, blur, variations in hue and saturation, ISO noise, and optical distortion, as can be seen in <xref rid="sensors-24-06774-f003" ref-type="fig">Figure 3</xref>. These altered images allowed us to transform the dataset to a more reliable approximation of the agricultural work environment, which is often unpredictable and subjected to harsh and variable climatic conditions [<xref rid="B26-sensors-24-06774" ref-type="bibr">26</xref>,<xref rid="B27-sensors-24-06774" ref-type="bibr">27</xref>].</p><table-wrap id="sensors-24-06774-t001" position="float"><?disp-level 3?><label>Table 1</label><caption><p>Augmentation operations implemented.</p></caption><table frame="hsides" rules="groups"><thead><tr><th align="left" valign="middle" style="border-bottom:solid thin;border-top:solid thin" rowspan="1" colspan="1">Augmentation Operation</th><th align="center" valign="middle" style="border-bottom:solid thin;border-top:solid thin" rowspan="1" colspan="1">Values</th><th align="left" valign="middle" style="border-bottom:solid thin;border-top:solid thin" rowspan="1" colspan="1">Description</th></tr></thead><tbody><tr><td align="left" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">Horizontal Flip</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">-</td><td align="left" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">Flips the image horizontally</td></tr><tr><td rowspan="2" align="left" valign="middle" style="border-bottom:solid thin" colspan="1">Scale</td><td align="center" valign="middle" rowspan="1" colspan="1">0.7×</td><td align="left" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">Scales down the image by 30%</td></tr><tr><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">1.3×</td><td align="left" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">Scales up the image by 30%</td></tr><tr><td rowspan="2" align="left" valign="middle" style="border-bottom:solid thin" colspan="1">Rotation</td><td align="center" valign="middle" rowspan="1" colspan="1">−15°</td><td align="left" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">Rotates the image −15°</td></tr><tr><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">+15°</td><td align="left" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">Rotates the image 15°</td></tr><tr><td rowspan="3" align="left" valign="middle" style="border-bottom:solid thin" colspan="1">Hue, Saturation and Value</td><td align="center" valign="middle" rowspan="1" colspan="1">−15 ≤ <italic>hue</italic> ≤ 1</td><td rowspan="3" align="left" valign="middle" style="border-bottom:solid thin" colspan="1">Changes the image’s hue, saturation and value levels</td></tr><tr><td align="center" valign="middle" rowspan="1" colspan="1">−20 ≤ <italic>saturation</italic> ≤ 20</td></tr><tr><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">−30 ≤ <italic>value</italic> ≤ 30</td></tr><tr><td rowspan="2" align="left" valign="middle" style="border-bottom:solid thin" colspan="1">CLAHE</td><td align="center" valign="middle" rowspan="1" colspan="1"><italic>contrast limit</italic> = 4</td><td rowspan="2" align="left" valign="middle" style="border-bottom:solid thin" colspan="1">Applies Contrast Limited Adaptive Histogram Equalization to the image</td></tr><tr><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1"><italic>grid size</italic> = 8 × 8</td></tr><tr><td rowspan="2" align="left" valign="middle" style="border-bottom:solid thin" colspan="1">Emboss</td><td align="center" valign="middle" rowspan="1" colspan="1">0.4 ≤ <italic>alpha</italic> ≤ 0.6</td><td rowspan="2" align="left" valign="middle" style="border-bottom:solid thin" colspan="1">Embosses the image and overlays the result with the original image</td></tr><tr><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">0.5 ≤ <italic>strength</italic> ≤ 1.5</td></tr><tr><td rowspan="2" align="left" valign="middle" style="border-bottom:solid thin" colspan="1">Sharpen</td><td align="center" valign="middle" rowspan="1" colspan="1">0.2 ≤ <italic>alpha</italic> ≤ 0.5</td><td rowspan="2" align="left" valign="middle" style="border-bottom:solid thin" colspan="1">Sharpens the image and overlays the result with the original image</td></tr><tr><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">0.5 ≤ <italic>lightness</italic> ≤ 1.5</td></tr><tr><td rowspan="2" align="left" valign="middle" style="border-bottom:solid thin" colspan="1">Optical Distortion</td><td align="center" valign="middle" rowspan="1" colspan="1">−0.15</td><td align="left" valign="middle" rowspan="1" colspan="1">Applies negative optical distortion to the image</td></tr><tr><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">+0.15</td><td align="left" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">Applies positive optical distortion to the image</td></tr><tr><td rowspan="2" align="left" valign="middle" style="border-bottom:solid thin" colspan="1">Gaussian Blur</td><td align="center" valign="middle" rowspan="1" colspan="1"><italic>blur</italic> ≤ 7</td><td rowspan="2" align="left" valign="middle" style="border-bottom:solid thin" colspan="1">Blurs the image using a Gaussian filter</td></tr><tr><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1"><inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="mm9" overflow="scroll"><mml:mrow><mml:mi>σ</mml:mi></mml:mrow></mml:math></inline-formula> ≤ 5</td></tr><tr><td rowspan="2" align="left" valign="middle" style="border-bottom:solid thin" colspan="1">Glass Blur</td><td align="center" valign="middle" rowspan="1" colspan="1"><inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="mm10" overflow="scroll"><mml:mrow><mml:mi>σ</mml:mi></mml:mrow></mml:math></inline-formula> = 0.5</td><td rowspan="2" align="left" valign="middle" style="border-bottom:solid thin" colspan="1">Applies glass blur to the image</td></tr><tr><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1"><inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="mm11" overflow="scroll"><mml:mrow><mml:mi>δ</mml:mi></mml:mrow></mml:math></inline-formula> = 2</td></tr><tr><td rowspan="2" align="left" valign="middle" style="border-bottom:solid thin" colspan="1">ISO Noise</td><td align="center" valign="middle" rowspan="1" colspan="1"><italic>colour shift</italic> = 0.15</td><td rowspan="2" align="left" valign="middle" style="border-bottom:solid thin" colspan="1">Applies ISO noise to the image</td></tr><tr><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1"><italic>intensity</italic> = 0.6</td></tr><tr><td align="left" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">Random Rain</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">-</td><td align="left" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">Adds random rain to the image</td></tr><tr><td align="left" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">Random Fog</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">-</td><td align="left" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">Adds random fog to the image</td></tr><tr><td align="left" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">Random Snow</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">-</td><td align="left" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">Adds random snow to the image</td></tr><tr><td align="left" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">Spatter Mud</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">-</td><td align="left" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">Adds mud spatter to the image</td></tr><tr><td align="left" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">Spatter Rain</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">-</td><td align="left" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">Adds rain spatter to the image</td></tr></tbody></table></table-wrap><fig id="sensors-24-06774-f003" position="float"><?disp-level 3?><label>Figure 3</label><caption><p>Examples of augmentations performed on the 3D2cut Single Guyot dataset [<xref rid="B11-sensors-24-06774" ref-type="bibr">11</xref>] in order to increase the variability of the dataset.</p></caption><alternatives><graphic xmlns:xlink="http://www.w3.org/1999/xlink" content-type="image" xlink:href="sensors-24-06774-g003.jpg"><?cloudpmc-path blobs/af01/11548613/47483d82125c/sensors-24-06774-g003.jpg?><?cloudpmc-bucket cdn?><?image-server-status LOAD_COMPLETED?><?original-height 2444?><?original-width 2948?><?scaled-height 611?><?scaled-width 737?></graphic><graphic xmlns:xlink="http://www.w3.org/1999/xlink" content-type="thumb" xlink:href="sensors-24-06774-g003.gif"><?cloudpmc-path blobs/af01/11548613/0680f4553998/sensors-24-06774-g003.gif?><?cloudpmc-bucket cdn?></graphic></alternatives></fig><p>The deep learning models used were initially trained with similar hyperparameters for 50 epochs. However, there was a lack of convergence on the training loss for all the models. Therefore, the number of training epochs was gradually incremented to 150 for all the models to guarantee that the training loss value stabilized. Despite using similar hyperparameters in the first set of training conducted, eventually, a few suffered different tuning strategies to guarantee that the best results from each model were achieved as well as to ensure correct adaptation to the hardware used. Parameters such as input and batch sizes had to be modified on all models due to constraints on the hardware of the device used for training (NVIDIA GeForce RTX 3090, NVIDIA Corporation, Santa Clara, CA, USA) (NVIDIA GeForce RTX 3090. <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://www.nvidia.com/en-eu/geforce/graphics-cards/30-series/rtx-3090-3090ti/" ext-link-type="uri">https://www.nvidia.com/en-eu/geforce/graphics-cards/30-series/rtx-3090-3090ti/</ext-link> (accessed on 17 October 2024)). However, YOLOv8s, YOLOv9-S and YOLOv10-S presented more difficulties to reach convergence during training. For this purpose, the Cosine learning Rate Scheduler Function [<xref rid="B28-sensors-24-06774" ref-type="bibr">28</xref>] was activated during training, and mosaic data augmentation was deactivated on the last 50 epochs to improve loss stabilization at the last training stages. Furthermore, different optimizers were tested on different trainings, and it was discovered that on all the models tested, the AdamW contributed to better convergence. The basic configurations of the models and the hyperparameters that suffered further modifications in each deep learning model used, which resulted in the most favourable results, can be seen in <xref rid="sensors-24-06774-t002" ref-type="table">Table 2</xref>.</p><table-wrap id="sensors-24-06774-t002" position="float"><?disp-level 3?><label>Table 2</label><caption><p>Tuned hyperparameters used in each YOLO model.</p></caption><table frame="hsides" rules="groups"><thead><tr><th align="center" valign="middle" style="border-bottom:solid thin;border-top:solid thin" rowspan="1" colspan="1">Parameter</th><th align="center" valign="middle" style="border-bottom:solid thin;border-top:solid thin" rowspan="1" colspan="1">YOLOv7-tiny</th><th align="center" valign="middle" style="border-bottom:solid thin;border-top:solid thin" rowspan="1" colspan="1">YOLOv8s</th><th align="center" valign="middle" style="border-bottom:solid thin;border-top:solid thin" rowspan="1" colspan="1">YOLOv9-S</th><th align="center" valign="middle" style="border-bottom:solid thin;border-top:solid thin" rowspan="1" colspan="1">YOLOv10-S</th></tr></thead><tbody><tr><td align="center" valign="middle" rowspan="1" colspan="1">Input Size</td><td align="center" valign="middle" rowspan="1" colspan="1">640 × 640 px</td><td align="center" valign="middle" rowspan="1" colspan="1">640 × 640 px</td><td align="center" valign="middle" rowspan="1" colspan="1">640 × 640 px</td><td align="center" valign="middle" rowspan="1" colspan="1">640 × 640 px</td></tr><tr><td align="center" valign="middle" rowspan="1" colspan="1">Batch Size</td><td align="center" valign="middle" rowspan="1" colspan="1">16</td><td align="center" valign="middle" rowspan="1" colspan="1">16</td><td align="center" valign="middle" rowspan="1" colspan="1">16</td><td align="center" valign="middle" rowspan="1" colspan="1">16</td></tr><tr><td align="center" valign="middle" rowspan="1" colspan="1">Initial Learning Rate</td><td align="center" valign="middle" rowspan="1" colspan="1">0.01</td><td align="center" valign="middle" rowspan="1" colspan="1">0.01</td><td align="center" valign="middle" rowspan="1" colspan="1">0.01</td><td align="center" valign="middle" rowspan="1" colspan="1">0.01</td></tr><tr><td align="center" valign="middle" rowspan="1" colspan="1">Final Learning Rate</td><td align="center" valign="middle" rowspan="1" colspan="1">0.0002</td><td align="center" valign="middle" rowspan="1" colspan="1">0.0002</td><td align="center" valign="middle" rowspan="1" colspan="1">0.0002</td><td align="center" valign="middle" rowspan="1" colspan="1">0.002</td></tr><tr><td align="center" valign="middle" rowspan="1" colspan="1">Optimizer</td><td align="center" valign="middle" rowspan="1" colspan="1">AdamW</td><td align="center" valign="middle" rowspan="1" colspan="1">AdamW</td><td align="center" valign="middle" rowspan="1" colspan="1">AdamW</td><td align="center" valign="middle" rowspan="1" colspan="1">AdamW</td></tr><tr><td align="center" valign="middle" rowspan="1" colspan="1">Cos_lr Function</td><td align="center" valign="middle" rowspan="1" colspan="1">False</td><td align="center" valign="middle" rowspan="1" colspan="1">True</td><td align="center" valign="middle" rowspan="1" colspan="1">True</td><td align="center" valign="middle" rowspan="1" colspan="1">True</td></tr><tr><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">Cls_mosaic Parameter</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">-</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">50</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">50</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">50</td></tr></tbody></table></table-wrap></sec></sec><sec id="sec3-sensors-24-06774" disp-level="1"><title>3. Results</title><p>The trained YOLO models demonstrated convergence within the 150 epochs of training. However, the YOLOv7-tiny resulted in smaller box loss values for both training and validation, and those values were more stable at the end of the training when compared to the other models. The other models started to present a slight increase in the validation loss, which can be an indicator that the model was starting to overfit [<xref rid="B29-sensors-24-06774" ref-type="bibr">29</xref>]. In contrast, YOLOv10-S exhibited higher loss values during training compared to the other models. The box loss values of the trained models are presented in <xref rid="sensors-24-06774-f004" ref-type="fig">Figure 4</xref>.</p><fig id="sensors-24-06774-f004" position="float"><?disp-level 2?><label>Figure 4</label><caption><p>Box loss values on the four models trained.</p></caption><alternatives><graphic xmlns:xlink="http://www.w3.org/1999/xlink" content-type="image" xlink:href="sensors-24-06774-g004.jpg"><?cloudpmc-path blobs/af01/11548613/8dd884b7399a/sensors-24-06774-g004.jpg?><?cloudpmc-bucket cdn?><?image-server-status LOAD_COMPLETED?><?original-height 1976?><?original-width 3128?><?scaled-height 494?><?scaled-width 782?></graphic><graphic xmlns:xlink="http://www.w3.org/1999/xlink" content-type="thumb" xlink:href="sensors-24-06774-g004.gif"><?cloudpmc-path blobs/af01/11548613/3934f07da9c7/sensors-24-06774-g004.gif?><?cloudpmc-bucket cdn?></graphic></alternatives></fig><sec id="sec9" disp-level="2"><title>Validation and Field Trials</title><p>Three different sets of images were selected to evaluate the performance of the models in detecting the nodes on grapevines. The first set was composed of images from the 3D2cut Single Guyot dataset [<xref rid="B11-sensors-24-06774" ref-type="bibr">11</xref>], containing an artificial background behind the captured grapevines. The remaining two sets consisted of images selected from the datasets gathered on the Portuguese <italic>Dão</italic> and <italic>Douro</italic> regions, containing 30 original images and 150 augmented images, which added geometric modifications to the original images in order to increase the variability of the dataset by containing grapevines captured in different layouts.</p><p>The images from the 3D2cut Single Guyot dataset [<xref rid="B11-sensors-24-06774" ref-type="bibr">11</xref>] were used as a benchmark to compare the robustness of the YOLO models trained. Since these images were similar to the ones used for training, it was expected that the models would achieve the best results. <italic>Dão</italic> and <italic>Douro</italic> datasets constituted of images taken in open fields without covering the objects behind the grapevines. This feature allowed these datasets to be used as indicators to evaluate the robustness of the models developed in detecting nodes in images with natural backgrounds (cluttered).</p><p>Furthermore, despite the fact that the models have been trained with an input size of 640 × 640 px, tests were also conducted with changes in the input parameter to consider a size of 1280 × 1280 px. Since nodes on grapevines are relatively small objects on the overall size of the image, this test was crucial to reduce the loss of information on the images used. However, it was expected that this increase in the input size would result in delays in the inference times of the models [<xref rid="B30-sensors-24-06774" ref-type="bibr">30</xref>].</p><p>The results obtained on the validation phase of the detection models trained on the four different datasets are presented in <xref rid="sensors-24-06774-t003" ref-type="table">Table 3</xref>. The metrics obtained were analysed at confidence levels of at least 10% and at the best F1-Score obtained in each model. Confidence levels below 10% were excluded from the analysis, as lower confidence scores are likely to lead to an increase in false positive detections [<xref rid="B31-sensors-24-06774" ref-type="bibr">31</xref>]. An increase in false positives would negatively affect the performance of the node detection task, resulting in inaccurate conclusions and potentially compromising the reliability of the system to be integrated into an autonomous pruning robot. Furthermore, <xref rid="sensors-24-06774-t004" ref-type="table">Table 4</xref> displays the average inference time per image on each of the models with different input sizes.</p><table-wrap id="sensors-24-06774-t003" position="float"><?disp-level 3?><label>Table 3</label><caption><p>Performance of the models in images from the different datasets.</p></caption><table frame="hsides" rules="groups"><thead><tr><th rowspan="2" align="center" valign="middle" style="border-bottom:solid thin;border-top:solid thin" colspan="1">Dataset</th><th rowspan="2" align="center" valign="middle" style="border-bottom:solid thin;border-top:solid thin" colspan="1">Model</th><th align="center" valign="middle" style="border-top:solid thin" rowspan="1" colspan="1">Input Size</th><th align="center" valign="middle" style="border-top:solid thin" rowspan="1" colspan="1">Precision</th><th align="center" valign="middle" style="border-top:solid thin" rowspan="1" colspan="1">Recall</th><th align="center" valign="middle" style="border-top:solid thin" rowspan="1" colspan="1">F1-Score</th><th align="center" valign="middle" style="border-top:solid thin" rowspan="1" colspan="1">mAP@50</th><th align="center" valign="middle" style="border-top:solid thin" rowspan="1" colspan="1">Precision</th><th align="center" valign="middle" style="border-top:solid thin" rowspan="1" colspan="1">Recall</th><th align="center" valign="middle" style="border-top:solid thin" rowspan="1" colspan="1">F1-Score</th><th align="center" valign="middle" style="border-top:solid thin" rowspan="1" colspan="1">mAP@50</th><th align="center" valign="middle" style="border-top:solid thin" rowspan="1" colspan="1">Average</th></tr><tr><th align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">(px)</th><th colspan="4" align="center" valign="middle" style="border-bottom:solid thin" rowspan="1">Confidence ≥ 10%</th><th colspan="4" align="center" valign="middle" style="border-bottom:solid thin" rowspan="1">On the Best F1-Score</th><th align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">IoU</th></tr></thead><tbody><tr><td rowspan="8" align="center" valign="middle" style="border-bottom:solid thin" colspan="1">3D2cut</td><td align="center" valign="middle" rowspan="1" colspan="1">YOLOv7-tiny</td><td rowspan="4" align="center" valign="middle" style="border-bottom:solid thin" colspan="1">640 × 640</td><td align="center" valign="middle" rowspan="1" colspan="1">84.5%</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>86.8%</bold>
</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>85.6%</bold>
</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>83.4%</bold>
</td><td align="center" valign="middle" rowspan="1" colspan="1">84.5%</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>86.8%</bold>
</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>85.6%</bold>
</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>83.4%</bold>
</td><td align="center" valign="middle" rowspan="1" colspan="1">78.6%</td></tr><tr><td align="center" valign="middle" rowspan="1" colspan="1">YOLOv8s</td><td align="center" valign="middle" rowspan="1" colspan="1">90.9%</td><td align="center" valign="middle" rowspan="1" colspan="1">73.8%</td><td align="center" valign="middle" rowspan="1" colspan="1">81.5%</td><td align="center" valign="middle" rowspan="1" colspan="1">71.3%</td><td align="center" valign="middle" rowspan="1" colspan="1">90.9%</td><td align="center" valign="middle" rowspan="1" colspan="1">73.8%</td><td align="center" valign="middle" rowspan="1" colspan="1">81.5%</td><td align="center" valign="middle" rowspan="1" colspan="1">71.3%</td><td align="center" valign="middle" rowspan="1" colspan="1">80.5%</td></tr><tr><td align="center" valign="middle" rowspan="1" colspan="1">YOLOv9-S</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>91.1%</bold>
</td><td align="center" valign="middle" rowspan="1" colspan="1">76.2%</td><td align="center" valign="middle" rowspan="1" colspan="1">83%</td><td align="center" valign="middle" rowspan="1" colspan="1">74.6%</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>91.1%</bold>
</td><td align="center" valign="middle" rowspan="1" colspan="1">76.2%</td><td align="center" valign="middle" rowspan="1" colspan="1">83%</td><td align="center" valign="middle" rowspan="1" colspan="1">74.6%</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>80.9%</bold>
</td></tr><tr><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">YOLOv10-S</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">87.9%</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">77.3%</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">82.3%</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">75.3%</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">87.9%</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">77.3%</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">82.3%</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">75.3%</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">80.6%</td></tr><tr><td align="center" valign="middle" rowspan="1" colspan="1">YOLOv7-tiny</td><td rowspan="4" align="center" valign="middle" style="border-bottom:solid thin" colspan="1">1280 × 1280</td><td align="center" valign="middle" rowspan="1" colspan="1">71.8%</td><td align="center" valign="middle" rowspan="1" colspan="1">92.6%</td><td align="center" valign="middle" rowspan="1" colspan="1">80.9%</td><td align="center" valign="middle" rowspan="1" colspan="1">86.6%</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>88.8%</bold>
</td><td align="center" valign="middle" rowspan="1" colspan="1">84.3%</td><td align="center" valign="middle" rowspan="1" colspan="1">86.5%</td><td align="center" valign="middle" rowspan="1" colspan="1">80.1%</td><td align="center" valign="middle" rowspan="1" colspan="1">76.3%</td></tr><tr><td align="center" valign="middle" rowspan="1" colspan="1">YOLOv8s</td><td align="center" valign="middle" rowspan="1" colspan="1">76.3%</td><td align="center" valign="middle" rowspan="1" colspan="1">91.4%</td><td align="center" valign="middle" rowspan="1" colspan="1">83.2%</td><td align="center" valign="middle" rowspan="1" colspan="1">84.9%</td><td align="center" valign="middle" rowspan="1" colspan="1">87.6%</td><td align="center" valign="middle" rowspan="1" colspan="1">86.8%</td><td align="center" valign="middle" rowspan="1" colspan="1">87.2%</td><td align="center" valign="middle" rowspan="1" colspan="1">80.8%</td><td align="center" valign="middle" rowspan="1" colspan="1">77%</td></tr><tr><td align="center" valign="middle" rowspan="1" colspan="1">YOLOv9-S</td><td align="center" valign="middle" rowspan="1" colspan="1">78.4%</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>93.1%</bold>
</td><td align="center" valign="middle" rowspan="1" colspan="1">85.1%</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>88.5%</bold>
</td><td align="center" valign="middle" rowspan="1" colspan="1">84.1%</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>90.5%</bold>
</td><td align="center" valign="middle" rowspan="1" colspan="1">87.2%</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>86.1%</bold>
</td><td align="center" valign="middle" rowspan="1" colspan="1">78.3%</td></tr><tr><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">YOLOv10-S</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">
<bold>80.6%</bold>
</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">91.4%</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">
<bold>85.7%</bold>
</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">86.8%</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">85.8%</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">89.8%</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">
<bold>87.8%</bold>
</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">85.2%</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">
<bold>79.1%</bold>
</td></tr><tr><td rowspan="8" align="center" valign="middle" style="border-bottom:solid thin" colspan="1">
<italic>Dão</italic>
</td><td align="center" valign="middle" rowspan="1" colspan="1">YOLOv7-tiny</td><td rowspan="4" align="center" valign="middle" style="border-bottom:solid thin" colspan="1">640 × 640</td><td align="center" valign="middle" rowspan="1" colspan="1">79%</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>49.5%</bold>
</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>60.8%</bold>
</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>44.5%</bold>
</td><td align="center" valign="middle" rowspan="1" colspan="1">79%</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>49.5%</bold>
</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>60.8%</bold>
</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>44.5%</bold>
</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>70.8%</bold>
</td></tr><tr><td align="center" valign="middle" rowspan="1" colspan="1">YOLOv8s</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>85.5%</bold>
</td><td align="center" valign="middle" rowspan="1" colspan="1">19%</td><td align="center" valign="middle" rowspan="1" colspan="1">31.1%</td><td align="center" valign="middle" rowspan="1" colspan="1">18.8%</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>85.5%</bold>
</td><td align="center" valign="middle" rowspan="1" colspan="1">19%</td><td align="center" valign="middle" rowspan="1" colspan="1">31.1%</td><td align="center" valign="middle" rowspan="1" colspan="1">18.8%</td><td align="center" valign="middle" rowspan="1" colspan="1">69.6%</td></tr><tr><td align="center" valign="middle" rowspan="1" colspan="1">YOLOv9-S</td><td align="center" valign="middle" rowspan="1" colspan="1">76.6%</td><td align="center" valign="middle" rowspan="1" colspan="1">16%</td><td align="center" valign="middle" rowspan="1" colspan="1">26.4%</td><td align="center" valign="middle" rowspan="1" colspan="1">13.9%</td><td align="center" valign="middle" rowspan="1" colspan="1">76.6%</td><td align="center" valign="middle" rowspan="1" colspan="1">16%</td><td align="center" valign="middle" rowspan="1" colspan="1">26.4%</td><td align="center" valign="middle" rowspan="1" colspan="1">13.9%</td><td align="center" valign="middle" rowspan="1" colspan="1">63.2%</td></tr><tr><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">YOLOv10-S</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">78.9%</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">13.9%</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">23.7%</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">12%</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">78.9%</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">13.9%</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">23.7%</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">12%</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">67.2%</td></tr><tr><td align="center" valign="middle" rowspan="1" colspan="1">YOLOv7-tiny</td><td rowspan="4" align="center" valign="middle" style="border-bottom:solid thin" colspan="1">1280 × 1280</td><td align="center" valign="middle" rowspan="1" colspan="1">73.4%</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>70.4%</bold>
</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>71.8%</bold>
</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>61.5%</bold>
</td><td align="center" valign="middle" rowspan="1" colspan="1">76.4%</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>69.7%</bold>
</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>72.9%</bold>
</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>60.8%</bold>
</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>74.8%</bold>
</td></tr><tr><td align="center" valign="middle" rowspan="1" colspan="1">YOLOv8s</td><td align="center" valign="middle" rowspan="1" colspan="1">76.4%</td><td align="center" valign="middle" rowspan="1" colspan="1">64.7%</td><td align="center" valign="middle" rowspan="1" colspan="1">70%</td><td align="center" valign="middle" rowspan="1" colspan="1">53.9%</td><td align="center" valign="middle" rowspan="1" colspan="1">76.4%</td><td align="center" valign="middle" rowspan="1" colspan="1">64.7%</td><td align="center" valign="middle" rowspan="1" colspan="1">70%</td><td align="center" valign="middle" rowspan="1" colspan="1">53.9%</td><td align="center" valign="middle" rowspan="1" colspan="1">73.6%</td></tr><tr><td align="center" valign="middle" rowspan="1" colspan="1">YOLOv9-S</td><td align="center" valign="middle" rowspan="1" colspan="1">79.3%</td><td align="center" valign="middle" rowspan="1" colspan="1">61.4%</td><td align="center" valign="middle" rowspan="1" colspan="1">69.2%</td><td align="center" valign="middle" rowspan="1" colspan="1">50.8%</td><td align="center" valign="middle" rowspan="1" colspan="1">79.3%</td><td align="center" valign="middle" rowspan="1" colspan="1">61.4%</td><td align="center" valign="middle" rowspan="1" colspan="1">69.2%</td><td align="center" valign="middle" rowspan="1" colspan="1">50.8%</td><td align="center" valign="middle" rowspan="1" colspan="1">72.2%</td></tr><tr><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">YOLOv10-S</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">
<bold>80%</bold>
</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">55.4%</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">65.5%</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">46.1%</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">
<bold>80%</bold>
</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">55.4%</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">65.5%</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">46.1%</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">73.3%</td></tr><tr><td rowspan="8" align="center" valign="middle" style="border-bottom:solid thin" colspan="1">
<italic>Douro</italic>
</td><td align="center" valign="middle" rowspan="1" colspan="1">YOLOv7-tiny</td><td rowspan="4" align="center" valign="middle" style="border-bottom:solid thin" colspan="1">640 × 640</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>72.2%</bold>
</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>47%</bold>
</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>56.9%</bold>
</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>42%</bold>
</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>72.2%</bold>
</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>47%</bold>
</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>56.9%</bold>
</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>42%</bold>
</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>66.5%</bold>
</td></tr><tr><td align="center" valign="middle" rowspan="1" colspan="1">YOLOv8s</td><td align="center" valign="middle" rowspan="1" colspan="1">67.3%</td><td align="center" valign="middle" rowspan="1" colspan="1">18.7%</td><td align="center" valign="middle" rowspan="1" colspan="1">29.2%</td><td align="center" valign="middle" rowspan="1" colspan="1">14.6%</td><td align="center" valign="middle" rowspan="1" colspan="1">67.3%</td><td align="center" valign="middle" rowspan="1" colspan="1">18.7%</td><td align="center" valign="middle" rowspan="1" colspan="1">29.2%</td><td align="center" valign="middle" rowspan="1" colspan="1">14.6%</td><td align="center" valign="middle" rowspan="1" colspan="1">61.7%</td></tr><tr><td align="center" valign="middle" rowspan="1" colspan="1">YOLOv9-S</td><td align="center" valign="middle" rowspan="1" colspan="1">70.6%</td><td align="center" valign="middle" rowspan="1" colspan="1">19.9%</td><td align="center" valign="middle" rowspan="1" colspan="1">31%</td><td align="center" valign="middle" rowspan="1" colspan="1">15.9%</td><td align="center" valign="middle" rowspan="1" colspan="1">70.6%</td><td align="center" valign="middle" rowspan="1" colspan="1">19.9%</td><td align="center" valign="middle" rowspan="1" colspan="1">31%</td><td align="center" valign="middle" rowspan="1" colspan="1">15.8%</td><td align="center" valign="middle" rowspan="1" colspan="1">64.1%</td></tr><tr><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">YOLOv10-S</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">68.5%</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">18.9%</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">29.6%</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">14.5%</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">68.5%</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">18.9%</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">29.6%</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">14.5%</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">62.4%</td></tr><tr><td align="center" valign="middle" rowspan="1" colspan="1">YOLOv7-tiny</td><td rowspan="4" align="center" valign="middle" style="border-bottom:solid thin" colspan="1">1280 × 1280</td><td align="center" valign="middle" rowspan="1" colspan="1">63.9%</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>70.6%</bold>
</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>67.1%</bold>
</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>62.8%</bold>
</td><td align="center" valign="middle" rowspan="1" colspan="1">74.2%</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>65.5%</bold>
</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>69.6%</bold>
</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>59.3%</bold>
</td><td align="center" valign="middle" rowspan="1" colspan="1">68.5%</td></tr><tr><td align="center" valign="middle" rowspan="1" colspan="1">YOLOv8s</td><td align="center" valign="middle" rowspan="1" colspan="1">72.7%</td><td align="center" valign="middle" rowspan="1" colspan="1">58.9%</td><td align="center" valign="middle" rowspan="1" colspan="1">65.1%</td><td align="center" valign="middle" rowspan="1" colspan="1">52%</td><td align="center" valign="middle" rowspan="1" colspan="1">72.7%</td><td align="center" valign="middle" rowspan="1" colspan="1">58.9%</td><td align="center" valign="middle" rowspan="1" colspan="1">65.1%</td><td align="center" valign="middle" rowspan="1" colspan="1">52%</td><td align="center" valign="middle" rowspan="1" colspan="1">68.2%</td></tr><tr><td align="center" valign="middle" rowspan="1" colspan="1">YOLOv9-S</td><td align="center" valign="middle" rowspan="1" colspan="1">71.1%</td><td align="center" valign="middle" rowspan="1" colspan="1">53.5%</td><td align="center" valign="middle" rowspan="1" colspan="1">61.1%</td><td align="center" valign="middle" rowspan="1" colspan="1">46.4%</td><td align="center" valign="middle" rowspan="1" colspan="1">71.1%</td><td align="center" valign="middle" rowspan="1" colspan="1">53.5%</td><td align="center" valign="middle" rowspan="1" colspan="1">61.1%</td><td align="center" valign="middle" rowspan="1" colspan="1">46.4%</td><td align="center" valign="middle" rowspan="1" colspan="1">67%</td></tr><tr><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">YOLOv10-S</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">
<bold>77.4%</bold>
</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">49.2%</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">60.2%</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">44.9%</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">
<bold>77.4%</bold>
</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">49.2%</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">60.2%</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">44.9%</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">
<bold>69.2%</bold>
</td></tr></tbody></table><table-wrap-foot><fn id="fn2"><p>Text in <bold>Bold</bold> indicates the best-performing results for each metric.</p></fn></table-wrap-foot></table-wrap><table-wrap id="sensors-24-06774-t004" position="float"><?disp-level 3?><label>Table 4</label><caption><p>Average inference time of the YOLO models.</p></caption><table frame="hsides" rules="groups"><thead><tr><th align="center" valign="middle" style="border-bottom:solid thin;border-top:solid thin" rowspan="1" colspan="1">Model</th><th align="center" valign="middle" style="border-bottom:solid thin;border-top:solid thin" rowspan="1" colspan="1">Input Size (px)</th><th align="center" valign="middle" style="border-bottom:solid thin;border-top:solid thin" rowspan="1" colspan="1">Average Inference Time (ms)</th></tr></thead><tbody><tr><td align="center" valign="middle" rowspan="1" colspan="1">YOLOv7-tiny</td><td rowspan="4" align="center" valign="middle" style="border-bottom:solid thin" colspan="1">640 × 640</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>20.52</bold>
</td></tr><tr><td align="center" valign="middle" rowspan="1" colspan="1">YOLOv8s</td><td align="center" valign="middle" rowspan="1" colspan="1">323.79</td></tr><tr><td align="center" valign="middle" rowspan="1" colspan="1">YOLOv9-S</td><td align="center" valign="middle" rowspan="1" colspan="1">63.82</td></tr><tr><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">YOLOv10-S</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">51.59</td></tr><tr><td align="center" valign="middle" rowspan="1" colspan="1">YOLOv7-tiny</td><td rowspan="4" align="center" valign="middle" style="border-bottom:solid thin" colspan="1">1280 × 1280</td><td align="center" valign="middle" rowspan="1" colspan="1">
<bold>88.79</bold>
</td></tr><tr><td align="center" valign="middle" rowspan="1" colspan="1">YOLOv8s</td><td align="center" valign="middle" rowspan="1" colspan="1">502.52</td></tr><tr><td align="center" valign="middle" rowspan="1" colspan="1">YOLOv9-S</td><td align="center" valign="middle" rowspan="1" colspan="1">288.53</td></tr><tr><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">YOLOv10-S</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">260.21</td></tr></tbody></table><table-wrap-foot><fn id="fn3"><p>Text in <bold>Bold</bold> indicates the shortest inference time for each input size.</p></fn></table-wrap-foot></table-wrap></sec></sec><sec id="sec4-sensors-24-06774" disp-level="1"><title>4. Discussion</title><p>Overall, YOLOv7-tiny achieved higher F1-Score values than the other models. The more recent models tested, despite typically presenting slightly higher Precision values, resulted in lower Recall values. Moreover, the mAP obtained by the novel YOLOv10-S was usually significantly lower than the other models benchmarked for scenarios significantly different from the training environment.</p><p>Observing the results previously presented in <xref rid="sensors-24-06774-t003" ref-type="table">Table 3</xref>, there was a tendency for the models to perform better with an input size of 1280 × 1280 px when compared to the standard input size of 640 × 640 px, which supports the possibility of information being lost in smaller images due to the reduced size of the nodes. Due to this characteristic, it was considered advantageous to use this larger input size on the developed models, which demonstrated performance improvements. <xref rid="sensors-24-06774-f005" ref-type="fig">Figure 5</xref> shows a comparison of the F1-Score values from the four YOLO models analysed on the different datasets used, considering an input size of 1280 × 1280 px.</p><fig id="sensors-24-06774-f005" position="float"><?disp-level 2?><label>Figure 5</label><caption><p>F1-Score of the models with an input size of 1280 × 1280 px in the three datasets used. (<bold>a</bold>) 3D2cut Single Guyot [<xref rid="B11-sensors-24-06774" ref-type="bibr">11</xref>]; (<bold>b</bold>) <italic>Dão</italic>; (<bold>c</bold>) <italic>Douro</italic>.</p></caption><alternatives><graphic xmlns:xlink="http://www.w3.org/1999/xlink" content-type="image" xlink:href="sensors-24-06774-g005a.jpg"><?cloudpmc-path blobs/af01/11548613/edd49f7223d0/sensors-24-06774-g005a.jpg?><?cloudpmc-bucket cdn?><?image-server-status LOAD_COMPLETED?><?original-height 3123?><?original-width 2492?><?scaled-height 892?><?scaled-width 712?></graphic><graphic xmlns:xlink="http://www.w3.org/1999/xlink" content-type="thumb" xlink:href="sensors-24-06774-g005a.gif"><?cloudpmc-path blobs/af01/11548613/3763f0383cf4/sensors-24-06774-g005a.gif?><?cloudpmc-bucket cdn?></graphic></alternatives><alternatives><graphic xmlns:xlink="http://www.w3.org/1999/xlink" content-type="image" xlink:href="sensors-24-06774-g005b.jpg"><?cloudpmc-path blobs/af01/11548613/d4c33e24872c/sensors-24-06774-g005b.jpg?><?cloudpmc-bucket cdn?><?image-server-status LOAD_COMPLETED?><?original-height 1573?><?original-width 2479?><?scaled-height 449?><?scaled-width 708?></graphic><graphic xmlns:xlink="http://www.w3.org/1999/xlink" content-type="thumb" xlink:href="sensors-24-06774-g005b.gif"><?cloudpmc-path blobs/af01/11548613/904b78fde9dc/sensors-24-06774-g005b.gif?><?cloudpmc-bucket cdn?></graphic></alternatives></fig><p>Despite these performance differences between the used versions, all the models were capable of detecting the grapevine’s nodes successfully. Furthermore, it was possible to observe that on the 3D2cut Single Guyot dataset [<xref rid="B11-sensors-24-06774" ref-type="bibr">11</xref>], the four models achieved F1-Score values of above 80%, despite the input size used. Meanwhile, on both <italic>Dão</italic> and <italic>Douro</italic> datasets, the highest F1-Score values were around 70%. However, since the objective of this study is to further implement a perception algorithm on a robotic system capable of performing pruning tasks autonomously, not only is it crucial to analyse the models’ accuracy and their adaptability to different real-case scenarios, but also how much time they take to complete the inference.</p><sec id="sec4dot1-sensors-24-06774" disp-level="2"><title>4.1. Adaptability to Different Grapevine’s Configurations and Environments</title><p>As expected, the models’ inference performance was considerably higher on the 3D2cut Single Guyot dataset [<xref rid="B11-sensors-24-06774" ref-type="bibr">11</xref>], where all the models presented a similar detection capability, as seen in <xref rid="sensors-24-06774-f006" ref-type="fig">Figure 6</xref>. This behaviour occurs not only because this set constituted of images similar to the ones used for training, but also due to the amount of additional information on the background of the <italic>Dão</italic> and <italic>Douro</italic> datasets, which induced the occurrence of some false positive detections on the background of the images.</p><fig id="sensors-24-06774-f006" position="float"><?disp-level 3?><label>Figure 6</label><caption><p>Inference on a grapevine from the 3D2cut Single Guyot dataset [<xref rid="B11-sensors-24-06774" ref-type="bibr">11</xref>] considering an input size of 1280 × 1280 px.</p></caption><alternatives><graphic xmlns:xlink="http://www.w3.org/1999/xlink" content-type="image" xlink:href="sensors-24-06774-g006.jpg"><?cloudpmc-path blobs/af01/11548613/08610a2bc63c/sensors-24-06774-g006.jpg?><?cloudpmc-bucket cdn?><?image-server-status LOAD_COMPLETED?><?original-height 3164?><?original-width 2512?><?scaled-height 903?><?scaled-width 717?></graphic><graphic xmlns:xlink="http://www.w3.org/1999/xlink" content-type="thumb" xlink:href="sensors-24-06774-g006.gif"><?cloudpmc-path blobs/af01/11548613/e65899ba050c/sensors-24-06774-g006.gif?><?cloudpmc-bucket cdn?></graphic></alternatives></fig><p>In both of these two datasets (<italic>Dão</italic> and <italic>Douro</italic>), we noticed that the YOLOv7-tiny achieved higher F1-Score values than the other three models, with this difference being more noticeable on these datasets in comparison to the results obtained on the 3D2cut Single Guyot dataset [<xref rid="B11-sensors-24-06774" ref-type="bibr">11</xref>]. Furthermore, although YOLOv8s was the second best performing model on the <italic>Dão</italic> dataset, the performance changed significantly on the <italic>Douro</italic> dataset, in which this model presented varying results according to the input size used, resulting in lower F1-Score values for an input size of 640 × 640 px. Although the Precision of YOLOv10-S was usually higher than the other models on the Portuguese datasets, the Recall was lower, especially when considering an input size of 640 × 640 px. This allowed us to notice that when using the YOLOv10-S, despite not encountering many false positives, the model presented more difficulties in detecting small objects on each frame, ignoring a few nodes in the image. This behaviour makes it less valuable for the intended task than the other models analysed, since this behaviour results in a considerable amount of nodes not being considered as such.</p><p>Furthermore, when comparing the results obtained between the two datasets without an artificial background, the <italic>Douro</italic> dataset resulted in less-accurate detections by the models than the <italic>Dão</italic> dataset. Analysing the images present in each one, it is possible to observe what features might be causing this discrepancy. The grapevines on the <italic>Dão</italic> dataset, despite having a structure different from the grapevines on the images of the training dataset, make distinguishing each cane on the plant easily feasible. Meanwhile, on the <italic>Douro</italic> dataset, the grapevines have more canes overlapped, and thus sometimes make it more difficult to distinguish the primary canes; this also results in nodes that can be occluded by nearby branches, as can be observed in <xref rid="sensors-24-06774-f007" ref-type="fig">Figure 7</xref>.</p><fig id="sensors-24-06774-f007" position="float"><?disp-level 3?><label>Figure 7</label><caption><p>Partially occluded node (red circle) not detected in image from <italic>Douro</italic> dataset.</p></caption><alternatives><graphic xmlns:xlink="http://www.w3.org/1999/xlink" content-type="image" xlink:href="sensors-24-06774-g007.jpg"><?cloudpmc-path blobs/af01/11548613/488d9eb8e93c/sensors-24-06774-g007.jpg?><?cloudpmc-bucket cdn?><?image-server-status LOAD_COMPLETED?><?original-height 1328?><?original-width 1415?><?scaled-height 664?><?scaled-width 707?></graphic><graphic xmlns:xlink="http://www.w3.org/1999/xlink" content-type="thumb" xlink:href="sensors-24-06774-g007.gif"><?cloudpmc-path blobs/af01/11548613/41edaedf481b/sensors-24-06774-g007.gif?><?cloudpmc-bucket cdn?></graphic></alternatives></fig><p>Regarding the annotations performed on the 3D2cut Single Guyot dataset [<xref rid="B11-sensors-24-06774" ref-type="bibr">11</xref>], only the nodes on the primary canes of the centred grapevine on the image were selected. The same logic was implemented on the construction of the ground truth for the <italic>Dão</italic> and <italic>Douro</italic> image sets. However, the models also detected a few nodes on canes in adjacent grapevines that slightly appeared on the images captured. This consequently resulted in these detections being considered as false positives, which is not entirely wrong since the nodes do not belong to the focused grapevine, but they are correctly assumed as being nodes. Although these situations slightly affect some of the images, they were not considered relevant for the overall performance of the models in detecting nodes on these datasets, since it only happened in a few cases. Furthermore, in these sets of images taken in open field, the pictures captured in a bottom-to-top perspective are preferable to achieve better results. This approach helped reduce the background information, resulting in less false positives detected. In <xref rid="sensors-24-06774-f008" ref-type="fig">Figure 8</xref>, it is possible to notice that a horizontal camera perspective captures additional information, not only from other grapevines behind the one in foreground, but also from the ground of the figure. Meanwhile, a bottom-to-top perspective reduces the amount of content in the background of the image, allowing for a more precise detection.</p><fig id="sensors-24-06774-f008" position="float"><?disp-level 3?><label>Figure 8</label><caption><p>YOLOv7 inference comparison considering two different camera perspectives.</p></caption><alternatives><graphic xmlns:xlink="http://www.w3.org/1999/xlink" content-type="image" xlink:href="sensors-24-06774-g008.jpg"><?cloudpmc-path blobs/af01/11548613/4e42c4217d73/sensors-24-06774-g008.jpg?><?cloudpmc-bucket cdn?><?image-server-status LOAD_COMPLETED?><?original-height 2776?><?original-width 1679?><?scaled-height 1109?><?scaled-width 671?></graphic><graphic xmlns:xlink="http://www.w3.org/1999/xlink" content-type="thumb" xlink:href="sensors-24-06774-g008.gif"><?cloudpmc-path blobs/af01/11548613/6f8c0c72022b/sensors-24-06774-g008.gif?><?cloudpmc-bucket cdn?></graphic></alternatives></fig><p>Regardless of the performance differences between datasets and between images containing artificial backgrounds or not, grapevine node detection recurring to YOLO detection models presents an efficient and more versatile approach when compared to the previously implemented methods. The previously published work by Gentilhomme et al. [<xref rid="B11-sensors-24-06774" ref-type="bibr">11</xref>] managed to achieve an overall Precision of 95% and Recall of 90% in the task of node detection, having an execution time between 0.55 s and 8 s. Despite obtaining good detection performance on the selected testing environment, the authors acknowledged that the algorithm requires optimization to be implemented on a real-time system and also considered that the algorithm proposed for pruning assistance would be severely impacted in work conditions without uniform artificial backgrounds. However, this work assures that grapevine node detection is feasible in heterogeneous environments recurring to state-of-the-art YOLO detection models, despite presenting a slight performance decrease.</p><p>Furthermore, it was possible to assess the capability of performing node detection on the entirety of the grapevines even when configurations different from those used in training are provided to the models. In <xref rid="sensors-24-06774-f009" ref-type="fig">Figure 9</xref> and <xref rid="sensors-24-06774-f010" ref-type="fig">Figure 10</xref>, it is possible to analyse the inference behaviour of the models used on each of the datasets acquired in Portugal, <italic>Dão</italic> and Douro, respectively, considering an input size of 1280 × 1280 px.</p><fig id="sensors-24-06774-f009" position="float"><?disp-level 3?><label>Figure 9</label><caption><p>Inference on grapevines from the <italic>Dão</italic> region.</p></caption><alternatives><graphic xmlns:xlink="http://www.w3.org/1999/xlink" content-type="image" xlink:href="sensors-24-06774-g009.jpg"><?cloudpmc-path blobs/af01/11548613/afdf45da93de/sensors-24-06774-g009.jpg?><?cloudpmc-bucket cdn?><?image-server-status LOAD_COMPLETED?><?original-height 2867?><?original-width 2246?><?scaled-height 955?><?scaled-width 748?></graphic><graphic xmlns:xlink="http://www.w3.org/1999/xlink" content-type="thumb" xlink:href="sensors-24-06774-g009.gif"><?cloudpmc-path blobs/af01/11548613/18c03e2bebc5/sensors-24-06774-g009.gif?><?cloudpmc-bucket cdn?></graphic></alternatives></fig><fig id="sensors-24-06774-f010" position="float"><?disp-level 3?><label>Figure 10</label><caption><p>Inference on grapevines from the <italic>Douro</italic> region.</p></caption><alternatives><graphic xmlns:xlink="http://www.w3.org/1999/xlink" content-type="image" xlink:href="sensors-24-06774-g010.jpg"><?cloudpmc-path blobs/af01/11548613/5da4a878257f/sensors-24-06774-g010.jpg?><?cloudpmc-bucket cdn?><?image-server-status LOAD_COMPLETED?><?original-height 2925?><?original-width 2312?><?scaled-height 974?><?scaled-width 770?></graphic><graphic xmlns:xlink="http://www.w3.org/1999/xlink" content-type="thumb" xlink:href="sensors-24-06774-g010.gif"><?cloudpmc-path blobs/af01/11548613/8fd4e369c5d2/sensors-24-06774-g010.gif?><?cloudpmc-bucket cdn?></graphic></alternatives></fig></sec><sec id="sec4dot2-sensors-24-06774" disp-level="2"><title>4.2. Capability of Implementation in a Real-Time System</title><p>According to the work by Botterill et al. [<xref rid="B1-sensors-24-06774" ref-type="bibr">1</xref>], a human takes approximately 2 min to prune a grapevine. In a robotic system, there might be several necessary additional steps from the detection of the cutting location until the cut is completed. These steps are mainly related to navigation towards the cutting location and actuation of the cutting tool [<xref rid="B7-sensors-24-06774" ref-type="bibr">7</xref>,<xref rid="B32-sensors-24-06774" ref-type="bibr">32</xref>]. Considering this, minimizing the inference time of the detection algorithm is crucial to ensure the maximum efficiency of the system by reducing the necessary time to execute the pruning task on each vine.</p><p>The average inference time of the analysed YOLO models varied significantly. These inference time differences were not only between the distinct versions of the models, but also between different input sizes. YOLOv7-tiny was the fastest model at grapevine node detection, presenting inference times below 100 ms. YOLOv8s, YOLOv9-S and YOLOv10-S presented significantly larger times, with YOLOv8s reaching an average of around 502.5 ms when detecting nodes with a 1280 × 1280 px input size.</p><p>Despite the most recent models (YOLOv9-S and YOLOv10-S) being faster than the 502 ms of YOLOv8s, their inference times on 1280 × 1280 px sized images are still near 300 ms, which is significantly higher than the inference time obtained using the YOLOv7-tiny. Furthermore, the YOLOv7-tiny model presents less computational complexity than the other models tested, having 6.2 million (M) parameters and 13.8 Giga (G) Floating Point Operations Per Second (FLOPS). This number of operations corresponds to around 36% fewer operations than YOLOv10-S, which is the second model tested with fewer FLOPS, as can be seen in <xref rid="sensors-24-06774-t005" ref-type="table">Table 5</xref>. Considering this, as a matter of the capability of implementation in a real-time system, the YOLOv7-tiny would be the preferred model of the four in order to ensure the system’s efficiency.</p><table-wrap id="sensors-24-06774-t005" position="float"><?disp-level 3?><label>Table 5</label><caption><p>Complexity comparison of the models tested.</p></caption><table frame="hsides" rules="groups"><thead><tr><th align="center" valign="middle" style="border-bottom:solid thin;border-top:solid thin" rowspan="1" colspan="1">Model</th><th align="center" valign="middle" style="border-bottom:solid thin;border-top:solid thin" rowspan="1" colspan="1">Number of Parameters</th><th align="center" valign="middle" style="border-bottom:solid thin;border-top:solid thin" rowspan="1" colspan="1">FLOPS</th></tr></thead><tbody><tr><td align="center" valign="middle" rowspan="1" colspan="1">YOLOv7-tiny</td><td align="center" valign="middle" rowspan="1" colspan="1">6.2 M</td><td align="center" valign="middle" rowspan="1" colspan="1">13.8 G</td></tr><tr><td align="center" valign="middle" rowspan="1" colspan="1">YOLOv8s</td><td align="center" valign="middle" rowspan="1" colspan="1">11.2 M</td><td align="center" valign="middle" rowspan="1" colspan="1">28.6 G</td></tr><tr><td align="center" valign="middle" rowspan="1" colspan="1">YOLOv9-S</td><td align="center" valign="middle" rowspan="1" colspan="1">7.1 M</td><td align="center" valign="middle" rowspan="1" colspan="1">26.4 G</td></tr><tr><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">YOLOv10-S</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">7.2 M</td><td align="center" valign="middle" style="border-bottom:solid thin" rowspan="1" colspan="1">21.6 G</td></tr></tbody></table></table-wrap></sec></sec><sec id="sec5-sensors-24-06774" disp-level="1"><title>5. Conclusions</title><p>This work allowed us to compare, side-by-side, four distinct YOLO models on the capability of successfully performing the task of node detection of a grapevine’s canes. The models were trained on the publicly available 3D2cut Single Guyot dataset, which contained an artificial background on the captured images. Furthermore, the models were tested on two different Portuguese grapevine datasets without artificial backgrounds, which allowed us to evaluate the robustness of the models on two distinct cultivars with different surrounding environments.</p><p>Considering this approach, it was possible to notice that YOLOv7-tiny achieved the best detection performance compared to the other used models. Furthermore, YOLOv7-tiny was the best model considering the trade-off between accuracy and inference speed. This detection model not only achieved overall better metrics compared to the state-of-the-art detection models, but also presented the fastest inference speed for both tested input sizes (640 × 640 px and 1280 × 1280 px).</p><p>Despite the significant performance variation between solutions, all the state-of-the-art YOLO models used are capable of detecting the nodes on grapevines even when additional environment information is present on the background of the frames behind the vines, which is an improvement to previously developed algorithms that lacked robustness to actuate in heterogeneous environments. This approach of node detection recurring to YOLO detection models fulfilled the objectives of presenting efficient and light-computing deep learning models for node detection on 2D images of grapevines. These models are capable of detecting nodes in cluttered environments while visualizing the entire grapevine, and they offer inference times suitable for real-time system implementation. Based on these features, the models demonstrate sufficient robustness for integration with decision-making algorithms to predict potential pruning locations on grapevine canes, even in highly heterogeneous settings. Notably, this performance is achieved without the need for artificial backgrounds to reduce background interference. Additionally, the low inference times, particularly for YOLOv7-tiny, enable the practical deployment of these algorithms on an autonomous pruning robot, allowing real-time operation without compromising system efficiency.</p><p>However, despite being proven the feasibility of implementing YOLO-detecting models on these types of environments, as future work, it would be interesting to perform the detections by firstly filtering the background of the images using data from a depth camera to remove unwanted information, which is expected to also improve inference times since the information available for inference will be reduced.</p></sec><sec id="ack1" sec-type="ack" disp-level="1"><title>Acknowledgments</title><p>The authors would like to acknowledge this work is co-financed by Component 5—Capitalization and Business Innovation, integrated in the Resilience Dimension of the Recovery and Resilience Plan within the scope of the Recovery and Resilience Mechanism (MRR) of the European Union (EU), framed in the Next-Generation EU, for the period 2021–2026, within project Vine&amp;Wine_PT, with reference 67. The authors would like to thank Centro de Estudos Vitivinícolas do Dão and Sogrape for making their vineyards available for data acquisition.</p></sec><sec id="glossary1" sec-type="glossary" disp-level="1"><title>Abbreviations</title><p>The following abbreviations are used in this manuscript:
</p><table-wrap position="anchor" id="array1"><table><tr><td align="left" valign="middle" rowspan="1" colspan="1">2D</td><td align="left" valign="middle" rowspan="1" colspan="1">Two-dimensional</td></tr><tr><td align="left" valign="middle" rowspan="1" colspan="1">AP</td><td align="left" valign="middle" rowspan="1" colspan="1">Average Precision</td></tr><tr><td align="left" valign="middle" rowspan="1" colspan="1">CIoU</td><td align="left" valign="middle" rowspan="1" colspan="1">Complete Intersection over Union</td></tr><tr><td align="left" valign="middle" rowspan="1" colspan="1">CNN</td><td align="left" valign="middle" rowspan="1" colspan="1">Convolutional Neural Network</td></tr><tr><td align="left" valign="middle" rowspan="1" colspan="1">CVAT</td><td align="left" valign="middle" rowspan="1" colspan="1">Computer Vision Annotation Tool</td></tr><tr><td align="left" valign="middle" rowspan="1" colspan="1">DFL</td><td align="left" valign="middle" rowspan="1" colspan="1">Distribution Focal Loss</td></tr><tr><td align="left" valign="middle" rowspan="1" colspan="1">FCN</td><td align="left" valign="middle" rowspan="1" colspan="1">Fully Convolutional Network</td></tr><tr><td align="left" valign="middle" rowspan="1" colspan="1">FLOPS</td><td align="left" valign="middle" rowspan="1" colspan="1">Floating Point Operations Per Second</td></tr><tr><td align="left" valign="middle" rowspan="1" colspan="1">GELAN</td><td align="left" valign="middle" rowspan="1" colspan="1">Generalized Efficient Layer Aggregation Network</td></tr><tr><td align="left" valign="middle" rowspan="1" colspan="1">mAP</td><td align="left" valign="middle" rowspan="1" colspan="1">Mean Average Precision</td></tr><tr><td align="left" valign="middle" rowspan="1" colspan="1">MS COCO</td><td align="left" valign="middle" rowspan="1" colspan="1">Microsoft Common Objects in Context</td></tr><tr><td align="left" valign="middle" rowspan="1" colspan="1">NMS</td><td align="left" valign="middle" rowspan="1" colspan="1">Non-Maximum Suppression</td></tr><tr><td align="left" valign="middle" rowspan="1" colspan="1">PGI</td><td align="left" valign="middle" rowspan="1" colspan="1">Programmable Gradient Information</td></tr><tr><td align="left" valign="middle" rowspan="1" colspan="1">SHG</td><td align="left" valign="middle" rowspan="1" colspan="1">Stacked Hourglass Network</td></tr><tr><td align="left" valign="middle" rowspan="1" colspan="1">YOLO</td><td align="left" valign="middle" rowspan="1" colspan="1">You Only Look Once</td></tr></table></table-wrap></sec><sec id="notes1" disp-level="1"><title>Author Contributions</title><p>Conceptualization, F.O., D.Q.d.S. and V.F.; Data curation, F.O. and D.Q.d.S.; Funding acquisition, F.N.d.S.; Investigation, F.O. and D.Q.d.S.; Methodology, F.O. and D.Q.d.S.; Project administration, F.N.d.S.; Software, F.O. and D.Q.d.S.; Supervision, T.M.P., J.B.C. and F.N.d.S.; Validation, V.F., T.M.P., M.C., J.B.C. and F.N.d.S.; Visualization, F.O.; Writing—original draft, F.O.; Writing—review and editing, D.Q.d.S., V.F., T.M.P., M.C., J.B.C. and F.N.d.S. All authors read and agreed to the published version of the manuscript.</p></sec><sec id="notes2" disp-level="1"><title>Institutional Review Board Statement</title><p>Not applicable.</p></sec><sec id="notes3" disp-level="1"><title>Informed Consent Statement</title><p>Not applicable.</p></sec><sec id="notes4" disp-level="1"><title>Data Availability Statement</title><p>The data presented in this study are openly available in the digital repository Zenodo: Douro &amp; Dão Grapevines Dataset for Node Detection—<ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://doi.org/10.5281/zenodo.10991688" ext-link-type="uri">https://doi.org/10.5281/zenodo.10991688</ext-link>.</p></sec><sec id="notes5" disp-level="1"><title>Conflicts of Interest</title><p>The authors declare no conflicts of interest. The funders had no role in the design of the study; in the collection, analyses, or interpretation of data; in the writing of the manuscript; or in the decision to publish the results.</p></sec><sec id="funding-statement1" xml:lang="en" disp-level="1"><title>Funding Statement</title><p>This research received no external funding.</p></sec><sec id="fn-group1" sec-type="fn-group" disp-level="1"><title>Footnotes</title><fn-group><fn id="fn1"><p><bold>Disclaimer/Publisher’s Note:</bold> The statements, opinions and data contained in all publications are solely those of the individual author(s) and contributor(s) and not of MDPI and/or the editor(s). MDPI and/or the editor(s) disclaim responsibility for any injury to people or property resulting from any ideas, methods, instructions or products referred to in the content.</p></fn></fn-group></sec><sec id="ref-list1" sec-type="ref-list" disp-level="1"><title>References</title><sec id="ref-list1_sec2" disp-level="2"><ref-list><ref id="B1-sensors-24-06774"><label>1.</label><mixed-citation><named-content content-type="citation-string">Botterill T., Paulin S., Green R.D., Williams S., Lin J., Saxton V., Mills S., Chen X., Corbett-Davies S. A Robot System for Pruning Grape Vines. J. Field Robot. 2017;34:1100–1122. doi: 10.1002/rob.21680.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1002/rob.21680"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=J. Field Robot.&amp;title=A Robot System for Pruning Grape Vines&amp;author=T. Botterill&amp;author=S. Paulin&amp;author=R.D. Green&amp;author=S. Williams&amp;author=J. Lin&amp;volume=34&amp;publication_year=2017&amp;pages=1100-1122&amp;doi=10.1002/rob.21680&amp;"/></mixed-citation></ref><ref id="B2-sensors-24-06774"><label>2.</label><mixed-citation><named-content content-type="citation-string">Poni S., Sabbatini P., Palliotti A. Facing Spring Frost Damage in Grapevine: Recent Developments and the Role of Delayed Winter Pruning—A Review. Am. J. Enol. Vitic. 2022;73:211–226. doi: 10.5344/ajev.2022.22011.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.5344/ajev.2022.22011"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Am. J. Enol. Vitic.&amp;title=Facing Spring Frost Damage in Grapevine: Recent Developments and the Role of Delayed Winter Pruning—A Review&amp;author=S. Poni&amp;author=P. Sabbatini&amp;author=A. Palliotti&amp;volume=73&amp;publication_year=2022&amp;pages=211-226&amp;doi=10.5344/ajev.2022.22011&amp;"/></mixed-citation></ref><ref id="B3-sensors-24-06774"><label>3.</label><mixed-citation><named-content content-type="citation-string">Reich L.  The Pruning Book. Taunton Press; Newtown, CT, USA: 2010. </named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="title=The Pruning Book&amp;author=L. Reich&amp;publication_year=2010&amp;"/></mixed-citation></ref><ref id="B4-sensors-24-06774"><label>4.</label><mixed-citation><named-content content-type="citation-string">Silwal A., Yandun F., Nellithimaru A., Bates T., Kantor G. Bumblebee: A Path Towards Fully Autonomous Robotic Vine Pruning. arXiv. 2021 doi: 10.55417/fr.2022051.2112.00291</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.55417/fr.2022051"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=arXiv&amp;title=Bumblebee: A Path Towards Fully Autonomous Robotic Vine Pruning&amp;author=A. Silwal&amp;author=F. Yandun&amp;author=A. Nellithimaru&amp;author=T. Bates&amp;author=G. Kantor&amp;publication_year=2021&amp;doi=10.55417/fr.2022051&amp;"/></mixed-citation></ref><ref id="B5-sensors-24-06774"><label>5.</label><mixed-citation><named-content content-type="citation-string">Williams H., Smith D., Shahabi J., Gee T., Nejati M., McGuinness B., Black K., Tobias J., Jangali R., Lim H., et al.  Modelling wine grapevines for autonomous robotic cane pruning. Biosyst. Eng. 2023;235:31–49. doi: 10.1016/j.biosystemseng.2023.09.006.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1016/j.biosystemseng.2023.09.006"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Biosyst. Eng.&amp;title=Modelling wine grapevines for autonomous robotic cane pruning&amp;author=H. Williams&amp;author=D. Smith&amp;author=J. Shahabi&amp;author=T. Gee&amp;author=M. Nejati&amp;volume=235&amp;publication_year=2023&amp;pages=31-49&amp;doi=10.1016/j.biosystemseng.2023.09.006&amp;"/></mixed-citation></ref><ref id="B6-sensors-24-06774"><label>6.</label><mixed-citation><named-content content-type="citation-string">Oliveira F., Tinoco V., Magalhães S., Santos F.N., Silva M.F. End-Effectors for Harvesting Manipulators—State of the Art Review; Proceedings of the 2022 IEEE International Conference on Autonomous Robot Systems and Competitions (ICARSC); Santa Maria da Feira, Portugal. 29–30 April 2022; pp. 98–103.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1109/ICARSC55462.2022.9784809"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Proceedings of the 2022 IEEE International Conference on Autonomous Robot Systems and Competitions (ICARSC)&amp;title=End-Effectors for Harvesting Manipulators—State of the Art Review&amp;author=F. Oliveira&amp;author=V. Tinoco&amp;author=S. Magalhães&amp;author=F.N. Santos&amp;author=M.F. Silva&amp;pages=98-103&amp;doi=10.1109/ICARSC55462.2022.9784809&amp;"/></mixed-citation></ref><ref id="B7-sensors-24-06774"><label>7.</label><mixed-citation><named-content content-type="citation-string">He L., Schupp J. Sensing and Automation in Pruning of Apple Trees: A Review. Agronomy. 2018;8:211.  doi: 10.3390/agronomy8100211.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.3390/agronomy8100211"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Agronomy&amp;title=Sensing and Automation in Pruning of Apple Trees: A Review&amp;author=L. He&amp;author=J. Schupp&amp;volume=8&amp;publication_year=2018&amp;pages=211&amp;doi=10.3390/agronomy8100211&amp;"/></mixed-citation></ref><ref id="B8-sensors-24-06774"><label>8.</label><mixed-citation><named-content content-type="citation-string">Collins C., Wang X., Lesefko S., De Bei R., Fuentes S. Effects of canopy management practices on grapevine bud fruitfulness. OENO ONE. 2020;54:313–325. doi: 10.20870/oeno-one.2020.54.2.3016.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.20870/oeno-one.2020.54.2.3016"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=OENO ONE&amp;title=Effects of canopy management practices on grapevine bud fruitfulness&amp;author=C. Collins&amp;author=X. Wang&amp;author=S. Lesefko&amp;author=R. De Bei&amp;author=S. Fuentes&amp;volume=54&amp;publication_year=2020&amp;pages=313-325&amp;doi=10.20870/oeno-one.2020.54.2.3016&amp;"/></mixed-citation></ref><ref id="B9-sensors-24-06774"><label>9.</label><mixed-citation><named-content content-type="citation-string">Cuevas-Velasquez H., Gallego A.J., Tylecek R., Hemming J., Van Tuijl B., Mencarelli A., Fisher R.B. Real-time Stereo Visual Servoing for Rose Pruning with Robotic Arm; Proceedings of the 2020 IEEE International Conference on Robotics and Automation (ICRA); Paris, France. 31 May–31 August 2020; pp. 7050–7056.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1109/ICRA40945.2020.9197272"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Proceedings of the 2020 IEEE International Conference on Robotics and Automation (ICRA)&amp;title=Real-time Stereo Visual Servoing for Rose Pruning with Robotic Arm&amp;author=H. Cuevas-Velasquez&amp;author=A.J. Gallego&amp;author=R. Tylecek&amp;author=J. Hemming&amp;author=B. Van Tuijl&amp;pages=7050-7056&amp;doi=10.1109/ICRA40945.2020.9197272&amp;"/></mixed-citation></ref><ref id="B10-sensors-24-06774"><label>10.</label><mixed-citation><named-content content-type="citation-string">Villegas Marset W., Pérez D.S., Díaz C.A., Bromberg F. Towards practical 2D grapevine bud detection with fully convolutional networks. Comput. Electron. Agric. 2021;182:105947. doi: 10.1016/j.compag.2020.105947.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1016/j.compag.2020.105947"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Comput. Electron. Agric.&amp;title=Towards practical 2D grapevine bud detection with fully convolutional networks&amp;author=W. Villegas Marset&amp;author=D.S. Pérez&amp;author=C.A. Díaz&amp;author=F. Bromberg&amp;volume=182&amp;publication_year=2021&amp;pages=105947&amp;doi=10.1016/j.compag.2020.105947&amp;"/></mixed-citation></ref><ref id="B11-sensors-24-06774"><label>11.</label><mixed-citation><named-content content-type="citation-string">Gentilhomme T., Villamizar M., Corre J., Odobez J.M. Towards smart pruning: ViNet, a deep-learning approach for grapevine structure estimation. Comput. Electron. Agric. 2023;207:107736. doi: 10.1016/j.compag.2023.107736.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1016/j.compag.2023.107736"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Comput. Electron. Agric.&amp;title=Towards smart pruning: ViNet, a deep-learning approach for grapevine structure estimation&amp;author=T. Gentilhomme&amp;author=M. Villamizar&amp;author=J. Corre&amp;author=J.M. Odobez&amp;volume=207&amp;publication_year=2023&amp;pages=107736&amp;doi=10.1016/j.compag.2023.107736&amp;"/></mixed-citation></ref><ref id="B12-sensors-24-06774"><label>12.</label><mixed-citation><named-content content-type="citation-string">Redmon J., Divvala S., Girshick R., Farhadi A. You Only Look Once: Unified, Real-Time Object Detection; Proceedings of the 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR); Las Vegas, NV, USA. 27–30 June 2016; pp. 779–788.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1109/CVPR.2016.91"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Proceedings of the 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)&amp;title=You Only Look Once: Unified, Real-Time Object Detection&amp;author=J. Redmon&amp;author=S. Divvala&amp;author=R. Girshick&amp;author=A. Farhadi&amp;pages=779-788&amp;doi=10.1109/CVPR.2016.91&amp;"/></mixed-citation></ref><ref id="B13-sensors-24-06774"><label>13.</label><mixed-citation><named-content content-type="citation-string">Wang C., Bochkovskiy A., Liao H. YOLOv7: Trainable Bag-of-Freebies Sets New State-of-the-Art for Real-Time Object Detectors; Proceedings of the 2023 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR); Vancouver, BC, Canada. 17–24 June 2023; pp. 7464–7475.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1109/CVPR52729.2023.00721"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Proceedings of the 2023 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)&amp;title=YOLOv7: Trainable Bag-of-Freebies Sets New State-of-the-Art for Real-Time Object Detectors&amp;author=C. Wang&amp;author=A. Bochkovskiy&amp;author=H. Liao&amp;pages=7464-7475&amp;doi=10.1109/CVPR52729.2023.00721&amp;"/></mixed-citation></ref><ref id="B14-sensors-24-06774"><label>14.</label><mixed-citation><named-content content-type="citation-string">Jocher G., Chaurasia A., Qiu J. Ultralytics YOLOv8. 2023.  [(accessed on 17 October 2024)].  Available online:  <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://docs.ultralytics.com/pt/models/yolov8/" ext-link-type="uri">https://docs.ultralytics.com/pt/models/yolov8/</ext-link></named-content></mixed-citation></ref><ref id="B15-sensors-24-06774"><label>15.</label><mixed-citation><named-content content-type="citation-string">Wang C.Y., Liao H.Y.M. YOLOv9: Learning What You Want to Learn Using Programmable Gradient Information. arXiv. 20242402.13616</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=arXiv&amp;title=YOLOv9: Learning What You Want to Learn Using Programmable Gradient Information&amp;author=C.Y. Wang&amp;author=H.Y.M. Liao&amp;publication_year=2024&amp;"/></mixed-citation></ref><ref id="B16-sensors-24-06774"><label>16.</label><mixed-citation><named-content content-type="citation-string">Wang A., Chen H., Liu L., Chen K., Lin Z., Han J., Ding G. YOLOv10: Real-Time End-to-End Object Detection. arXiv. 20242405.14458</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=arXiv&amp;title=YOLOv10: Real-Time End-to-End Object Detection&amp;author=A. Wang&amp;author=H. Chen&amp;author=L. Liu&amp;author=K. Chen&amp;author=Z. Lin&amp;publication_year=2024&amp;"/></mixed-citation></ref><ref id="B17-sensors-24-06774"><label>17.</label><mixed-citation><named-content content-type="citation-string">Lavrador da Silva A., João Fernão-Pires M., Bianchi-de-Aguiar F. Portuguese vines and wines: Heritage, quality symbol, tourism asset. Ciência Téc. Vitiv. 2018;33:31–46. doi: 10.1051/ctv/20183301031.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1051/ctv/20183301031"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Ciência Téc. Vitiv.&amp;title=Portuguese vines and wines: Heritage, quality symbol, tourism asset&amp;author=A. Lavrador da Silva&amp;author=M. João Fernão-Pires&amp;author=F. Bianchi-de-Aguiar&amp;volume=33&amp;publication_year=2018&amp;pages=31-46&amp;doi=10.1051/ctv/20183301031&amp;"/></mixed-citation></ref><ref id="B18-sensors-24-06774"><label>18.</label><mixed-citation><named-content content-type="citation-string">Oliveira F.A., Silva D.Q.  Douro &amp; Dão Grapevines Dataset for Node Detection. CERN Data Centre; Prévessin-Moëns, France: 2024. </named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.5281/zenodo.10991688"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="title=Douro &amp; Dão Grapevines Dataset for Node Detection&amp;author=F.A. Oliveira&amp;author=D.Q. Silva&amp;publication_year=2024&amp;"/></mixed-citation></ref><ref id="B19-sensors-24-06774"><label>19.</label><mixed-citation><named-content content-type="citation-string">Casas G.G., Ismail Z.H., Limeira M.M.C., da Silva A.A.L., Leite H.G. Automatic Detection and Counting of Stacked Eucalypt Timber Using the YOLOv8 Model. Forests. 2023;14:2369.  doi: 10.3390/f14122369.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.3390/f14122369"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Forests&amp;title=Automatic Detection and Counting of Stacked Eucalypt Timber Using the YOLOv8 Model&amp;author=G.G. Casas&amp;author=Z.H. Ismail&amp;author=M.M.C. Limeira&amp;author=A.A.L. da Silva&amp;author=H.G. Leite&amp;volume=14&amp;publication_year=2023&amp;pages=2369&amp;doi=10.3390/f14122369&amp;"/></mixed-citation></ref><ref id="B20-sensors-24-06774"><label>20.</label><mixed-citation><named-content content-type="citation-string">Xie S., Sun H. Tea-YOLOv8s: A Tea Bud Detection Model Based on Deep Learning and Computer Vision. Sensors. 2023;23:6576.  doi: 10.3390/s23146576.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.3390/s23146576"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC10383684"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="37514870"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Sensors&amp;title=Tea-YOLOv8s: A Tea Bud Detection Model Based on Deep Learning and Computer Vision&amp;author=S. Xie&amp;author=H. Sun&amp;volume=23&amp;publication_year=2023&amp;pages=6576&amp;pmid=37514870&amp;doi=10.3390/s23146576&amp;"/></mixed-citation></ref><ref id="B21-sensors-24-06774"><label>21.</label><mixed-citation><named-content content-type="citation-string">Lin T.Y., Maire M., Belongie S., Bourdev L., Girshick R., Hays J., Perona P., Ramanan D., Zitnick C.L., Dollár P. Microsoft COCO: Common Objects in Context. arXiv. 20151405.0312</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=arXiv&amp;title=Microsoft COCO: Common Objects in Context&amp;author=T.Y. Lin&amp;author=M. Maire&amp;author=S. Belongie&amp;author=L. Bourdev&amp;author=R. Girshick&amp;publication_year=2015&amp;"/></mixed-citation></ref><ref id="B22-sensors-24-06774"><label>22.</label><mixed-citation><named-content content-type="citation-string">Terven J., Córdova-Esparza D.M., Romero-González J.A. A Comprehensive Review of YOLO Architectures in Computer Vision: From YOLOv1 to YOLOv8 and YOLO-NAS. Mach. Learn. Knowl. Extr. 2023;5:1680–1716. doi: 10.3390/make5040083.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.3390/make5040083"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Mach. Learn. Knowl. Extr.&amp;title=A Comprehensive Review of YOLO Architectures in Computer Vision: From YOLOv1 to YOLOv8 and YOLO-NAS&amp;author=J. Terven&amp;author=D.M. Córdova-Esparza&amp;author=J.A. Romero-González&amp;volume=5&amp;publication_year=2023&amp;pages=1680-1716&amp;doi=10.3390/make5040083&amp;"/></mixed-citation></ref><ref id="B23-sensors-24-06774"><label>23.</label><mixed-citation><named-content content-type="citation-string">Jocher G.  YOLOv5 by Ultralytics. CERN Data Centre; Prévessin-Moëns, France: 2020. </named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.5281/zenodo.3908559"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="title=YOLOv5 by Ultralytics&amp;author=G. Jocher&amp;publication_year=2020&amp;"/></mixed-citation></ref><ref id="B24-sensors-24-06774"><label>24.</label><mixed-citation><named-content content-type="citation-string">Li X., Wang W., Wu L., Chen S., Hu X., Li J., Tang J., Yang J. Generalized focal loss: Learning qualified and distributed bounding boxes for dense object detection; Proceedings of the 34th International Conference on Neural Information Processing Systems; Red Hook, NY, USA. 6–12 December 2020;  NIPS ’20.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Proceedings of the 34th International Conference on Neural Information Processing Systems&amp;title=Generalized focal loss: Learning qualified and distributed bounding boxes for dense object detection&amp;author=X. Li&amp;author=W. Wang&amp;author=L. Wu&amp;author=S. Chen&amp;author=X. Hu&amp;"/></mixed-citation></ref><ref id="B25-sensors-24-06774"><label>25.</label><mixed-citation><named-content content-type="citation-string">Zheng Z., Wang P., Liu W., Li J., Ye R., Ren D. Distance-IoU Loss: Faster and Better Learning for Bounding Box Regression. Proc. AAAI Conf. Artif. Intell. 2020;34:12993–13000. doi: 10.1609/aaai.v34i07.6999.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1609/aaai.v34i07.6999"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Proc. AAAI Conf. Artif. Intell.&amp;title=Distance-IoU Loss: Faster and Better Learning for Bounding Box Regression&amp;author=Z. Zheng&amp;author=P. Wang&amp;author=W. Liu&amp;author=J. Li&amp;author=R. Ye&amp;volume=34&amp;publication_year=2020&amp;pages=12993-13000&amp;doi=10.1609/aaai.v34i07.6999&amp;"/></mixed-citation></ref><ref id="B26-sensors-24-06774"><label>26.</label><mixed-citation><named-content content-type="citation-string">Vougioukas S.G. Agricultural Robotics. Annu. Rev. Control. Robot. Auton. Syst. 2019;2:365–392. doi: 10.1146/annurev-control-053018-023617.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1146/annurev-control-053018-023617"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Annu. Rev. Control. Robot. Auton. Syst.&amp;title=Agricultural Robotics&amp;author=S.G. Vougioukas&amp;volume=2&amp;publication_year=2019&amp;pages=365-392&amp;doi=10.1146/annurev-control-053018-023617&amp;"/></mixed-citation></ref><ref id="B27-sensors-24-06774"><label>27.</label><mixed-citation><named-content content-type="citation-string">Bechar A., Vigneault C. Agricultural robots for field operations: Concepts and components. Biosyst. Eng. 2016;149:94–111. doi: 10.1016/j.biosystemseng.2016.06.014.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1016/j.biosystemseng.2016.06.014"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Biosyst. Eng.&amp;title=Agricultural robots for field operations: Concepts and components&amp;author=A. Bechar&amp;author=C. Vigneault&amp;volume=149&amp;publication_year=2016&amp;pages=94-111&amp;doi=10.1016/j.biosystemseng.2016.06.014&amp;"/></mixed-citation></ref><ref id="B28-sensors-24-06774"><label>28.</label><mixed-citation><named-content content-type="citation-string">Hsueh B.Y., Li W., Wu I.C. Stochastic Gradient Descent With Hyperbolic-Tangent Decay on Classification; Proceedings of the 2019 IEEE Winter Conference on Applications of Computer Vision (WACV); Waikoloa, HI, USA. 7–11 January 2019; pp. 435–442.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1109/WACV.2019.00052"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Proceedings of the 2019 IEEE Winter Conference on Applications of Computer Vision (WACV)&amp;title=Stochastic Gradient Descent With Hyperbolic-Tangent Decay on Classification&amp;author=B.Y. Hsueh&amp;author=W. Li&amp;author=I.C. Wu&amp;pages=435-442&amp;doi=10.1109/WACV.2019.00052&amp;"/></mixed-citation></ref><ref id="B29-sensors-24-06774"><label>29.</label><mixed-citation><named-content content-type="citation-string">Prechelt L.  Early Stopping—But When? In: Orr G.B., Müller K.R., editors. Neural Networks: Tricks of the Trade. Springer; Berlin/Heidelberg, Germany: 1998. pp. 55–69.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1007/3-540-49430-8_3"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="title=Neural Networks: Tricks of the Trade&amp;author=L. Prechelt&amp;publication_year=1998&amp;"/></mixed-citation></ref><ref id="B30-sensors-24-06774"><label>30.</label><mixed-citation><named-content content-type="citation-string">Butko N.J., Movellan J.R. Optimal scanning for faster object detection; Proceedings of the 2009 IEEE Conference on Computer Vision and Pattern Recognition; Miami, FL, USA. 20–25 June 2009; pp. 2751–2758.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1109/CVPR.2009.5206540"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Proceedings of the 2009 IEEE Conference on Computer Vision and Pattern Recognition&amp;title=Optimal scanning for faster object detection&amp;author=N.J. Butko&amp;author=J.R. Movellan&amp;pages=2751-2758&amp;doi=10.1109/CVPR.2009.5206540&amp;"/></mixed-citation></ref><ref id="B31-sensors-24-06774"><label>31.</label><mixed-citation><named-content content-type="citation-string">Wenkel S., Alhazmi K., Liiv T., Alrshoud S., Simon M. Confidence Score: The Forgotten Dimension of Object Detection Performance Evaluation. Sensors. 2021;21:4350.  doi: 10.3390/s21134350.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.3390/s21134350"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC8271464"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="34202089"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Sensors&amp;title=Confidence Score: The Forgotten Dimension of Object Detection Performance Evaluation&amp;author=S. Wenkel&amp;author=K. Alhazmi&amp;author=T. Liiv&amp;author=S. Alrshoud&amp;author=M. Simon&amp;volume=21&amp;publication_year=2021&amp;pages=4350&amp;pmid=34202089&amp;doi=10.3390/s21134350&amp;"/></mixed-citation></ref><ref id="B32-sensors-24-06774"><label>32.</label><mixed-citation><named-content content-type="citation-string">Zahid A., Mahmud M.S., He L., Heinemann P., Choi D., Schupp J. Technological advancements towards developing a robotic pruner for apple trees: A review. Comput. Electron. Agric. 2021;189:106383. doi: 10.1016/j.compag.2021.106383.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1016/j.compag.2021.106383"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Comput. Electron. Agric.&amp;title=Technological advancements towards developing a robotic pruner for apple trees: A review&amp;author=A. Zahid&amp;author=M.S. Mahmud&amp;author=L. He&amp;author=P. Heinemann&amp;author=D. Choi&amp;volume=189&amp;publication_year=2021&amp;pages=106383&amp;doi=10.1016/j.compag.2021.106383&amp;"/></mixed-citation></ref></ref-list></sec></sec><sec id="_ad93_" xml:lang="en" sec-type="associated-data" disp-level="1"><title>Associated Data</title><sec id="_adda93_" xml:lang="en" sec-type="data-availability-statement" disp-level="2"><title>Data Availability Statement</title><p>The data presented in this study are openly available in the digital repository Zenodo: Douro &amp; Dão Grapevines Dataset for Node Detection—<ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://doi.org/10.5281/zenodo.10991688" ext-link-type="uri">https://doi.org/10.5281/zenodo.10991688</ext-link>.</p></sec></sec></body></article>