
<!DOCTYPE article
  PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD with MathML3 v1.4 20241031//EN" "JATS-archivearticle1-4-mathml3.dtd">
<article article-type="research-article" xml:lang="en" dtd-version="1.4"><?da-xref-anchor-style superscripted?><processing-meta base-tagset="archiving" mathml-version="3.0" table-model="xhtml" tagset-family="jats"><restricted-by>pmc</restricted-by></processing-meta><front><journal-meta><journal-id journal-id-type="nlm-ta">Nat Commun</journal-id><journal-id journal-id-type="iso-abbrev">Nat Commun</journal-id><journal-id journal-id-type="pmc-domain-id">2873</journal-id><journal-id journal-id-type="pmc-domain">ncomms</journal-id><journal-id journal-id-type="nlm-id">101528555</journal-id><journal-title-group><journal-title>Nature Communications</journal-title></journal-title-group><issn pub-type="epub">2041-1723</issn><?publisher_abbrev naturepg?><publisher><publisher-name>Nature Publishing Group</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="pmcid">PMC13376813</article-id><article-id pub-id-type="pmcid-ver">PMC13376813.1</article-id><article-id pub-id-type="pmcaid">13376813</article-id><article-id pub-id-type="pmcaiid">13376813</article-id><article-id pub-id-type="pmid">41932896</article-id><article-id pub-id-type="doi">10.1038/s41467-026-71090-y</article-id><article-id pub-id-type="publisher-id">71090</article-id><article-version article-version-type="pmc-version">1</article-version><article-categories><subj-group subj-group-type="heading"><subject>Article</subject></subj-group></article-categories><title-group><article-title>A conversational multi-agent AI system for automated plant phenotyping</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><contrib-id contrib-id-type="orcid" authenticated="false">http://orcid.org/0000-0003-2915-599X</contrib-id><name name-style="western"><surname>Chen</surname><given-names initials="F">Feng</given-names></name><address><email>Feng.Chen@ed.ac.uk</email></address><xref ref-type="aff" rid="Aff1">1</xref></contrib><contrib contrib-type="author"><contrib-id contrib-id-type="orcid" authenticated="false">http://orcid.org/0009-0005-5803-1138</contrib-id><name name-style="western"><surname>Stogiannidis</surname><given-names initials="I">Ilias</given-names></name><xref ref-type="aff" rid="Aff1">1</xref></contrib><contrib contrib-type="author"><contrib-id contrib-id-type="orcid" authenticated="false">http://orcid.org/0000-0003-1343-0774</contrib-id><name name-style="western"><surname>Wood</surname><given-names initials="A">Andrew</given-names></name><xref ref-type="aff" rid="Aff1">1</xref></contrib><contrib contrib-type="author"><contrib-id contrib-id-type="orcid" authenticated="false">http://orcid.org/0009-0008-3036-7010</contrib-id><name name-style="western"><surname>Bueno</surname><given-names initials="D">Danilo</given-names></name><xref ref-type="aff" rid="Aff1">1</xref></contrib><contrib contrib-type="author"><contrib-id contrib-id-type="orcid" authenticated="false">http://orcid.org/0000-0001-8908-6696</contrib-id><name name-style="western"><surname>Williams</surname><given-names initials="D">Dominic</given-names></name><xref ref-type="aff" rid="Aff2">2</xref></contrib><contrib contrib-type="author"><contrib-id contrib-id-type="orcid" authenticated="false">http://orcid.org/0000-0002-7411-1446</contrib-id><name name-style="western"><surname>Macfarlane</surname><given-names initials="F">Fraser</given-names></name><xref ref-type="aff" rid="Aff2">2</xref></contrib><contrib contrib-type="author"><contrib-id contrib-id-type="orcid" authenticated="false">http://orcid.org/0000-0002-5130-3592</contrib-id><name name-style="western"><surname>Grieve</surname><given-names initials="BD">Bruce D.</given-names></name><xref ref-type="aff" rid="Aff3">3</xref></contrib><contrib contrib-type="author"><contrib-id contrib-id-type="orcid" authenticated="false">http://orcid.org/0000-0002-4246-4909</contrib-id><name name-style="western"><surname>Wells</surname><given-names initials="D">Darren</given-names></name><xref ref-type="aff" rid="Aff4">4</xref></contrib><contrib contrib-type="author"><contrib-id contrib-id-type="orcid" authenticated="false">http://orcid.org/0000-0003-2815-0812</contrib-id><name name-style="western"><surname>Atkinson</surname><given-names initials="JA">Jonathan A.</given-names></name><xref ref-type="aff" rid="Aff4">4</xref></contrib><contrib contrib-type="author"><contrib-id contrib-id-type="orcid" authenticated="false">http://orcid.org/0000-0001-8759-3969</contrib-id><name name-style="western"><surname>Hawkesford</surname><given-names initials="MJ">Malcolm J.</given-names></name><xref ref-type="aff" rid="Aff5">5</xref></contrib><contrib contrib-type="author"><contrib-id contrib-id-type="orcid" authenticated="false">http://orcid.org/0000-0003-2141-4707</contrib-id><name name-style="western"><surname>Rolfe</surname><given-names initials="SA">Stephen A.</given-names></name><xref ref-type="aff" rid="Aff6">6</xref></contrib><contrib contrib-type="author"><contrib-id contrib-id-type="orcid" authenticated="false">http://orcid.org/0000-0002-4073-7221</contrib-id><name name-style="western"><surname>Lawson</surname><given-names initials="T">Tracy</given-names></name><xref ref-type="aff" rid="Aff7">7</xref><xref ref-type="aff" rid="Aff8">8</xref><xref ref-type="aff" rid="Aff9">9</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Pridmore</surname><given-names initials="T">Tony</given-names></name><xref ref-type="aff" rid="Aff10">10</xref></contrib><contrib contrib-type="author" corresp="yes"><contrib-id contrib-id-type="orcid" authenticated="false">http://orcid.org/0000-0002-8795-9294</contrib-id><name name-style="western"><surname>Tsaftaris</surname><given-names initials="SA">Sotirios A.</given-names></name><address><email>S.Tsaftaris@ed.ac.uk</email></address><xref ref-type="aff" rid="Aff1">1</xref><xref ref-type="aff" rid="Aff11">11</xref></contrib><contrib contrib-type="author" corresp="yes"><contrib-id contrib-id-type="orcid" authenticated="false">http://orcid.org/0000-0002-5232-677X</contrib-id><name name-style="western"><surname>Giuffrida</surname><given-names initials="MV">Mario Valerio</given-names></name><address><email>Valerio.Giuffrida@nottingham.ac.uk</email></address><xref ref-type="aff" rid="Aff10">10</xref></contrib><aff id="Aff1"><label>1</label><institution-wrap><institution-id institution-id-type="ROR">https://ror.org/01nrxwf90</institution-id><institution-id institution-id-type="GRID">grid.4305.2</institution-id><institution-id institution-id-type="ISNI">0000 0004 1936 7988</institution-id><institution>Institute for Imaging, Data and Communications (IDCOM), School of Engineering, </institution><institution>University of Edinburgh, </institution></institution-wrap>Edinburgh, UK </aff><aff id="Aff2"><label>2</label><institution-wrap><institution-id institution-id-type="ROR">https://ror.org/03rzp5127</institution-id><institution-id institution-id-type="GRID">grid.43641.34</institution-id><institution-id institution-id-type="ISNI">0000 0001 1014 6626</institution-id><institution>James Hutton Institute, </institution></institution-wrap>Dundee, UK </aff><aff id="Aff3"><label>3</label><institution-wrap><institution-id institution-id-type="ROR">https://ror.org/027m9bs27</institution-id><institution-id institution-id-type="GRID">grid.5379.8</institution-id><institution-id institution-id-type="ISNI">0000 0001 2166 2407</institution-id><institution>Department of Electrical and Electronic Engineering, </institution><institution>University of Manchester, </institution></institution-wrap>Manchester, UK </aff><aff id="Aff4"><label>4</label><institution-wrap><institution-id institution-id-type="ROR">https://ror.org/01ee9ar58</institution-id><institution-id institution-id-type="GRID">grid.4563.4</institution-id><institution-id institution-id-type="ISNI">0000 0004 1936 8868</institution-id><institution>School of Biosciences, </institution><institution>University of Nottingham, </institution></institution-wrap>Nottingham, UK </aff><aff id="Aff5"><label>5</label><institution-wrap><institution-id institution-id-type="ROR">https://ror.org/0347fy350</institution-id><institution-id institution-id-type="GRID">grid.418374.d</institution-id><institution-id institution-id-type="ISNI">0000 0001 2227 9389</institution-id><institution>Rothamsted Research, </institution></institution-wrap>Harpenden, UK </aff><aff id="Aff6"><label>6</label><institution-wrap><institution-id institution-id-type="ROR">https://ror.org/05krs5044</institution-id><institution-id institution-id-type="GRID">grid.11835.3e</institution-id><institution-id institution-id-type="ISNI">0000 0004 1936 9262</institution-id><institution>School of Biosciences, </institution><institution>University of Sheffield, </institution></institution-wrap>Sheffield, UK </aff><aff id="Aff7"><label>7</label><institution-wrap><institution-id institution-id-type="ROR">https://ror.org/047426m28</institution-id><institution-id institution-id-type="GRID">grid.35403.31</institution-id><institution-id institution-id-type="ISNI">0000 0004 1936 9991</institution-id><institution>Department of Plant Biology and Department of Crop Sciences, </institution><institution>University of Illinois Urbana-Champaign, </institution></institution-wrap>Urbana, IL USA </aff><aff id="Aff8"><label>8</label><institution-wrap><institution-id institution-id-type="ROR">https://ror.org/047426m28</institution-id><institution-id institution-id-type="GRID">grid.35403.31</institution-id><institution-id institution-id-type="ISNI">0000 0004 1936 9991</institution-id><institution>Institute for Genomic Biology, </institution><institution>University of Illinois Urbana-Champaign, </institution></institution-wrap>Urbana, IL USA </aff><aff id="Aff9"><label>9</label><institution-wrap><institution-id institution-id-type="ROR">https://ror.org/02nkf1q06</institution-id><institution-id institution-id-type="GRID">grid.8356.8</institution-id><institution-id institution-id-type="ISNI">0000 0001 0942 6946</institution-id><institution>School of Life Sciences, </institution><institution>University of Essex, </institution></institution-wrap>Colchester, UK </aff><aff id="Aff10"><label>10</label><institution-wrap><institution-id institution-id-type="ROR">https://ror.org/01ee9ar58</institution-id><institution-id institution-id-type="GRID">grid.4563.4</institution-id><institution-id institution-id-type="ISNI">0000 0004 1936 8868</institution-id><institution>School of Computer Science, </institution><institution>University of Nottingham, </institution></institution-wrap>Nottingham, UK </aff><aff id="Aff11"><label>11</label>Causality in Healthcare AI Hub (CHAI), Edinburgh, UK </aff></contrib-group><pub-date pub-type="epub"><day>3</day><month>4</month><year>2026</year></pub-date><pub-date pub-type="collection"><year>2026</year></pub-date><volume>17</volume><issue-id pub-id-type="pmc-issue-id">503853</issue-id><elocation-id>6391</elocation-id><history><date date-type="received"><day>1</day><month>5</month><year>2025</year></date><date date-type="accepted"><day>12</day><month>3</month><year>2026</year></date></history><pub-history><event event-type="pmc-release"><date><day>03</day><month>04</month><year>2026</year></date></event><event event-type="pmc-live"><date><day>18</day><month>07</month><year>2026</year></date></event><event event-type="pmc-last-change"><date iso-8601-date="2026-08-19 11:25:18.120"><day>19</day><month>08</month><year>2026</year></date></event></pub-history><permissions><copyright-statement>© The Author(s) 2026</copyright-statement><copyright-year>2026</copyright-year><license><ali:license_ref xmlns:ali="http://www.niso.org/schemas/ali/1.0/" specific-use="textmining" content-type="ccbylicense">https://creativecommons.org/licenses/by/4.0/</ali:license_ref><license-p><bold>Open Access</bold> This article is licensed under a Creative Commons Attribution 4.0 International License, which permits use, sharing, adaptation, distribution and reproduction in any medium or format, as long as you give appropriate credit to the original author(s) and the source, provide a link to the Creative Commons licence, and indicate if changes were made. The images or other third party material in this article are included in the article’s Creative Commons licence, unless indicated otherwise in a credit line to the material. If material is not included in the article’s Creative Commons licence and your intended use is not permitted by statutory regulation or exceeds the permitted use, you will need to obtain permission directly from the copyright holder. To view a copy of this licence, visit <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">http://creativecommons.org/licenses/by/4.0/</ext-link>.</license-p></license></permissions><self-uri xmlns:xlink="http://www.w3.org/1999/xlink" content-type="pmc-pdf" xlink:href="41467_2026_Article_71090.pdf"><?pdf-name 41467_2026_Article_71090.pdf?><?pdf-size 3146400?><?pdf-md5 46668486061208d7cc7290a06eb04238?><?pdf-image-server-status NEVER_LOAD?><?pdf-cloudpmc-urn urn:app:f947/13376813/466684860612/41467_2026_Article_71090.pdf?></self-uri><abstract id="Abs1"><p id="Par1">Plant phenotyping increasingly relies on (semi-)automated image-based analysis workflows to improve its accuracy and scalability. However, many existing solutions remain overly complex, difficult to reimplement and maintain, and pose high barriers for users without substantial computational expertise. To address these challenges, we introduce PhenoAssistant: a pioneering AI-driven system that streamlines plant phenotyping via intuitive natural language interaction. PhenoAssistant leverages a large language model to orchestrate a curated toolkit supporting tasks including automated phenotype extraction, data visualisation and automated model training. We validate PhenoAssistant through several representative case studies and a set of evaluation tasks. By lowering technical hurdles, PhenoAssistant underscores the promise of AI-driven methodologies to democratising AI adoption in plant biology.</p></abstract><abstract id="Abs2" abstract-type="web-summary"><p id="Par2">Application of large language models (LLMs) for automating complex and data-intensive crop phenotype analysis remains unexplored. Here, the authors integrate LLM with curated toolkit for phenotype extraction, visualization, and model training and show its ability to reduce technical barriers and enhance accessibility.</p></abstract><kwd-group kwd-group-type="npg-subject"><title>Subject terms</title><kwd>Natural variation in plants</kwd><kwd>Optical imaging</kwd><kwd>Information technology</kwd><kwd>Machine learning</kwd></kwd-group><funding-group><award-group><funding-source><institution-wrap><institution-id institution-id-type="FundRef">https://doi.org/10.13039/501100000268</institution-id><institution>RCUK | Biotechnology and Biological Sciences Research Council (BBSRC)</institution></institution-wrap></funding-source><award-id>BB/Y512333/1</award-id><award-id>BB/Y512333/1</award-id><award-id>BB/Y512333/1</award-id><award-id>BB/Y512333/1</award-id><award-id>BB/Y512333/1</award-id><award-id>BB/Y512333/1</award-id><award-id>BB/Y512333/1</award-id><award-id>BB/Y512333/1</award-id><award-id>BB/Y512333/1</award-id><award-id>BB/Y512333/1</award-id><award-id>BB/Y512333/1</award-id><award-id>BB/X011003/1</award-id><award-id>BB/Y512333/1</award-id><award-id>BB/Y512333/1</award-id><award-id>BB/Y512333/1</award-id><award-id>BB/Y512333/1</award-id><principal-award-recipient><name name-style="western"><surname>Chen</surname><given-names>Feng</given-names></name><name name-style="western"><surname>Stogiannidis</surname><given-names>Ilias</given-names></name><name name-style="western"><surname>Wood</surname><given-names>Andrew</given-names></name><name name-style="western"><surname>Bueno</surname><given-names>Danilo</given-names></name><name name-style="western"><surname>Williams</surname><given-names>Dominic</given-names></name><name name-style="western"><surname>Macfarlane</surname><given-names>Fraser</given-names></name><name name-style="western"><surname>Grieve</surname><given-names>Bruce D.</given-names></name><name name-style="western"><surname>Wells</surname><given-names>Darren</given-names></name><name name-style="western"><surname>Atkinson</surname><given-names>Jonathan A.</given-names></name><name name-style="western"><surname>Hawkesford</surname><given-names>Malcolm J.</given-names></name><name name-style="western"><surname>Rolfe</surname><given-names>Stephen A.</given-names></name><name name-style="western"><surname>Lawson</surname><given-names>Tracy</given-names></name><name name-style="western"><surname>Pridmore</surname><given-names>Tony</given-names></name><name name-style="western"><surname>Tsaftaris</surname><given-names>Sotirios A.</given-names></name><name name-style="western"><surname>Giuffrida</surname><given-names>Mario Valerio</given-names></name></principal-award-recipient></award-group></funding-group><funding-group><award-group><funding-source><institution>Microsoft Accelerating Foundation Models Research (AFMR): Agricultural Foundation Models via Domain-Specific Pre-Training</institution></funding-source></award-group></funding-group><funding-group><award-group><funding-source><institution-wrap><institution-id institution-id-type="FundRef">https://doi.org/10.13039/501100000266</institution-id><institution>RCUK | Engineering and Physical Sciences Research Council (EPSRC)</institution></institution-wrap></funding-source><award-id>EP/X033686/1</award-id><award-id>EP/Y028856/1</award-id><principal-award-recipient><name name-style="western"><surname>Chen</surname><given-names>Feng</given-names></name><name name-style="western"><surname>Tsaftaris</surname><given-names>Sotirios A.</given-names></name></principal-award-recipient></award-group></funding-group><custom-meta-group><custom-meta><meta-name>pmc-status-qastatus</meta-name><meta-value>0</meta-value></custom-meta><custom-meta><meta-name>pmc-status-live</meta-name><meta-value>yes</meta-value></custom-meta><custom-meta><meta-name>pmc-status-embargo</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-status-released</meta-name><meta-value>yes</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-open-access</meta-name><meta-value>yes</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-olf</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-manuscript</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-legally-suppressed</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-has-pdf</meta-name><meta-value>yes</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-has-supplement</meta-name><meta-value>yes</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-pdf-only</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-suppress-copyright</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-is-real-version</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-is-scanned-article</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-preprint</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-in-epmc</meta-name><meta-value>yes</meta-value></custom-meta><custom-meta><meta-name>pmc-license-ref</meta-name><meta-value>CC BY</meta-value></custom-meta><custom-meta><meta-name>issue-copyright-statement</meta-name><meta-value>© Springer Nature Limited 2026</meta-value></custom-meta></custom-meta-group></article-meta></front><body><sec id="Sec1" sec-type="introduction"><title>Introduction</title><p id="Par3">Plant phenotyping aims to quantify functional and structural traits (phenotypes) of crops and plants which result from complex interactions between genetics and environmental factors<sup><xref ref-type="bibr" rid="CR1">1</xref></sup>. Phenotype analysis allows breeders and researchers to disentangle genetic effects from environmental adaptation enabling the development of crops with improved yields and climate resilience—an urgent need given that the global population is projected to reach 9.7 billion by 2050<sup><xref ref-type="bibr" rid="CR2">2</xref></sup> and extreme weather events become increasingly frequent<sup><xref ref-type="bibr" rid="CR3">3</xref></sup>.</p><p id="Par4">Accurate plant phenotyping often relies on computational workflows that chain multiple software tools for image processing to measure plant traits and subsequently perform data analysis. However, the steep learning curve associated with these workflows, such as programming, machine learning and data science, may prevent plant phenotyping researchers and practitioners from fully leveraging them<sup><xref ref-type="bibr" rid="CR4">4</xref></sup>. The complexity of operating several software tools across different platforms present additional communication obstacles for both intra and inter-domain collaboration (e.g. between plant scientists with different specialisations, or between plant scientists and data scientists). Moreover, existing workflows typically employ fixed pipelines, which are difficult to extend or modify, restricting their applicability to broader tasks and scenarios. Despite calls for more accessible and versatile phenotyping systems<sup><xref ref-type="bibr" rid="CR5">5</xref></sup>, this need remains inadequately addressed.</p><p id="Par5">A promising approach to lowering technical barriers bridging domain expertise, and enabling more flexible phenotype analysis is by harnessing the power of large language models (LLMs)<sup><xref ref-type="bibr" rid="CR6">6</xref>,<xref ref-type="bibr" rid="CR7">7</xref></sup>. LLMs have emerged as powerful tools capable of solving diverse tasks from natural language instructions. Rather than focusing on developing computational skills, researchers can focus on critical scientific discovery and exploration by simply prompting an LLM to address the necessary computational and analytical tasks. However, this promise faces the challenge that LLMs alone often lack the specialised domain knowledge and requisite computational workflows to fulfil user-specified tasks<sup><xref ref-type="bibr" rid="CR8">8</xref>,<xref ref-type="bibr" rid="CR9">9</xref></sup>. An effective strategy is to augment LLMs with instructions and external tools forming AI agents capable of automatically executing tailored computational routines and data analysis pipelines.</p><p id="Par6">An AI agent is a goal-directed software designed to perform tasks within a defined digital environment<sup><xref ref-type="bibr" rid="CR10">10</xref>,<xref ref-type="bibr" rid="CR11">11</xref></sup>. LLMs have recently become the core intelligence for many such agents (and hence also termed LLM agents) due to their strong abilities in natural language understanding and reasoning. These agents can receive task instructions, interpret and reason over inputs, and act to achieve specific objectives. They can also utilise external tools such as search engines or databases to extend their abilities<sup><xref ref-type="bibr" rid="CR12">12</xref></sup>. A multi-agent AI system extends this paradigm by integrating several specialised LLM agents and tools within a coordinated framework, enabling division of labour across subtasks and allowing the system to address broader and more complex tasks than a single agent could manage alone<sup><xref ref-type="bibr" rid="CR11">11</xref></sup>. Agentic AI represents multi-agent AI systems that exhibit higher adaptability and autonomy. Rather than operating with fixed roles, such systems often allow agents to interact with each another flexibly, negotiate responsibilities, and adapt their behaviour to dynamic contexts<sup><xref ref-type="bibr" rid="CR12">12</xref>–<xref ref-type="bibr" rid="CR14">14</xref></sup>. These different levels of complexity and flexibility distinguish the three paradigms and make them suitable for different scenarios: AI agents can operate independently on single bounded tasks; multi-agent AI systems are well suited for tasks requiring multi-step workflows; and agentic AI leverages emergent behaviours to support more exploratory or creative tasks.</p><p id="Par7">Recently, LLM-based agent systems have successfully been applied across different scientific domains, including chemistry<sup><xref ref-type="bibr" rid="CR15">15</xref>,<xref ref-type="bibr" rid="CR16">16</xref></sup>, material sciences<sup><xref ref-type="bibr" rid="CR17">17</xref>,<xref ref-type="bibr" rid="CR18">18</xref></sup>, biology<sup><xref ref-type="bibr" rid="CR19">19</xref>,<xref ref-type="bibr" rid="CR20">20</xref></sup> and healthcare<sup><xref ref-type="bibr" rid="CR21">21</xref>–<xref ref-type="bibr" rid="CR23">23</xref></sup>. They not only reduce the technical difficulties of domain-specific computation and data analysis, but also serve as effective platforms for multidisciplinary collaboration. Experts from diverse fields can contribute to the analytical process by interacting with the system through natural language. For example, a plant scientist unsure about which data analysis approach to use could consult a data scientist, then relay the detailed instructions to the AI system to automate the task. Feature comparison of several existing LLMs and LLM-based agent systems for plant science and agriculture is presented in Supplementary Table <xref rid="MOESM1" ref-type="media">1</xref>. While LLMs and LLM-based agent systems have also been applied to agriculture and plant sciences (e.g. agricultural Q&amp;A<sup><xref ref-type="bibr" rid="CR24">24</xref>–<xref ref-type="bibr" rid="CR28">28</xref></sup>, disease and stress phenotyping<sup><xref ref-type="bibr" rid="CR29">29</xref>–<xref ref-type="bibr" rid="CR32">32</xref></sup>, genome prediction<sup><xref ref-type="bibr" rid="CR33">33</xref>–<xref ref-type="bibr" rid="CR35">35</xref></sup>), their potential for automating complex and data-intensive phenotype analysis workflows remains underexplored.</p><p id="Par8">In this work, we introduce PhenoAssistant, an open-source multi-agent AI system for automating plant phenotype analysis. By integrating a generalist LLM with a specialised toolkit for plant research, PhenoAssistant allows plant scientists to use free-text task descriptions to prompt a wide range of phenotyping-related data analysis pipelines, such as phenotype extraction from images, phenotypic statistics analysis, and data visualisation. The toolkit features cutting-edge deep learning models and LLM agents to support users’ dynamic needs in extracting and analysing plant traits. PhenoAssistant can be extended (e.g. through built-in model training functions) to accommodate customised and emerging needs. To improve reproducibility, it also allows users to save the generated pipelines and re-execute them on similar datasets. Overall, PhenoAssistant serves as a pioneer for leveraging multi-agent systems in plant phenotyping. We demonstrate its potential to streamline complex workflows and democratise AI adoption in plant biology through three case studies and evaluations covering three aspects. By lowering technical barriers and facilitating collaborations across different areas of expertise, PhenoAssistant highlights the promising future of LLM-based agent systems in accelerating scientific discovery in plant science.</p></sec><sec id="Sec2" sec-type="results"><title>Results</title><sec id="Sec3"><title>Design of PhenoAssistant</title><p id="Par9">The goal of PhenoAssistant is to generate and execute suitable workflows by chaining AI models and other computational tools to address plant phenotype analysis requests from users. To achieve this, PhenoAssistant is built around a centralised multi-agent architecture, comprising an augmented LLM (the manager) and a specialised toolkit consisting of dedicated computational modules and LLM agents designed to enhance its capabilities for image-based plant phenotyping, as illustrated in Fig. <xref rid="Fig1" ref-type="fig">1</xref>.<fig id="Fig1" position="float" orientation="portrait"><label>Fig. 1</label><caption><title>Design of PhenoAssistant.</title><p>Users provide data and task description to PhenoAssistant. The manager creates a step-by-step plan selects and executes appropriate tools, and then summarises the tool outputs to fulfil the task. Users retain full control to refine intermediate steps as needed. Manager icon made by Surang from Flaticon (<ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="http://www.flaticon.com">www.flaticon.com</ext-link>). Outputs and Toolkit icons made by Freepik from Flaticon (<ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="http://www.flaticon.com">www.flaticon.com</ext-link>).</p></caption><graphic xmlns:xlink="http://www.w3.org/1999/xlink" id="d33e544" position="float" orientation="portrait" xlink:href="41467_2026_71090_Fig1_HTML.jpg"><?image-name 41467_2026_71090_Fig1_HTML.jpg?><?image-size 120175?><?image-md5 a0b4a1082494bb0276c4424aeae8229f?><?image-image-server-status LOAD_COMPLETED?><?image-original-height 828?><?image-original-width 1496?><?image-scaled-height 414?><?image-scaled-width 748?><?image-cloudpmc-urn urn:cdn:blobs/f947/13376813/a0b4a1082494/41467_2026_71090_Fig1_HTML.jpg?><?thumb-name 41467_2026_71090_Fig1_HTML.gif?><?thumb-size 5194?><?thumb-md5 42a48968548ee35c902af6e970b5c3cc?><?thumb-image-server-status NEVER_LOAD?><?thumb-scaled-height 80?><?thumb-scaled-width 144?><?thumb-cloudpmc-urn urn:cdn:blobs/f947/13376813/42a48968548e/41467_2026_71090_Fig1_HTML.gif?></graphic></fig></p><p id="Par10">In this architecture, a single manager agent is responsible for coordinating workflow creation and execution. The role of each LLM agent in the system is pre-defined and fixed. This architecture is chosen because it best supports our aim of reliable workflow generation and execution. It offers several advantages for plant phenotype analysis: it is less prone to coordination errors, produces more consistent workflows for similar task descriptions, and operates within a clearly defined set of available tools and agents. This architecture also increases transparency and user control by allowing users to intervene at any step of the generated workflow to correct errors or unexpected outcomes simply by interacting with the manager.</p><p id="Par11">Specifically, upon receiving user-provided task descriptions and accompanying data, the manager first creates a plan detailing the necessary steps to fulfil the task. It then selects and executes the suitable tools with appropriate parameters, producing various types of outputs that can be saved for further analysis (either by the user or PhenoAssistant). Finally, the manager summarises these outputs, presenting essential information for task completion. Users retain control throughout this process, and can provide textual feedback to refine plans or select alternative tools and parameters, minimising errors from PhenoAssistant.</p><p id="Par12">A core component of many plant phenotype analysis workflows is accurate visual information extraction. While recent generalist LLMs and agents often include vision capabilities<sup><xref ref-type="bibr" rid="CR6">6</xref>,<xref ref-type="bibr" rid="CR36">36</xref>,<xref ref-type="bibr" rid="CR37">37</xref></sup>, they are not always able to generalise to extracting plant phenotypes due to the limited representation of plant-specific data and tasks during their development. To demonstrate this limitation, we prompt the state-of-the-art generalist agent, ChatGPT 4o, in various ways, to extract the projected leaf area (PLA) and leaf count from an image sourced from the same dataset that will later be used in case study 1. The actual PLA and leaf count for this image are 14,260 pixels and 14, respectively.</p><p id="Par13">As illustrated in Fig. <xref rid="Fig2" ref-type="fig">2a</xref>, ChatGPT is directly prompted to extract the requested phenotypes without additional instructions. However, it cannot correctly predict either the leaf count or PLA, despite it invokes the code writing function, attempting to use image processing techniques to solve the tasks. Furthermore, ChatGPT produces inconsistent results when processing the same prompt repeatedly (Fig. <xref rid="Fig2" ref-type="fig">2b</xref>). Next, as shown in Fig. <xref rid="Fig2" ref-type="fig">2c</xref>, we prompt ChatGPT to perform a computer vision pre-task (leaf instance segmentation), which allows for manual extraction of PLA and leaf count from the segmentation results. However, the results are inadequate for further analysis: several edges are incorrectly segmented, and not all individual leaves are successfully identified. Finally, we prompt ChatGPT to rely solely on its vision capabilities to extract the phenotypes, but the results are still unsatisfactory: tasks, such as leaf instance segmentation, cannot be performed using ChatGPT’s vision abilities (Fig. <xref rid="Fig2" ref-type="fig">2d</xref>), and leaf count predictions are incorrect (Fig. <xref rid="Fig2" ref-type="fig">2e</xref>).<fig id="Fig2" position="float" orientation="portrait"><label>Fig. 2</label><caption><title>Examples of using ChatGPT 4o to extract plant phenotypes and perform computer vision tasks on a top-view image of <italic toggle="yes">A. thaliana.</italic></title><p><bold>a</bold>, <bold>b</bold> ChatGPT is prompted to extract projected leaf area and leaf count in two separate attempts, but the results are incorrect and inconsistent between attempts. <bold>c</bold> ChatGPT is prompted to perform leaf instance segmentation. It gradually refines its segmentation approach from producing no meaningful output, to generating a rough plant mask, and finally several leaf masks. However, the final output still fails to separate each leaf individually with clear boundaries. ChatGPT is prompted to rely solely on its inherent vision capabilities to perform leaf instance segmentation (<bold>d</bold>) and leaf counting (<bold>e</bold>). However, these capabilities do not support instance segmentation and also produce an incorrect leaf count. The same <italic toggle="yes">A. thaliana</italic> image from (<bold>a</bold>) is used as visual input for all examples. Screen captures of user-ChatGPT conversations corresponding to these examples are provided in Supplementary Fig. <xref rid="MOESM1" ref-type="media">1</xref>.</p></caption><graphic xmlns:xlink="http://www.w3.org/1999/xlink" id="d33e614" position="float" orientation="portrait" xlink:href="41467_2026_71090_Fig2_HTML.jpg"><?image-name 41467_2026_71090_Fig2_HTML.jpg?><?image-size 102787?><?image-md5 156bb16a8202b7adffa7bc6b64c090d8?><?image-image-server-status LOAD_COMPLETED?><?image-original-height 605?><?image-original-width 1999?><?image-scaled-height 242?><?image-scaled-width 799?><?image-cloudpmc-urn urn:cdn:blobs/f947/13376813/156bb16a8202/41467_2026_71090_Fig2_HTML.jpg?><?thumb-name 41467_2026_71090_Fig2_HTML.gif?><?thumb-size 5273?><?thumb-md5 bd5c4a2d565d5c7c00d38ba48961f2e7?><?thumb-image-server-status NEVER_LOAD?><?thumb-scaled-height 61?><?thumb-scaled-width 200?><?thumb-cloudpmc-urn urn:cdn:blobs/f947/13376813/bd5c4a2d565d/41467_2026_71090_Fig2_HTML.gif?></graphic></fig></p><p id="Par14">To overcome this limitation in generalist LLMs and agents, PhenoAssistant is equipped with phenotype extraction tools, incorporating a model zoo of computer vision models trained on plant-specific datasets, with utilities for computing traits, such as plant area and diameter, from these models’ outputs. The toolkit also supports automatic model training on new data, enabling users to expand PhenoAssistant’s vision capabilities to their unique data and analytical requirements.</p><p id="Par15">Complementing these phenotype extraction tools, PhenoAssistant is also augmented with tools for phenotype analysis. This includes dedicated modules with pre-defined logic and deterministic processes for statistical tests (e.g. ANOVA). In addition, the toolkit contains various LLM agents engineered for different purposes, including the table analyser for extracting information from CSV files, the data visualiser for generating plots, and the plot analyser for interpreting and analysing generated plots. Additionally, the code writer agent supports essential tasks necessary for developing comprehensive end-to-end analysis workflows, such as merging and saving files, as well as executing novel functions not predefined within the toolkit. To enhance PhenoAssistant’s expertise in plant-related domains, a retrieval augmented generation (RAG) agent provides access to scientific literature. The Pipeline Reproducer agent enables users to extract and re-execute previously conducted phenotype extraction and analyses, including all tool calls and generated code, ensuring reproducibility for similar data.</p></sec><sec id="Sec4"><title>Case studies</title><p id="Par16">Traditionally, executing plant phenotyping tasks requires substantial expertise in computer science, including image processing, machine learning, and coding. With PhenoAssistant, users can complete these tasks effortlessly by interacting through natural language. To demonstrate PhenoAssistant’s effectiveness, we present three case studies showcasing its ability to streamline plant phenotyping analysis.</p><p id="Par17">In case study 1, we use PhenoAssistant to replicate part of the Phenotiki analysis workflow<sup><xref ref-type="bibr" rid="CR38">38</xref></sup>, aiming to visualise and analyse the growth patterns of different <italic toggle="yes">Arabidopsis thaliana</italic> ecotypes. We provide PhenoAssistant with the same dataset as in Minervini et al.<sup><xref ref-type="bibr" rid="CR38">38</xref></sup>, which contains 24 plants from 5 different ecotypes: wild type (Col-0, 5 plants), constitutive triple response 1 (<italic toggle="yes">ctr1</italic>, 5 plants), ethylene insensitive 2 (<italic toggle="yes">ein2.1</italic>, 5 plants), <italic toggle="yes">pgm</italic> (mutation in the plastidic isoform of phosphoglucomutase, 5 plants), and <italic toggle="yes">adh1</italic> (mutation causing defects in alcohol dehydrogenase, 4 plants). Each plant was grown for 26 days with images captured every 12 h, resulting in a total of 1248 images.</p><p id="Par18">As shown in Fig. <xref rid="Fig3" ref-type="fig">3</xref>, we prompt PhenoAssistant to address five relevant tasks, each including a task description, tools used by PhenoAssistant, and the generated results. We begin with the fundamental step of any plant phenotyping task, extracting phenotypes from data: in task 1, PhenoAssistant is prompted to extract phenotypes including projected leaf area (PLA) and leaf count from images. It identifies the need for a leaf instance segmentation model and hence selects the one suitable for <italic toggle="yes">Arabidopsis</italic> from the model zoo, executes it, and computes the requested phenotypes from the segmentation results. The code writer is then employed to write and execute Python code to merge the computed phenotypes with metadata and save them into a CSV file. To illustrate how users can interact with PhenoAssistant in practice, we provide the chat logs for case study 1 task 1 in Supplementary Note <xref rid="MOESM1" ref-type="media">1</xref>. At the end of task 1, the user can request PhenoAssistant to save the executed pipeline (i.e. from instance segmentation to saving the extracted phenotypes, as demonstrated at Supplementary Note <xref rid="MOESM1" ref-type="media">2</xref>), which can later be reapplied to a different dataset.<fig id="Fig3" position="float" orientation="portrait"><label>Fig. 3</label><caption><title>Case study 1—<italic toggle="yes">A. thaliana</italic> growth pattern analysis.</title><p>PhenoAssistant automatically completes five tasks: computing phenotypes from images, plotting phenotypic statistics, analysing a generated plot, performing statistical tests for different ecotypes, and comparing findings with literature. Each task is presented as task description (grey), tools used by PhenoAssistant (blue), and results (white). Enlarged versions of the saved plots in task 2 are provided in Supplementary Figs. <xref rid="MOESM1" ref-type="media">2</xref>–<xref rid="MOESM1" ref-type="media">5</xref>.</p></caption><graphic xmlns:xlink="http://www.w3.org/1999/xlink" id="d33e681" position="float" orientation="portrait" xlink:href="41467_2026_71090_Fig3_HTML.jpg"><?image-name 41467_2026_71090_Fig3_HTML.jpg?><?image-size 357062?><?image-md5 c5059a646cb34f563c7bfebca35a369e?><?image-image-server-status LOAD_COMPLETED?><?image-original-height 2499?><?image-original-width 1449?><?image-scaled-height 1249?><?image-scaled-width 724?><?image-cloudpmc-urn urn:cdn:blobs/f947/13376813/c5059a646cb3/41467_2026_71090_Fig3_HTML.jpg?><?thumb-name 41467_2026_71090_Fig3_HTML.gif?><?thumb-size 9396?><?thumb-md5 64c32a49b273ee0d2e574209b9e1b930?><?thumb-image-server-status NEVER_LOAD?><?thumb-scaled-height 172?><?thumb-scaled-width 100?><?thumb-cloudpmc-urn urn:cdn:blobs/f947/13376813/64c32a49b273/41467_2026_71090_Fig3_HTML.gif?></graphic></fig></p><p id="Par19">From tasks 2 to 5, we demonstrate PhenoAssistant’s capabilities for processing and analysing the extracted phenotypes (i.e. from task 1), which are essential for identifying plant growth trends and key differences among ecotypes. In task 2, PhenoAssistant leverages the table analyser and data visualiser agents to compute and plot statistics based on user-specified plot requirements to visualise the growth patterns of different ecotypes. It can further use the plot analyser to interpret data visualisations, such as ranking ecotypes according to their PLA (Task 3). PhenoAssistant can also invoke relevant tools to conduct statistical analyses (e.g. ANOVA and Tukey–Kramer) using user-defined parameters and then summarise the key findings (task 4). For example, it categorises ecotypes by PLA size as large (<italic toggle="yes">ein2.1</italic>, Col-0), medium (<italic toggle="yes">adh1</italic>, <italic toggle="yes">pgm</italic>), and small (<italic toggle="yes">ctr1</italic>), based on Tukey–Kramer post hoc tests (<italic toggle="yes">p</italic> &lt; 0.05). Furthermore, PhenoAssistant can validate these findings against previous literature by leveraging the RAG agent to embed paper contents and retrieve related knowledge (task 5). In tasks 4 and 5, PhenoAssistant performs the same statistical tests and reaches conclusions consistent with the Phenotiki paper<sup><xref ref-type="bibr" rid="CR38">38</xref></sup>.</p><p id="Par20">In case study 2, we show that PhenoAssistant can handle other data (e.g. non-model plant) and tasks when provided with appropriate tools, such as vision models developed on other datasets. To illustrate this flexibility, we prompt PhenoAssistant to replicate part of the workflow in Williams et al.<sup><xref ref-type="bibr" rid="CR39">39</xref></sup> for potato leaves, assessing the correlation between manually measured leaf area (<inline-formula id="IEq1"><alternatives><tex-math id="d33e710"><?equation-image-name d33e710.gif?><?equation-image-status READY?><?equation-image-md5 ed95e90ef9185201e5c3d1cf70003e59?><?equation-image-cloudpmc-urn urn:cdn:blobs/f947/13376813/ed95e90ef918/d33e710.gif?>\documentclass[12pt]{minimal}
				\usepackage{amsmath}
				\usepackage{wasysym} 
				\usepackage{amsfonts} 
				\usepackage{amssymb} 
				\usepackage{amsbsy}
				\usepackage{mathrsfs}
				\usepackage{upgreek}
				\setlength{\oddsidemargin}{-69pt}
				\begin{document}$$A$$\end{document}</tex-math><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="d33e713"><mml:mi>A</mml:mi></mml:math></alternatives></inline-formula>) and dry weight (<inline-formula id="IEq2"><alternatives><tex-math id="d33e717"><?equation-image-name d33e717.gif?><?equation-image-status READY?><?equation-image-md5 478ca78cac853971e376dccde2245395?><?equation-image-cloudpmc-urn urn:cdn:blobs/f947/13376813/478ca78cac85/d33e717.gif?>\documentclass[12pt]{minimal}
				\usepackage{amsmath}
				\usepackage{wasysym} 
				\usepackage{amsfonts} 
				\usepackage{amssymb} 
				\usepackage{amsbsy}
				\usepackage{mathrsfs}
				\usepackage{upgreek}
				\setlength{\oddsidemargin}{-69pt}
				\begin{document}$$W$$\end{document}</tex-math><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="d33e720"><mml:mi>W</mml:mi></mml:math></alternatives></inline-formula>), and evaluating if this correlation holds with algorithmically computed PLA. This analysis can help plant scientists identify potential discrepancies between manual and automated measurements. For this demonstration, we integrate the leaf-only SAM model (for potato leaf segmentation) and use a dataset consisting of 32 potato images with <inline-formula id="IEq3"><alternatives><tex-math id="d33e724"><?equation-image-name d33e724.gif?><?equation-image-status READY?><?equation-image-md5 ed95e90ef9185201e5c3d1cf70003e59?><?equation-image-cloudpmc-urn urn:cdn:blobs/f947/13376813/ed95e90ef918/d33e724.gif?>\documentclass[12pt]{minimal}
				\usepackage{amsmath}
				\usepackage{wasysym} 
				\usepackage{amsfonts} 
				\usepackage{amssymb} 
				\usepackage{amsbsy}
				\usepackage{mathrsfs}
				\usepackage{upgreek}
				\setlength{\oddsidemargin}{-69pt}
				\begin{document}$$A$$\end{document}</tex-math><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="d33e727"><mml:mi>A</mml:mi></mml:math></alternatives></inline-formula> and <inline-formula id="IEq4"><alternatives><tex-math id="d33e731"><?equation-image-name d33e731.gif?><?equation-image-status READY?><?equation-image-md5 478ca78cac853971e376dccde2245395?><?equation-image-cloudpmc-urn urn:cdn:blobs/f947/13376813/478ca78cac85/d33e731.gif?>\documentclass[12pt]{minimal}
				\usepackage{amsmath}
				\usepackage{wasysym} 
				\usepackage{amsfonts} 
				\usepackage{amssymb} 
				\usepackage{amsbsy}
				\usepackage{mathrsfs}
				\usepackage{upgreek}
				\setlength{\oddsidemargin}{-69pt}
				\begin{document}$$W$$\end{document}</tex-math><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="d33e734"><mml:mi>W</mml:mi></mml:math></alternatives></inline-formula> annotated; both the model and dataset originate from Williams et al.<sup><xref ref-type="bibr" rid="CR39">39</xref></sup>.</p><p id="Par21">As shown in Fig. <xref rid="Fig4" ref-type="fig">4</xref>, PhenoAssistant is asked to compute PLA, and it follows the similar procedure to the previous case study while using a different segmentation model (task 1). It is then tasked to measure and analyse the Pearson correlation coefficient between <inline-formula id="IEq5"><alternatives><tex-math id="d33e748"><?equation-image-name d33e748.gif?><?equation-image-status READY?><?equation-image-md5 ed95e90ef9185201e5c3d1cf70003e59?><?equation-image-cloudpmc-urn urn:cdn:blobs/f947/13376813/ed95e90ef918/d33e748.gif?>\documentclass[12pt]{minimal}
				\usepackage{amsmath}
				\usepackage{wasysym} 
				\usepackage{amsfonts} 
				\usepackage{amssymb} 
				\usepackage{amsbsy}
				\usepackage{mathrsfs}
				\usepackage{upgreek}
				\setlength{\oddsidemargin}{-69pt}
				\begin{document}$$A$$\end{document}</tex-math><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="d33e751"><mml:mi>A</mml:mi></mml:math></alternatives></inline-formula> and <inline-formula id="IEq6"><alternatives><tex-math id="d33e755"><?equation-image-name d33e755.gif?><?equation-image-status READY?><?equation-image-md5 478ca78cac853971e376dccde2245395?><?equation-image-cloudpmc-urn urn:cdn:blobs/f947/13376813/478ca78cac85/d33e755.gif?>\documentclass[12pt]{minimal}
				\usepackage{amsmath}
				\usepackage{wasysym} 
				\usepackage{amsfonts} 
				\usepackage{amssymb} 
				\usepackage{amsbsy}
				\usepackage{mathrsfs}
				\usepackage{upgreek}
				\setlength{\oddsidemargin}{-69pt}
				\begin{document}$$W$$\end{document}</tex-math><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="d33e758"><mml:mi>W</mml:mi></mml:math></alternatives></inline-formula>, as well as PLA and <inline-formula id="IEq7"><alternatives><tex-math id="d33e762"><?equation-image-name d33e762.gif?><?equation-image-status READY?><?equation-image-md5 478ca78cac853971e376dccde2245395?><?equation-image-cloudpmc-urn urn:cdn:blobs/f947/13376813/478ca78cac85/d33e762.gif?>\documentclass[12pt]{minimal}
				\usepackage{amsmath}
				\usepackage{wasysym} 
				\usepackage{amsfonts} 
				\usepackage{amssymb} 
				\usepackage{amsbsy}
				\usepackage{mathrsfs}
				\usepackage{upgreek}
				\setlength{\oddsidemargin}{-69pt}
				\begin{document}$$W$$\end{document}</tex-math><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="d33e765"><mml:mi>W</mml:mi></mml:math></alternatives></inline-formula> (task 2). As this correlation analysis is not predefined in the toolkit, PhenoAssistant leverages the code writer to accomplish it. Finally, PhenoAssistant summarises the results, highlighting the correlation coefficients and providing further explanations and context. The findings indicate that while the PLA derived from leaf-only SAM is useful, it may introduce errors compared to manually measured leaf area when predicting dry weight. In task 2, PhenoAssistant performs the same correlation analysis and reaches conclusions consistent with Williams et al.<sup><xref ref-type="bibr" rid="CR39">39</xref></sup>.<fig id="Fig4" position="float" orientation="portrait"><label>Fig. 4</label><caption><title>Case study 2—potato leaf area and dry weight correlation analysis.</title><p>In response to the user’s requests, PhenoAssistant first extracts phenotypes from the provided data and then compares correlations between different plant-related variables.</p></caption><graphic xmlns:xlink="http://www.w3.org/1999/xlink" id="d33e781" position="float" orientation="portrait" xlink:href="41467_2026_71090_Fig4_HTML.jpg"><?image-name 41467_2026_71090_Fig4_HTML.jpg?><?image-size 271663?><?image-md5 7f08416533506b27a5bf4f2c5dbb2fac?><?image-image-server-status LOAD_COMPLETED?><?image-original-height 2496?><?image-original-width 1615?><?image-scaled-height 998?><?image-scaled-width 646?><?image-cloudpmc-urn urn:cdn:blobs/f947/13376813/7f0841653350/41467_2026_71090_Fig4_HTML.jpg?><?thumb-name 41467_2026_71090_Fig4_HTML.gif?><?thumb-size 7991?><?thumb-md5 5ff255ee8549b87a3cc166474925fa17?><?thumb-image-server-status NEVER_LOAD?><?thumb-scaled-height 155?><?thumb-scaled-width 100?><?thumb-cloudpmc-urn urn:cdn:blobs/f947/13376813/5ff255ee8549/41467_2026_71090_Fig4_HTML.gif?></graphic></fig></p><p id="Par22">In case study 3, we demonstrate that PhenoAssistant can expand its vision capabilities via the built-in automatic vision model training function, enabling it to tackle previously unseen phenotyping challenges (e.g. experiments with new data or in new environments).</p><p id="Par23">As shown in Fig. <xref rid="Fig5" ref-type="fig">5</xref>, PhenoAssistant is prompted to identify nutrient deficiency for winter wheats in field, where no capable model is integrated in the toolkit. Recognising the need for image classification, it instructs the user to upload labelled data in the required format. After receiving the dataset, it invokes tools to automatically preprocess the data (e.g. splitting into training and evaluation subsets, applying data augmentation), initiate model training using the user-specified finetuning method (full-parameter finetuning or low-rank adaptation (LoRA)<sup><xref ref-type="bibr" rid="CR40">40</xref></sup>, whose differences are described in the automatic vision model training subsection of the 'Methods' section), and evaluate the model’s performance. Finally, the evaluation results are presented to the user, and the trained model is added to the model zoo for future use. The data used for this case study are from Yi et al.<sup><xref ref-type="bibr" rid="CR41">41</xref>,<xref ref-type="bibr" rid="CR42">42</xref></sup>.<fig id="Fig5" position="float" orientation="portrait"><label>Fig. 5</label><caption><title>Case study 3—automatic model training for nutrient deficiency identification.</title><p>When no suitable model is available to solve a given task, PhenoAssistant first prompts the user to provide a dataset in the desired format. The user can select between full-parameter finetuning or LoRA<sup><xref ref-type="bibr" rid="CR40">40</xref></sup>, depending on computational resource constraints and performance requirements. It then automatically applies data preprocessing, followed by training and evaluating the model. The trained model is saved in the model zoo for future use.</p></caption><graphic xmlns:xlink="http://www.w3.org/1999/xlink" id="d33e812" position="float" orientation="portrait" xlink:href="41467_2026_71090_Fig5_HTML.jpg"><?image-name 41467_2026_71090_Fig5_HTML.jpg?><?image-size 286630?><?image-md5 e030daf732af7fa944422fdca04b5b79?><?image-image-server-status LOAD_COMPLETED?><?image-original-height 2496?><?image-original-width 1657?><?image-scaled-height 997?><?image-scaled-width 662?><?image-cloudpmc-urn urn:cdn:blobs/f947/13376813/e030daf732af/41467_2026_71090_Fig5_HTML.jpg?><?thumb-name 41467_2026_71090_Fig5_HTML.gif?><?thumb-size 7853?><?thumb-md5 a6da6ea1f920b524a403c577e56adfc5?><?thumb-image-server-status NEVER_LOAD?><?thumb-scaled-height 151?><?thumb-scaled-width 100?><?thumb-cloudpmc-urn urn:cdn:blobs/f947/13376813/a6da6ea1f920/41467_2026_71090_Fig5_HTML.gif?></graphic></fig></p><p id="Par24">In summary, we present three case studies illustrating how PhenoAssistant streamlines plant phenotyping with natural language instructions, allowing plant researchers to concentrate on scientific discovery rather than technical training. These cases span various plant species, environmental conditions, and tasks, including phenotype extraction, pipeline reproduction, statistics computation and visualisation, plot analysis, statistical testing, knowledge retrieval, correlation assessment, and model training. Chat logs for all case studies are available in our code repository.</p></sec><sec id="Sec5"><title>Evaluation of PhenoAssistant</title><p id="Par25">Beyond the presented case studies, we evaluate PhenoAssistant’s adaptability and potential for broader applications via a series of tasks. Since PhenoAssistant leverages rapidly evolving AI technologies, this evaluation also offers a method for assessing new AI components (e.g. DeepSeek<sup><xref ref-type="bibr" rid="CR43">43</xref>,<xref ref-type="bibr" rid="CR44">44</xref></sup>) as replacements for PhenoAssistant’s original modules. The evaluation covers three aspects: tool selection, vision model selection, and data analysis. Prompts and detailed results for the evaluation are presented at Supplementary Notes <xref rid="MOESM1" ref-type="media">3</xref>–<xref rid="MOESM1" ref-type="media">5</xref>. Chat logs for the evaluation are available in our code repository. During the evaluation, no human feedback (e.g. corrections to tool selection or tool arguments) is provided.</p><p id="Par26">We evaluate the tool selection ability of PhenoAssistant by assessing whether it can select appropriate tools with the correct parameters and in the correct order to complete a task. This evaluates PhenoAssistant’s abilities of chaining multiple tools for task completion rather than the correctness of task outputs.</p><p id="Par27">To conduct this evaluation, we create 20 tasks to represent various phenotyping requests with different plant species. For each task, we prompt PhenoAssistant to generate a sequence of tools with their arguments in a standalised format. The outputs from PhenoAssistant are evaluated by an LLM-based evaluator, termed tool evaluator (LLM-based evaluators have been applied in different domains<sup><xref ref-type="bibr" rid="CR15">15</xref>,<xref ref-type="bibr" rid="CR45">45</xref></sup>). The tool evaluator is an LLM (GPT-4o) provided with detailed information about PhenoAssistant’s toolkit (e.g. descriptions of tool functionalities and arguments) and the user-specified task description. It is prompted to score the response of PhenoAssistant for each task on the following four aspects using a scale of 1–5 (where 1 is the lowest and 5 the highest):<list list-type="order"><list-item><p id="Par28">Overall chain: whether the tool chain is reasonable for addressing the task and whether any crucial steps are missing or incorrectly ordered.</p></list-item><list-item><p id="Par29">Tool existence: whether each selected tool exists within PhenoAssistant’s toolkit.</p></list-item><list-item><p id="Par30">Tool appropriateness: whether each tool is suitable for the specific step where it is applied.</p></list-item><list-item><p id="Par31">Arguments: whether the arguments for each tool are correctly specified.</p></list-item></list></p><p id="Par32">The average scores across the 20 tasks are reported in Table <xref rid="Tab1" ref-type="table">1</xref>, under the row manager. Overall, PhenoAssistant consistently achieves a full score in tool existence correctness (5), indicating that it does not hallucinate tools outside its toolkit. Slightly lower scores in overall chain correctness (4.25) and argument correctness (4.3) suggest that PhenoAssistant might generate toolchains in an inappropriate order, omit certain steps, or provide less accurate arguments. These issues may also be caused by vague task descriptions, highlighting the importance of allowing users to clarify tasks through interaction with PhenoAssistant. PhenoAssistant also performs well on tool appropriateness (4.65), although this score still indicates potential limitations in the tool understanding abilities of LLMs, as well as in the clarity and comprehensiveness of the tool descriptions provided within the toolkit in PhenoAssistant.<table-wrap id="Tab1" position="float" orientation="portrait"><label>Table 1</label><caption><p>Tool selection evaluation results under two settings (manager and manager + critic)</p></caption><table frame="hsides" rules="groups"><thead><tr><th colspan="1" rowspan="1">Setting</th><th colspan="1" rowspan="1">Overall chain</th><th colspan="1" rowspan="1">Tool existence</th><th colspan="1" rowspan="1">Tool appropriateness</th><th colspan="1" rowspan="1">Argument</th><th colspan="1" rowspan="1">Average</th></tr></thead><tbody><tr><td align="left" colspan="1" rowspan="1">Manager</td><td char="." align="char" colspan="1" rowspan="1">4.25</td><td char="." align="char" colspan="1" rowspan="1">5.00</td><td char="." align="char" colspan="1" rowspan="1">4.65</td><td char="." align="char" colspan="1" rowspan="1">4.30</td><td char="." align="char" colspan="1" rowspan="1">4.55</td></tr><tr><td align="left" colspan="1" rowspan="1">Manger + critic</td><td char="." align="char" colspan="1" rowspan="1">4.35</td><td char="." align="char" colspan="1" rowspan="1">5.00</td><td char="." align="char" colspan="1" rowspan="1">4.90</td><td char="." align="char" colspan="1" rowspan="1">4.40</td><td char="." align="char" colspan="1" rowspan="1">4.66</td></tr></tbody></table><table-wrap-foot><p>For each setting, the scores at each column are averaged across 20 tasks.</p></table-wrap-foot></table-wrap></p><p id="Par33">Beyond evaluating the original design of PhenoAssistant, we also assess the potential of introducing a critic agent. This agent is implemented as a prompted LLM (GPT-4o), provided with both the user-specified task and PhenoAssistant’s toolkit information, to critique and refine the toolchains generated by the manager. Each toolchain produced by the manager is reviewed by the critic and will be refined if issues are identified. The toolchains generated by the manager–critic interactions are assessed with the same tool evaluator in the manager-only setting.</p><p id="Par34">The average scores with the critic are presented in Table <xref rid="Tab1" ref-type="table">1</xref> under the row manager + critic. The results show consistent improvements across all aspects compared to using the manager solely. This suggests that strategy verification mechanisms (e.g. adding a critic agent) can enhance the reliability of PhenoAssistant on broader tasks and reduce the burden and frequency of human verification, which should be considered in future extensions.</p><p id="Par35">The system prompts for the tool evaluator and the critic are provided in Supplementary Note <xref rid="MOESM1" ref-type="media">6</xref>. Detailed per-task scores for the tool selection evaluation are reported in Supplementary Table <xref rid="MOESM1" ref-type="media">2</xref>.</p><p id="Par36">Next, we evaluate whether PhenoAssistant can recommend the appropriate type of computer vision model (instance segmentation, image classification, or image regression) for a given plant phenotyping task. We term this evaluation vision model selection I. It is crucial for automatic model training, as users may not have substantial machine learning knowledge to associate plant tasks with computer vision models. Accurate suggestions from PhenoAssistant prevent users from preparing incorrect training data.</p><p id="Par37">To conduct this evaluation, we create 50 plant tasks expressed in varied formats: short phrases (e.g. leaf counting), statements (e.g. highlight the weeds in these field images), and questions (e.g. what is the maturity status of these pods?). This variation aims to cover different ways users might naturally express tasks. PhenoAssistant is then asked to select the most appropriate model type corresponding to each task description. The outputs of PhenoAssistant are manually inspected.</p><p id="Par38">PhenoAssistant achieves 100% accuracy in vision model selection I, indicating its potential to activate the correct model training function for a wide range of plant tasks.</p><p id="Par39">In addition to evaluating PhenoAssistant’s ability to select the correct model type, we further assess whether it can identify the most suitable specific model from a vision model zoo for a given plant phenotyping task. We term this evaluation vision model selection II. It is designed to test whether PhenoAssistant can handle broader phenotyping scenarios when its model zoo expands.</p><p id="Par40">To perform this evaluation, we adapt tasks from vision model selection I and create 20 tasks covering a wide range of plant species and phenotyping needs (e.g. identify all the wheat spikes in these images, and how badly are these sunflowers affected by disease? Provide a score for each image). We also create a vision model zoo containing 30 model identifiers. Each identifier follows the same naming convention used in the vision model zoo of PhenoAssistant, namely {plant-species}_{task}_{model} (e.g. wheat_spike-instance-segmentation_m2fb, sunflower_disease-severity-regress_dino2b), as described in the vision model zoo subsection of the 'Methods' section. Among these 30 model identifiers, 15 identifiers correspond exactly to 15 of the 20 plant tasks, while the remaining 15 identifiers serve as distractors.</p><p id="Par41">PhenoAssistant is then asked to match each plant task with the most suitable model identifier from the model zoo. It achieves 100% accuracy in this evaluation: for the 15 tasks with corresponding model identifiers, it provides the correct match; for the remaining 5 tasks without a direct match, it not only recognises that no suitable model is available, but also recommends an appropriate model type for fine-tuning. This result demonstrates PhenoAssistant’s potential to address a wider range of plant phenotyping tasks with an extended vision model zoo.</p><p id="Par42">Finally, we assess whether PhenoAssistant can produce the correct outputs for different data analysis requests. This aims to assess the reliability of the built-in LLM agents (e.g. data visualiser) for data analysis.</p><p id="Par43">To conduct this evaluation, we create 20 phenotype analysis tasks using actual data. These tasks cover a range of operations including extracting values from files, computing statistics, generating plots, analysing plots, converting file formats, computing and analysing new metrics. The outputs of PhenoAssistant are manually inspected.</p><p id="Par44">PhenoAssistant achieves 85% (17/20) accuracy in this evaluation. All failed cases fall within the plot analysis category. These tasks require the use of the plot analyser agent (a prompted GPT-4o) to interpret and extract fine-grained information from figures. This suggests that fine-grained visual reasoning remains a bottleneck for LLMs.</p><p id="Par45">In summary, our evaluation shows that PhenoAssistant, as a multi-agent AI system, exhibits promising adaptability to broader plant phenotyping scenarios and tasks, while strategy verification mechanisms and improved visual interpretation capabilities should be further explored.</p></sec></sec><sec id="Sec6" sec-type="discussion"><title>Discussion</title><p id="Par46">In summary, we introduce PhenoAssistant, a pioneering multi-agent AI system designed to streamline data-intensive and complex workflows for plant phenotyping researchers and practitioners while reducing the requirements for computational expertise and the burden of programming. By integrating cutting-edge AI technologies, PhenoAssistant transforms simple, interactive conversations into complex image-based plant analyses. Our case studies demonstrate its versatility across a range of phenotyping tasks and plant species, while evaluations highlight its potential of extending to broader scenarios. Overall, PhenoAssistant marks an important step toward democratising AI adoption in plant phenotyping and lays a solid foundation for future AI-driven advancements in plant science. To facilitate ongoing development and validation, PhenoAssistant is made available as open source.</p><p id="Par47">Despite its success, PhenoAssistant has limitations. Due to current LLMs’ constraints in understanding and decomposing complex tasks, some tasks may need to be carefully instructed or broken down into multiple steps. This can be improved with more effective prompt engineering or fine-tuning on LLMs. We also integrate a function to reproduce executed pipelines, reducing the user-LLM interaction for these complex tasks.</p><p id="Par48">In addition, our evaluations reveal problems on tool selection and data analysis, indicating that a certain level of human inspection of the generated workflows and results is still required. We show that introducing an additional critic agent to verify the responses generated by the manager agent can improve tool selection. In the future, other strategy verification mechanisms, such as sandboxing and agent self-evaluation strategies should also be explored<sup><xref ref-type="bibr" rid="CR46">46</xref>,<xref ref-type="bibr" rid="CR47">47</xref></sup>. Although the current evaluations cover the essential functions of PhenoAssistant, they still require a degree of human involvement to design the evaluation tasks and perform the assessments. This may limit the scalability of evaluations when extending PhenoAssistant to a larger set of tools or tasks. Future work should aim to automate evaluations to enhance scalability.</p><p id="Par49">PhenoAssistant currently relies on users to provide information about their data source or inspect the extracted phenotypes to mitigate the risks of selecting and applying vision models to conditions different from those they were originally trained on (e.g. variations in imaging setups, lighting, or growth environments). This limitation can be addressed by using the integrated training function to develop new models on users’ own datasets; looking ahead, it could be further reduced by integrating out-of-distribution (OOD) detection mechanisms<sup><xref ref-type="bibr" rid="CR48">48</xref>,<xref ref-type="bibr" rid="CR49">49</xref></sup> or vision foundation models capable of handling more general agricultural and plant conditions<sup><xref ref-type="bibr" rid="CR50">50</xref>,<xref ref-type="bibr" rid="CR51">51</xref></sup>.</p><p id="Par50">PhenoAssistant’s adaptability to new tasks and scenarios is mainly constrained by the coverage and capabilities of its toolkit. To extend vision capabilities, PhenoAssistant supports automatic model training, but this requires users to collect and label the training data. Meanwhile, AI community platforms (e.g. Hugging Face<sup><xref ref-type="bibr" rid="CR52">52</xref></sup> and Kaggle<sup><xref ref-type="bibr" rid="CR53">53</xref></sup>) host many trained computer vision models for plant-related tasks. In the future, dynamically fetching suitable models from these platforms based on user-specified plant species and phenotyping tasks should be explored to further reduce user effort.</p><p id="Par51">However, several requirements must be met before such a mechanism can be reliably integrated. First, model identifiers and descriptions on these platforms should be standardised and made more comprehensive, so that the training data sources as well as plant species and phenotypic traits to which each model can be applied, are clearly specified and easily retrieved. Second, mechanisms are required to assess compatibility between fetched models and user-provided data, since models may fail when applied to scenarios different from those they were originally trained on. Third, external model repositories should ensure long-term stability through consistent access protocols and transparent versioning, mitigate issues arising from model removal or changes in authentication and data formats. Finally, safeguards are needed to ensure that dynamically fetched models are free from malicious components or adversarial behaviours that affect the safety and robustness in plant phenotyping.</p><p id="Par52">In addition, standards such as model context protocol (MCP)<sup><xref ref-type="bibr" rid="CR54">54</xref></sup> should be explored to allow users integrate other tools more easily. These standards enable software developers to make their products readily integrable with LLMs, fostering a growing ecosystem of LLM compatible tools with diverse functionalities. As the needs and interests of applying LLM agent systems in plant phenotyping grow, more related tools will adopt such standards and make this form of extension more practical.</p><p id="Par53">While PhenoAssistant focuses on streamlining phenotype extraction and analysis, plant phenotyping (more broadly, plant science and agriculture) encompasses other tasks that could benefit from LLM agents. For example, simulating plant-related hypotheses (e.g. responses to different nutrient deficiencies under different conditions) could help researchers narrow down experimental options, reduce risks, and ultimately save resources and time before deploying costly real-world experiments. One emerging direction is agent-based simulation, where agents explore what-if scenarios to probe underlying mechanisms and assess potential outcomes by mimicking real-world processes. Such approaches have been investigated in domains such as economic systems<sup><xref ref-type="bibr" rid="CR55">55</xref>,<xref ref-type="bibr" rid="CR56">56</xref></sup> and physical environments<sup><xref ref-type="bibr" rid="CR57">57</xref>,<xref ref-type="bibr" rid="CR58">58</xref></sup>, though challenges in efficiency and robustness remains insufficiently addressed<sup><xref ref-type="bibr" rid="CR59">59</xref></sup>.</p><p id="Par54">Closely aligned with the goals of simulation, reasoning about cause–effect relationships in plant physiology can help researchers formulate mechanistic hypotheses and predict phenotypic responses to novel environments or management strategies. Although PhenoAssistant already demonstrates abilities in statistical testing, literature-based verification and correlation analysis (case study 1 and 2), it does not explicitly support causal modelling. Recently, agent-based causal reasoning and validation has emerged as an active research area<sup><xref ref-type="bibr" rid="CR60">60</xref>,<xref ref-type="bibr" rid="CR61">61</xref></sup>. However, existing studies also identify that current agent-based causal modelling often lacks deep causal understanding and are limited to shallow, low-level reasoning<sup><xref ref-type="bibr" rid="CR12">12</xref>,<xref ref-type="bibr" rid="CR62">62</xref></sup>.</p><p id="Par55">PhenoAssistant adopts a centralised multi-agent architecture where a single manager coordinates tools and agents with pre-defined roles to accomplish user-specified tasks. This architecture is well-suited to the current objective on plant phenotyping workflow generation and execution. However, it may limit an agent system’s capacity for emergent intelligence and constrain its ability to address more open-ended or creative tasks. Recent research on agentic AI (i.e. more autonomous and self-adapting multi-agent AI system) investigates decentralised designs in which different agents interact more flexibly through peer-to-peer communication with shared task contexts<sup><xref ref-type="bibr" rid="CR63">63</xref>,<xref ref-type="bibr" rid="CR64">64</xref></sup>; this trend is further supported by emerging multi-agent frameworks<sup><xref ref-type="bibr" rid="CR65">65</xref>,<xref ref-type="bibr" rid="CR66">66</xref></sup>. Alongside this, dynamic role negotiation has emerged as an alternative to static role assignments, allowing agents to self-assign responsibilities and adapt them in response to evolving task requirements<sup><xref ref-type="bibr" rid="CR67">67</xref>,<xref ref-type="bibr" rid="CR68">68</xref></sup>. Such advances in agent orchestration and communication are promising for fostering emergent behaviours and scalability<sup><xref ref-type="bibr" rid="CR12">12</xref></sup>, and hence are suitable for tackling exploratory tasks in agriculture and plant science beyond workflow automation, such as proposing novel research directions or unconventional experimental designs. At the same time, decentralised coordination and dynamic role assignment introduce challenges, such as longer inference time due to negotiation overhead, instability or inefficiency when roles are poorly defined, and additional safety and governance concerns<sup><xref ref-type="bibr" rid="CR12">12</xref>,<xref ref-type="bibr" rid="CR69">69</xref></sup>.</p><p id="Par56">To address these concerns and extend multi-agent systems such as PhenoAssistant more broadly, recent surveys<sup><xref ref-type="bibr" rid="CR12">12</xref>,<xref ref-type="bibr" rid="CR69">69</xref></sup> suggest that proper scalable infrastructure and governance are required. Scalable infrastructure should include flexible agent orchestration mechanisms (from centralised managers with fixed-role agents to decentralised agents with dynamic role negotiation). It should also provide standard protocols for integrating external tools (e.g. MCP as mentioned above) and allowing agents built from different frameworks to interact securely and effectively (e.g. Agent2Agent protocol<sup><xref ref-type="bibr" rid="CR70">70</xref></sup>). In addition, it should include robust resource scheduling and conflict handling mechanisms between agents to prevent potential bottlenecks and instability of agent systems. Moreover, a shared memory layer (e.g. vector databases) is important to providing different agents with a consistent and queryable context on task progress, while observability pipelines that log conversation histories and tool use are crucial for human users to audit and debug.</p><p id="Par57">Lastly, complementing scalable infrastructure, governance policies are required to keep multi-agent AI systems safe and accountable as they scale. Mechanisms for automatic or manual supervision are needed to ensure each agent has transparent responsibility, aligns to the overall goal, and accountable for its subtasks. Access controls and sandboxing protect privacy and safety when agents interact with external data and tools. Configurable memory retention policies (e.g. for vector databases) can prevent unintended leakage of task execution context. Finally, certain level of human-in-the-loop involvement should be preserved to trace agent actions and intervene if problems arise, such as biased interpretation of data or agents drifting away from the intended objectives.</p></sec><sec id="Sec7"><title>Methods</title><sec id="Sec8"><title>LLMs and agents</title><p id="Par58">LLMs are deep neural networks with billions of parameters trained on large-scale text corpora to predict the next token in a sequence. Although this objective is simple, it enables LLMs to acquire emergent abilities, such as natural language understanding, reasoning and code generation. Recent models such as GPT-4o<sup><xref ref-type="bibr" rid="CR6">6</xref></sup>, have also extended to multi-modal training (e.g. using both text and image data), and hence expanded their capabilities to vision-language interpretation, making them suitable to support more complex tasks.</p><p id="Par59">The versatile abilities of LLMs make them effective foundations for AI agents (and hence also termed LLM agents). These agents leverage LLMs as core intelligence for reasoning, decision-making, and performing actions to achieve specified goals, such as operating structural data or retrieving knowledge from literature. This is typically achieved by providing an LLM with (i) a carefully crafted system prompt, which defines the agent’s role and acting logics (e.g. to follow specific instructions or avoid common mistakes), and (ii) access to external tools, such as computational functions, APIs, or databases, which extend the abilities of an LLM beyond those acquired from model training. When multiple agents and tools with specialised functionalities are integrated into a multi-agent AI system, the system as a whole can collectively address complex tasks that exceed the capacity of a single agent. To ensure that a such system aligns with its objectives, both system-level design (e.g. incorporating human-in-the-loop mechanisms) and element-level design (i.e. selecting appropriate agents and tools) need to be carefully considered.</p></sec><sec id="Sec9"><title>Implementation of PhenoAssistant</title><p id="Par60">PhenoAssistant comprises a manager agent and a specialised toolkit designed for plant phenotyping tasks. The manager provides an interactive interface for handling user requests, planning tasks, selecting and executing tools, and summarising results. The manager is implemented using GPT-4o (version: 2024-08-06, via OpenAI Azure API), which was the state-of-the-art LLM when PhenoAssistant was developed, with the model temperature set to 0.1, guided by a system prompt (Supplementary Note <xref rid="MOESM1" ref-type="media">6</xref>) that instructs it to ensure desired functionality and behaviour.</p><p id="Par61">Specifically, the system prompt instructs the manager agent to (i) generate a step-by-step plan and refine it iteratively, (ii) coordinate the use of the toolkit by chaining outputs and inputs across different tools and agents, (iii) follow specialised guidelines when using specific tools or agents (e.g. the manager must verify the availability of a vision model before invoking it), and (iv) conclude interactions with a concise summary.</p><p id="Par62">The toolkit consists of Python-based functions, each accompanied by structured schema (name, description, parameters, input/output format). This enables the manager agent to understand the tools’ capabilities and compose them properly into valid workflows. Notably, PhenoAssistant’s toolkit includes both deterministic modules (e.g. instance segmentation and statistical tests) and LLM agents (e.g. code writer and plot analyser).</p><p id="Par63">PhenoAssistant is implemented in Python, with tool invocation and multi-agent orchestration supported by AutoGen. Computer vision models and model training functions are implemented using PyTorch and the Transformers library. The table analyser agent is implemented using Pandas AI. We introduce the key tools below; other tools, as well as a complete list of software dependencies is available in our code repository.</p></sec><sec id="Sec10"><title>Vision model zoo</title><p id="Par64">Many plant phenotypes cannot be extracted without the aid of image processing or computer vision techniques. Although recent LLMs have incorporated vision capabilities, they are not yet powerful enough for effective extraction of plant traits. For example, tasks such as delineating individual leaf boundaries (known as instance segmentation in computer vision), which is a preliminary task for computing leaf area and count, remain challenging.</p><p id="Par65">To overcome this limitation, PhenoAssistant integrates external computer vision models. Each vision model is uniquely identified using the naming convention: {plant-species}_{task}_{(optional) training-dataset}_{model}_{(optional) finetuning-method}. These identifiers are maintained within a structured file (e.g. model_zoo.json), which the manager agent accesses whenever a vision model is required, ensuring appropriate vision model selection for the given plant task.</p><p id="Par66">For demonstration purposes, in case study 1, we integrate Mask2Former<sup><xref ref-type="bibr" rid="CR71">71</xref></sup> for segmenting individual <italic toggle="yes">Arabidopsis</italic> leaves. Following the similar practice in Chen et al.<sup><xref ref-type="bibr" rid="CR72">72</xref></sup>, Mask2Former is fine-tuned on subsets A1 and A4 of the publicly available CVPPP LSC dataset<sup><xref ref-type="bibr" rid="CR73">73</xref></sup>. In case study 2, we integrate Leaf-only SAM<sup><xref ref-type="bibr" rid="CR39">39</xref></sup> for potato leaf instance segmentation.</p></sec><sec id="Sec11"><title>Automatic vision model training</title><p id="Par67">Given the diverse conditions encountered in plant phenotyping, such as varying illumination, indoor/outdoor environments, species diversity and diverse phenotyping needs, it is practically impossible to pre-integrate every required vision model into PhenoAssistant. Currently, there is also no universal vision model capable of solving all phenotyping tasks. Furthermore, new tasks and customised user requirements continuously emerge, highlighting the need for a convenient mechanism to expand PhenoAssistant’s vision model zoo.</p><p id="Par68">To satisfy this need, PhenoAssistant incorporates an automated model training pipeline for common phenotyping tasks (e.g. image classification). When users identify the need for a new vision model or when PhenoAssistant determines that its current model zoo cannot adequately solve a provided task, PhenoAssistant prompts the user to upload a dataset formatted according to predefined specifications. PhenoAssistant calls tools to automatically split the uploaded dataset into training and validation sets, as well as initiate model training. Upon completion, the newly trained model is automatically added to the vision model zoo using the naming convention described earlier and becomes available for inference.</p><p id="Par69">The core idea behind this automated training pipeline is to fine-tune pre-trained vision models using plant-specific datasets<sup><xref ref-type="bibr" rid="CR74">74</xref></sup>. These models are initially trained on large-scale general data and can be fine-tuned to specific plant analysis tasks with limited additional data, thereby minimising development and deployment efforts. Specifically, for image classification (as shown in case study 3), we employ the DINOv2-base model<sup><xref ref-type="bibr" rid="CR75">75</xref></sup> as the pre-trained model.</p><p id="Par70">PhenoAssistant supports two fine-tuning strategies: low-rank adaptation (LoRA)<sup><xref ref-type="bibr" rid="CR40">40</xref></sup> and full fine-tuning, to accommodate users with varying levels of computational resources. LoRA updates only a small subset of parameters by inserting low-rank trainable matrices into existing model layers, substantially reducing computational and memory requirements during training. This approach enables rapid adaptation at the cost of a modest performance trade-off. In contrast, full fine-tuning updates all model parameters, potentially achieving higher accuracy while at increased computational and memory demands.</p></sec><sec id="Sec12"><title>LLM agents for phenotype analysis</title><p id="Par71">Beyond phenotype extraction, PhenoAssistant supports diverse data analysis tasks, such as statistics computations, data visualisation, and interpretative plot analysis. These analyses often involve complex and customised requirements that cannot always follow pre-defined logic. For example, users may request specific visualisation styles (e.g. plotting plant growth curves for different ecotypes using specific colour) or novel analyses prompted spontaneously during exploration.</p><p id="Par72">To support these dynamic and evolving analytical needs, we repurpose GPT-4o (same as the manager agent, version 2024-08-06 via the OpenAI Azure API with temperature set to 0.1) into different roles using different system prompts (Supplementary Note <xref rid="MOESM1" ref-type="media">6</xref>). This configuration creates a set of distinct LLM agents, each of which can be invoked by the manager when required:<list list-type="order"><list-item><p id="Par73">Code writer: generates and executes Python code for performing tasks that are not previously defined in the toolkit. Code execution is performed with automatic error handling and retry logic. Its system prompt explicitly instructs the agent to output complete, runnable scripts.</p></list-item><list-item><p id="Par74">Data visualiser: generates and executes Python code for producing plots with user specification. Guidance for creating clear and well-designed plots (e.g. including legends and beautifying layout) is provided within its system prompt.</p></list-item><list-item><p id="Par75">Plot analyser: interprets plots using GPT-4o’s visual understanding capabilities.</p></list-item><list-item><p id="Par76">Table analyser: queries values and computes statistics from CSV files using Pandas AI. Pandas AI is an agent framework that augments an LLM (GPT-4o in our case) with predefined logic and access to specialised tools (e.g. Pandas data analysis library) to allow natural language queries for data analysis.</p></list-item><list-item><p id="Par77">Pipeline reproducer: extracts executed function calls and code from a chat history, organising them into a new Python function that can be reused in the future. Its system prompt instructs it to extract only the functions and code snippets that were actually executed, while excluding the failed ones. The system prompt also provides an example of an extracted pipeline to ensure the output is valid and can be re-run on new datasets without modification.</p></list-item><list-item><p id="Par78">RAG agent: expands PhenoAssistant’s domain-specific knowledge by embedding and retrieving information from literature related to  user-specified tasks.</p></list-item></list></p></sec><sec id="Sec13"><title>Reporting summary</title><p id="Par79">Further information on research design is available in the <xref rid="MOESM3" ref-type="media">Nature Portfolio Reporting Summary</xref> linked to this article.</p></sec></sec><sec id="Sec14" sec-type="supplementary-material"><title>Supplementary information</title><p>
<supplementary-material content-type="local-data" id="MOESM1" position="float" orientation="portrait"><media xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="41467_2026_71090_MOESM1_ESM.pdf" position="float" orientation="portrait"><?suppdata-name 41467_2026_71090_MOESM1_ESM.pdf?><?suppdata-size 842107?><?suppdata-md5 be3a8b00835d4e57c66367353eeab97d?><?suppdata-image-server-status NEVER_LOAD?><?suppdata-mime-type application?><?suppdata-mime-sub-type pdf?><?suppdata-cloudpmc-urn urn:app:f947/13376813/be3a8b00835d/41467_2026_71090_MOESM1_ESM.pdf?><caption><p>Supplementary Information</p></caption></media></supplementary-material>
<supplementary-material content-type="local-data" id="MOESM2" position="float" orientation="portrait"><media xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="41467_2026_71090_MOESM2_ESM.pdf" position="float" orientation="portrait"><?suppdata-name 41467_2026_71090_MOESM2_ESM.pdf?><?suppdata-size 722769?><?suppdata-md5 255f68f7f15d957d29aaa1e4554ab921?><?suppdata-image-server-status NEVER_LOAD?><?suppdata-mime-type application?><?suppdata-mime-sub-type pdf?><?suppdata-cloudpmc-urn urn:app:f947/13376813/255f68f7f15d/41467_2026_71090_MOESM2_ESM.pdf?><caption><p>Peer Review File</p></caption></media></supplementary-material>
<supplementary-material content-type="local-data" id="MOESM3" position="float" orientation="portrait"><media xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="41467_2026_71090_MOESM3_ESM.pdf" position="float" orientation="portrait"><?suppdata-name 41467_2026_71090_MOESM3_ESM.pdf?><?suppdata-size 97050?><?suppdata-md5 6cd4ffc868a9ef76636c75e05b689b82?><?suppdata-image-server-status NEVER_LOAD?><?suppdata-mime-type application?><?suppdata-mime-sub-type pdf?><?suppdata-cloudpmc-urn urn:app:f947/13376813/6cd4ffc868a9/41467_2026_71090_MOESM3_ESM.pdf?><caption><p>Reporting Summary</p></caption></media></supplementary-material>
</p></sec><sec id="Sec15" sec-type="supplementary-material"><title>Source data</title><p>
<supplementary-material content-type="local-data" id="MOESM4" position="float" orientation="portrait"><media xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="41467_2026_71090_MOESM4_ESM.xlsx" position="float" orientation="portrait"><?suppdata-name 41467_2026_71090_MOESM4_ESM.xlsx?><?suppdata-size 48482?><?suppdata-md5 003fec33ae67399679c0c23e3c573e9d?><?suppdata-image-server-status NEVER_LOAD?><?suppdata-mime-type application?><?suppdata-mime-sub-type vnd.openxmlformats-officedocument.spreadsheetml.sheet?><?suppdata-cloudpmc-urn urn:app:f947/13376813/003fec33ae67/41467_2026_71090_MOESM4_ESM.xlsx?><caption><p>Source Data</p></caption></media></supplementary-material>
</p></sec></body><back><fn-group><fn><p><bold>Publisher’s note</bold> Springer Nature remains neutral with regard to jurisdictional claims in published maps and institutional affiliations.</p></fn></fn-group><sec><title>Supplementary information</title><p>The online version contains Supplementary material available at 10.1038/s41467-026-71090-y.</p></sec><ack><title>Acknowledgements</title><p>This research was funded by the Biotechnology and Biological Sciences Research Council (BBSRC) through PhenomUK-RI: The UK Plant and Crop Phenotyping Infrastructure (grant no. BB/Y512333/1). F.C. acknowledges support from the Engineering and Physical Sciences Research Council (EPSRC) through Real-time Digital Twin Assisted Surgery (grant no. EP/X033686/1). M.J.H. acknowledges support from the Biotechnology and Biological Sciences Research Council (BBSRC) through Delivering Sustainable Wheat (grant no. BB/X011003/1). S.A.T. acknowledges support of the UKRI AI programme, and the Engineering and Physical Sciences Research Council (EPSRC), for CHAI-EPSRC Causality in Healthcare AI Hub (grant no. EP/Y028856/1). F.C., S.A.T., and M.V.G. acknowledge support from the Microsoft Accelerating Foundation Models Research (AFMR) for Agricultural Foundation Models via Domain-Specific Pre-Training. We thank Jingyu Sun for exploring foundation models and Yuyang Xue for technical support during the early stages of this research.</p></ack><notes notes-type="author-contribution"><title>Author contributions</title><p>F.C. contributed to study conceptualisation, model development, case studies, model evaluation, funding acquisition, manuscript preparation and revision. I.S. contributed to model development and evaluation. A.W., D.B. contributed to technical supports for computational resources. D. Williams, F.M. contributed to advice and insights for case study 2. B.G., D. Wells, J.A.A., M.J.H., S.A.R., T.L., and T.P. contributed to manuscript revision and funding acquisition. M.V.G., S.A.T. contributed to study conceptualisation, funding acquisition, manuscript preparation and revision, and project supervision. All authors contributed to the manuscript and approved the submission.</p></notes><notes notes-type="peer-review"><title>Peer review</title><sec id="FPar1"><title>Peer review information</title><p id="Par80"><italic toggle="yes">Nature Communications</italic> thanks the anonymous reviewers for their contribution to the peer review of this work. A peer review file is available.</p></sec></notes><notes notes-type="data-availability"><title>Data availability</title><p>The <italic toggle="yes">Arabidopsis thaliana</italic> data used in case study 1 have been deposited in the Zenodo repository [10.5281/zenodo.18940282]<sup><xref ref-type="bibr" rid="CR76">76</xref></sup>. The data used for training and evaluating the computer vision model used in case study 1 are publicly available from the CVPPP2017 Leaf Segmentation Challenge dataset (A1 and A4 subsets) at CodaLab [<ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="https://codalab.lisn.upsaclay.fr/competitions/8970">https://codalab.lisn.upsaclay.fr/competitions/8970</ext-link>]. The potato data used in case study 2 are publicly available in the Zenodo repository [10.5281/zenodo.7938231]<sup><xref ref-type="bibr" rid="CR77">77</xref></sup>. The winter wheat data used in case study 3 are publicly available from the CVPPA@ICCV'23: image classification of nutrient deficiencies in winter wheat and winter rye dataset (WW2020 subset) at CodaLab [<ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="https://codalab.lisn.upsaclay.fr/competitions/13833">https://codalab.lisn.upsaclay.fr/competitions/13833</ext-link>]. <xref ref-type="sec" rid="Sec15">Source data</xref> are provided with this paper.</p></notes><notes notes-type="data-availability"><title>Code availability</title><p>The code for this research, as well as the chat logs and generated outputs of the case studies and evaluations, are available at Github [<ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="https://github.com/vios-s/PhenoAssistant/">https://github.com/vios-s/PhenoAssistant/</ext-link>]<sup><xref ref-type="bibr" rid="CR78">78</xref></sup>.</p></notes><notes id="FPar2" notes-type="COI-statement"><title>Competing interests</title><p id="Par81">The authors declare no competing interests.</p></notes><ref-list id="Bib1"><title>References</title><ref id="CR1"><label>1.</label><citation-alternatives><element-citation id="ec-CR1" publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fiorani</surname><given-names>F</given-names></name><name name-style="western"><surname>Schurr</surname><given-names>U</given-names></name></person-group><article-title>Future scenarios for plant phenotyping</article-title><source>Annu. Rev. Plant Biol.</source><year>2013</year><volume>64</volume><fpage>267</fpage><lpage>291</lpage><pub-id pub-id-type="doi">10.1146/annurev-arplant-050312-120137</pub-id><pub-id pub-id-type="pmid">23451789</pub-id></element-citation><mixed-citation id="mc-CR1" publication-type="journal">Fiorani, F. &amp; Schurr, U. Future scenarios for plant phenotyping. <italic toggle="yes">Annu. Rev. Plant Biol.</italic><bold>64</bold>, 267–291 (2013).<pub-id pub-id-type="pmid">23451789</pub-id>
<pub-id pub-id-type="doi" assigning-authority="pmc">10.1146/annurev-arplant-050312-120137</pub-id></mixed-citation></citation-alternatives></ref><ref id="CR2"><label>2.</label><mixed-citation publication-type="other">United Nations. Population. <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="https://www.un.org/en/global-issues/population">https://www.un.org/en/global-issues/population</ext-link>.</mixed-citation></ref><ref id="CR3"><label>3.</label><citation-alternatives><element-citation id="ec-CR3" publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lesk</surname><given-names>C</given-names></name><name name-style="western"><surname>Rowhani</surname><given-names>P</given-names></name><name name-style="western"><surname>Ramankutty</surname><given-names>N</given-names></name></person-group><article-title>Influence of extreme weather disasters on global crop production</article-title><source>Nature</source><year>2016</year><volume>529</volume><fpage>84</fpage><lpage>87</lpage><pub-id pub-id-type="doi">10.1038/nature16467</pub-id><pub-id pub-id-type="pmid">26738594</pub-id></element-citation><mixed-citation id="mc-CR3" publication-type="journal">Lesk, C., Rowhani, P. &amp; Ramankutty, N. Influence of extreme weather disasters on global crop production. <italic toggle="yes">Nature</italic><bold>529</bold>, 84–87 (2016).<pub-id pub-id-type="pmid">26738594</pub-id>
<pub-id pub-id-type="doi" assigning-authority="pmc">10.1038/nature16467</pub-id></mixed-citation></citation-alternatives></ref><ref id="CR4"><label>4.</label><mixed-citation publication-type="other">Murphy, K. M., Ludwig, E., Gutierrez, J. &amp; Gehan, M. A. Deep learning in image-based plant phenotyping. <italic toggle="yes">Annu. Rev. Plant Biol.</italic><bold>75</bold>, 771–795 (2024).<pub-id pub-id-type="doi" assigning-authority="pmc">10.1146/annurev-arplant-070523-042828</pub-id><pub-id pub-id-type="pmid">38382904</pub-id></mixed-citation></ref><ref id="CR5"><label>5.</label><citation-alternatives><element-citation id="ec-CR5" publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Coppens</surname><given-names>F</given-names></name><name name-style="western"><surname>Wuyts</surname><given-names>N</given-names></name><name name-style="western"><surname>Inzé</surname><given-names>D</given-names></name><name name-style="western"><surname>Dhondt</surname><given-names>S</given-names></name></person-group><article-title>Unlocking the potential of plant phenotyping data through integration and data-driven approaches</article-title><source>Curr. Opin. Syst. Biol.</source><year>2017</year><volume>4</volume><fpage>58</fpage><lpage>63</lpage><pub-id pub-id-type="doi">10.1016/j.coisb.2017.07.002</pub-id><pub-id pub-id-type="pmid">32923745</pub-id><pub-id pub-id-type="pmcid">PMC7477990</pub-id></element-citation><mixed-citation id="mc-CR5" publication-type="journal">Coppens, F., Wuyts, N., Inzé, D. &amp; Dhondt, S. Unlocking the potential of plant phenotyping data through integration and data-driven approaches. <italic toggle="yes">Curr. Opin. Syst. Biol.</italic><bold>4</bold>, 58–63 (2017).<pub-id pub-id-type="pmid">32923745</pub-id>
<pub-id pub-id-type="doi" assigning-authority="pmc">10.1016/j.coisb.2017.07.002</pub-id><pub-id pub-id-type="pmcid">PMC7477990</pub-id></mixed-citation></citation-alternatives></ref><ref id="CR6"><label>6.</label><mixed-citation publication-type="other">Achiam, J. et al. GPT-4 Technical Report. Preprint at <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="https://arxiv.org/abs/2303.08774">https://arxiv.org/abs/2303.08774</ext-link> (2024).</mixed-citation></ref><ref id="CR7"><label>7.</label><mixed-citation publication-type="other">Touvron, H. et al. LLaMA: open and efficient foundation language models. Preprint at <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="https://arxiv.org/abs/2302.13971">https://arxiv.org/abs/2302.13971</ext-link> (2023).</mixed-citation></ref><ref id="CR8"><label>8.</label><mixed-citation publication-type="other">Shen, J., Tenenholtz, N., Hall, J. B., Alvarez-Melis, D. &amp; Fusi, N. Tag-LLM: repurposing general-purpose LLMs for specialized domains. In <italic toggle="yes">Proc. 41st International Conference on Machine Learning</italic> Vol. <bold>235</bold>, 44759–44773 (JMLR.org, 2024).</mixed-citation></ref><ref id="CR9"><label>9.</label><mixed-citation publication-type="other">Yildiz, O. &amp; Peterka, T. Do large language models speak scientific workflows? In <italic toggle="yes">Proc. SC ’25 Workshops of the International Conference for High Performance Computing, Networking, Storage and Analysis</italic>, 2225–2233 (Association for Computing Machinery, 2025).</mixed-citation></ref><ref id="CR10"><label>10.</label><citation-alternatives><element-citation id="ec-CR10" publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sado</surname><given-names>F</given-names></name><name name-style="western"><surname>Loo</surname><given-names>CK</given-names></name><name name-style="western"><surname>Liew</surname><given-names>WS</given-names></name><name name-style="western"><surname>Kerzel</surname><given-names>M</given-names></name><name name-style="western"><surname>Wermter</surname><given-names>S</given-names></name></person-group><article-title>Explainable goal-driven agents and robots—a comprehensive review</article-title><source>ACM Comput. Surv.</source><year>2023</year><volume>55</volume><fpage>1</fpage><lpage>41</lpage><pub-id pub-id-type="doi">10.1145/3564240</pub-id></element-citation><mixed-citation id="mc-CR10" publication-type="journal">Sado, F., Loo, C. K., Liew, W. S., Kerzel, M. &amp; Wermter, S. Explainable goal-driven agents and robots—a comprehensive review. <italic toggle="yes">ACM Comput. Surv.</italic><bold>55</bold>, 1–41 (2023).</mixed-citation></citation-alternatives></ref><ref id="CR11"><label>11.</label><citation-alternatives><element-citation id="ec-CR11" publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Acharya</surname><given-names>DB</given-names></name><name name-style="western"><surname>Kuppan</surname><given-names>K</given-names></name><name name-style="western"><surname>Divya</surname><given-names>B</given-names></name></person-group><article-title>Agentic AI: autonomous intelligence for complex goals—a comprehensive survey</article-title><source>IEEE Access</source><year>2025</year><volume>13</volume><fpage>18912</fpage><lpage>18936</lpage><pub-id pub-id-type="doi">10.1109/ACCESS.2025.3532853</pub-id></element-citation><mixed-citation id="mc-CR11" publication-type="journal">Acharya, D. B., Kuppan, K. &amp; Divya, B. Agentic AI: autonomous intelligence for complex goals—a comprehensive survey. <italic toggle="yes">IEEE Access</italic><bold>13</bold>, 18912–18936 (2025).</mixed-citation></citation-alternatives></ref><ref id="CR12"><label>12.</label><citation-alternatives><element-citation id="ec-CR12" publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sapkota</surname><given-names>R</given-names></name><name name-style="western"><surname>Roumeliotis</surname><given-names>KI</given-names></name><name name-style="western"><surname>Karkee</surname><given-names>M</given-names></name></person-group><article-title>AI agents vs. agentic AI: a conceptual taxonomy, applications and challenges</article-title><source>Inf. Fusion</source><year>2026</year><volume>126</volume><fpage>103599</fpage><pub-id pub-id-type="doi">10.1016/j.inffus.2025.103599</pub-id></element-citation><mixed-citation id="mc-CR12" publication-type="journal">Sapkota, R., Roumeliotis, K. I. &amp; Karkee, M. AI agents vs. agentic AI: a conceptual taxonomy, applications and challenges. <italic toggle="yes">Inf. Fusion</italic><bold>126</bold>, 103599 (2026).</mixed-citation></citation-alternatives></ref><ref id="CR13"><label>13.</label><citation-alternatives><element-citation id="ec-CR13" publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hughes</surname><given-names>L</given-names></name><etal/></person-group><article-title>AI agents and agentic systems: a multi-expert analysis</article-title><source>J. Comput. Inf. Syst.</source><year>2025</year><volume>65</volume><fpage>1</fpage><lpage>29</lpage></element-citation><mixed-citation id="mc-CR13" publication-type="journal">Hughes, L. et al. AI agents and agentic systems: a multi-expert analysis. <italic toggle="yes">J. Comput. Inf. Syst.</italic><bold>65</bold>, 1–29 (2025).</mixed-citation></citation-alternatives></ref><ref id="CR14"><label>14.</label><citation-alternatives><element-citation id="ec-CR14" publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Borghoff</surname><given-names>UM</given-names></name><name name-style="western"><surname>Bottoni</surname><given-names>P</given-names></name><name name-style="western"><surname>Pareschi</surname><given-names>R</given-names></name></person-group><article-title>Human-artificial interaction in the age of agentic AI: a system-theoretical approach</article-title><source>Front. Hum. Dyn.</source><year>2025</year><volume>7</volume><fpage>1579166</fpage><pub-id pub-id-type="doi">10.3389/fhumd.2025.1579166</pub-id></element-citation><mixed-citation id="mc-CR14" publication-type="journal">Borghoff, U. M., Bottoni, P. &amp; Pareschi, R. Human-artificial interaction in the age of agentic AI: a system-theoretical approach. <italic toggle="yes">Front. Hum. Dyn.</italic><bold>7</bold>, 1579166 (2025).</mixed-citation></citation-alternatives></ref><ref id="CR15"><label>15.</label><citation-alternatives><element-citation id="ec-CR15" publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>M. Bran</surname><given-names>A</given-names></name><etal/></person-group><article-title>Augmenting large language models with chemistry tools</article-title><source>Nat. Mach. Intell.</source><year>2024</year><volume>6</volume><fpage>525</fpage><lpage>535</lpage><pub-id pub-id-type="doi">10.1038/s42256-024-00832-8</pub-id><pub-id pub-id-type="pmid">38799228</pub-id><pub-id pub-id-type="pmcid">PMC11116106</pub-id></element-citation><mixed-citation id="mc-CR15" publication-type="journal">M. Bran, A. et al. Augmenting large language models with chemistry tools. <italic toggle="yes">Nat. Mach. Intell.</italic><bold>6</bold>, 525–535 (2024).<pub-id pub-id-type="pmid">38799228</pub-id>
<pub-id pub-id-type="doi" assigning-authority="pmc">10.1038/s42256-024-00832-8</pub-id><pub-id pub-id-type="pmcid">PMC11116106</pub-id></mixed-citation></citation-alternatives></ref><ref id="CR16"><label>16.</label><citation-alternatives><element-citation id="ec-CR16" publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Boiko</surname><given-names>DA</given-names></name><name name-style="western"><surname>MacKnight</surname><given-names>R</given-names></name><name name-style="western"><surname>Kline</surname><given-names>B</given-names></name><name name-style="western"><surname>Gomes</surname><given-names>G</given-names></name></person-group><article-title>Autonomous chemical research with large language models</article-title><source>Nature</source><year>2023</year><volume>624</volume><fpage>570</fpage><lpage>578</lpage><pub-id pub-id-type="doi">10.1038/s41586-023-06792-0</pub-id><pub-id pub-id-type="pmid">38123806</pub-id><pub-id pub-id-type="pmcid">PMC10733136</pub-id></element-citation><mixed-citation id="mc-CR16" publication-type="journal">Boiko, D. A., MacKnight, R., Kline, B. &amp; Gomes, G. Autonomous chemical research with large language models. <italic toggle="yes">Nature</italic><bold>624</bold>, 570–578 (2023).<pub-id pub-id-type="pmid">38123806</pub-id>
<pub-id pub-id-type="doi" assigning-authority="pmc">10.1038/s41586-023-06792-0</pub-id><pub-id pub-id-type="pmcid">PMC10733136</pub-id></mixed-citation></citation-alternatives></ref><ref id="CR17"><label>17.</label><citation-alternatives><element-citation id="ec-CR17" publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kang</surname><given-names>Y</given-names></name><name name-style="western"><surname>Kim</surname><given-names>J</given-names></name></person-group><article-title>ChatMOF: an artificial intelligence system for predicting and generating metal-organic frameworks using large language models</article-title><source>Nat. Commun.</source><year>2024</year><volume>15</volume><fpage>4705</fpage><pub-id pub-id-type="doi">10.1038/s41467-024-48998-4</pub-id><pub-id pub-id-type="pmid">38830856</pub-id><pub-id pub-id-type="pmcid">PMC11148193</pub-id></element-citation><mixed-citation id="mc-CR17" publication-type="journal">Kang, Y. &amp; Kim, J. ChatMOF: an artificial intelligence system for predicting and generating metal-organic frameworks using large language models. <italic toggle="yes">Nat. Commun.</italic><bold>15</bold>, 4705 (2024).<pub-id pub-id-type="pmid">38830856</pub-id>
<pub-id pub-id-type="doi" assigning-authority="pmc">10.1038/s41467-024-48998-4</pub-id><pub-id pub-id-type="pmcid">PMC11148193</pub-id></mixed-citation></citation-alternatives></ref><ref id="CR18"><label>18.</label><citation-alternatives><element-citation id="ec-CR18" publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ghafarollahi</surname><given-names>A</given-names></name><name name-style="western"><surname>Buehler</surname><given-names>MJ</given-names></name></person-group><article-title>Automating alloy design and discovery with physics-aware multimodal multiagent AI</article-title><source>Proc. Natl. Acad. Sci. USA</source><year>2025</year><volume>122</volume><fpage>e2414074122</fpage><pub-id pub-id-type="doi">10.1073/pnas.2414074122</pub-id><pub-id pub-id-type="pmid">39854228</pub-id><pub-id pub-id-type="pmcid">PMC11789045</pub-id></element-citation><mixed-citation id="mc-CR18" publication-type="journal">Ghafarollahi, A. &amp; Buehler, M. J. Automating alloy design and discovery with physics-aware multimodal multiagent AI. <italic toggle="yes">Proc. Natl. Acad. Sci. USA</italic><bold>122</bold>, e2414074122 (2025).<pub-id pub-id-type="pmid">39854228</pub-id>
<pub-id pub-id-type="doi" assigning-authority="pmc">10.1073/pnas.2414074122</pub-id><pub-id pub-id-type="pmcid">PMC11789045</pub-id></mixed-citation></citation-alternatives></ref><ref id="CR19"><label>19.</label><citation-alternatives><element-citation id="ec-CR19" publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lei</surname><given-names>W</given-names></name><etal/></person-group><article-title>Chatbot: a community-driven AI assistant for integrative computational bioimaging</article-title><source>Nat. Methods</source><year>2024</year><volume>21</volume><fpage>1368</fpage><lpage>1370</lpage><pub-id pub-id-type="doi">10.1038/s41592-024-02370-y</pub-id><pub-id pub-id-type="pmid">39122937</pub-id></element-citation><mixed-citation id="mc-CR19" publication-type="journal">Lei, W. et al. Chatbot: a community-driven AI assistant for integrative computational bioimaging. <italic toggle="yes">Nat. Methods</italic><bold>21</bold>, 1368–1370 (2024).<pub-id pub-id-type="pmid">39122937</pub-id>
<pub-id pub-id-type="doi" assigning-authority="pmc">10.1038/s41592-024-02370-y</pub-id></mixed-citation></citation-alternatives></ref><ref id="CR20"><label>20.</label><citation-alternatives><element-citation id="ec-CR20" publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Royer</surname><given-names>LA</given-names></name></person-group><article-title>Omega—harnessing the power of large language models for bioimage analysis</article-title><source>Nat. Methods</source><year>2024</year><volume>21</volume><fpage>1371</fpage><lpage>1373</lpage><pub-id pub-id-type="doi">10.1038/s41592-024-02310-w</pub-id><pub-id pub-id-type="pmid">38858592</pub-id></element-citation><mixed-citation id="mc-CR20" publication-type="journal">Royer, L. A. Omega—harnessing the power of large language models for bioimage analysis. <italic toggle="yes">Nat. Methods</italic><bold>21</bold>, 1371–1373 (2024).<pub-id pub-id-type="pmid">38858592</pub-id>
<pub-id pub-id-type="doi" assigning-authority="pmc">10.1038/s41592-024-02310-w</pub-id></mixed-citation></citation-alternatives></ref><ref id="CR21"><label>21.</label><citation-alternatives><element-citation id="ec-CR21" publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tu</surname><given-names>T</given-names></name><etal/></person-group><article-title>Towards conversational diagnostic artificial intelligence</article-title><source>Nature</source><year>2025</year><volume>642</volume><fpage>442</fpage><lpage>450</lpage><pub-id pub-id-type="doi">10.1038/s41586-025-08866-7</pub-id><pub-id pub-id-type="pmid">40205050</pub-id><pub-id pub-id-type="pmcid">PMC12158756</pub-id></element-citation><mixed-citation id="mc-CR21" publication-type="journal">Tu, T. et al. Towards conversational diagnostic artificial intelligence. <italic toggle="yes">Nature</italic><bold>642</bold>, 442–450 (2025).<pub-id pub-id-type="pmid">40205050</pub-id>
<pub-id pub-id-type="doi" assigning-authority="pmc">10.1038/s41586-025-08866-7</pub-id><pub-id pub-id-type="pmcid">PMC12158756</pub-id></mixed-citation></citation-alternatives></ref><ref id="CR22"><label>22.</label><citation-alternatives><element-citation id="ec-CR22" publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Singhal</surname><given-names>K</given-names></name><etal/></person-group><article-title>Toward expert-level medical question answering with large language models</article-title><source>Nat. Med.</source><year>2025</year><volume>31</volume><fpage>943</fpage><lpage>950</lpage><pub-id pub-id-type="doi">10.1038/s41591-024-03423-7</pub-id><pub-id pub-id-type="pmid">39779926</pub-id><pub-id pub-id-type="pmcid">PMC11922739</pub-id></element-citation><mixed-citation id="mc-CR22" publication-type="journal">Singhal, K. et al. Toward expert-level medical question answering with large language models. <italic toggle="yes">Nat. Med.</italic><bold>31</bold>, 943–950 (2025).<pub-id pub-id-type="pmid">39779926</pub-id>
<pub-id pub-id-type="doi" assigning-authority="pmc">10.1038/s41591-024-03423-7</pub-id><pub-id pub-id-type="pmcid">PMC11922739</pub-id></mixed-citation></citation-alternatives></ref><ref id="CR23"><label>23.</label><mixed-citation publication-type="other">Gottweis, J. et al. Towards an AI co-scientist. Preprint at <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="https://arxiv.org/abs/2502.18864">https://arxiv.org/abs/2502.18864</ext-link> (2025).</mixed-citation></ref><ref id="CR24"><label>24.</label><mixed-citation publication-type="other">Yang, X., Gao, J., Xue, W. &amp; Alexandersson, E. PLLaMA: an open-source large language model for plant science. Preprint at <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="https://arxiv.org/abs/2401.01600">https://arxiv.org/abs/2401.01600</ext-link> (2024).</mixed-citation></ref><ref id="CR25"><label>25.</label><mixed-citation publication-type="other">Awais, M., Salem Abdulla Alharthi, A. H., Kumar, A., Cholakkal, H. &amp; Anwer, R. M. AgroGPT: efficient agricultural vision-language model with expert tuning. In <italic toggle="yes">Proc. 2025 IEEE/CVF Winter Conference on Applications of Computer Vision (WACV)</italic> 5687–5696, 10.1109/WACV61041.2025.00555 (2025).</mixed-citation></ref><ref id="CR26"><label>26.</label><mixed-citation publication-type="other">Yang, S. et al. ShizishanGPT: an agricultural large language model integrating tools and resources. In <italic toggle="yes">Proc. International Conference on Web Information Systems Engineering</italic>, 284–298 (Springer, 2024).</mixed-citation></ref><ref id="CR27"><label>27.</label><citation-alternatives><element-citation id="ec-CR27" publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>Y</given-names></name><etal/></person-group><article-title>IPM-AgriGPT: a large language model for pest and disease management with a G-EA framework and agricultural contextual reasoning</article-title><source>Mathematics</source><year>2025</year><volume>13</volume><fpage>566</fpage><pub-id pub-id-type="doi">10.3390/math13040566</pub-id></element-citation><mixed-citation id="mc-CR27" publication-type="journal">Zhang, Y. et al. IPM-AgriGPT: a large language model for pest and disease management with a G-EA framework and agricultural contextual reasoning. <italic toggle="yes">Mathematics</italic><bold>13</bold>, 566 (2025).</mixed-citation></citation-alternatives></ref><ref id="CR28"><label>28.</label><citation-alternatives><element-citation id="ec-CR28" publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ravindran</surname><given-names>DJS</given-names></name><name name-style="western"><surname>Skarga-Bandurova</surname><given-names>I</given-names></name><name name-style="western"><surname>V</surname><given-names>S</given-names></name><name name-style="western"><surname>Awais</surname><given-names>M</given-names></name><name name-style="western"><surname>S</surname><given-names>M</given-names></name></person-group><article-title>AgroLLM: connecting farmers and agricultural practices through large language models for enhanced knowledge transfer and practical application</article-title><source>AgriEngineering</source><year>2026</year><volume>8</volume><fpage>38</fpage><pub-id pub-id-type="doi">10.3390/agriengineering8010038</pub-id></element-citation><mixed-citation id="mc-CR28" publication-type="journal">Ravindran, D. J. S., Skarga-Bandurova, I., V, S., Awais, M. &amp; S, M. AgroLLM: connecting farmers and agricultural practices through large language models for enhanced knowledge transfer and practical application. <italic toggle="yes">AgriEngineering</italic><bold>8</bold>, 38 (2026).</mixed-citation></citation-alternatives></ref><ref id="CR29"><label>29.</label><mixed-citation publication-type="other">Arshad, M. A. et al. Leveraging vision language models for specialized agricultural tasks. In <italic toggle="yes">Proc. 2025 IEEE/CVF Winter Conference on Applications of Computer Vision (WACV)</italic> 6320–6329, 10.1109/WACV61041.2025.00616 (2025).</mixed-citation></ref><ref id="CR30"><label>30.</label><citation-alternatives><element-citation id="ec-CR30" publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhao</surname><given-names>X</given-names></name><etal/></person-group><article-title>Implementation of large language models and agricultural knowledge graphs for efficient plant disease detection</article-title><source>Agriculture</source><year>2024</year><volume>14</volume><fpage>1359</fpage><pub-id pub-id-type="doi">10.3390/agriculture14081359</pub-id></element-citation><mixed-citation id="mc-CR30" publication-type="journal">Zhao, X. et al. Implementation of large language models and agricultural knowledge graphs for efficient plant disease detection. <italic toggle="yes">Agriculture</italic><bold>14</bold>, 1359 (2024).</mixed-citation></citation-alternatives></ref><ref id="CR31"><label>31.</label><citation-alternatives><element-citation id="ec-CR31" publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Roumeliotis</surname><given-names>KI</given-names></name><name name-style="western"><surname>Sapkota</surname><given-names>R</given-names></name><name name-style="western"><surname>Karkee</surname><given-names>M</given-names></name><name name-style="western"><surname>Tselikas</surname><given-names>ND</given-names></name></person-group><article-title>Agentic AI with orchestrator-agent trust: a modular visual classification framework with trust-aware orchestration and rag-based reasoning</article-title><source>IEEE Access</source><year>2026</year><volume>14</volume><fpage>26965</fpage><lpage>26982</lpage><pub-id pub-id-type="doi">10.1109/ACCESS.2026.3662282</pub-id></element-citation><mixed-citation id="mc-CR31" publication-type="journal">Roumeliotis, K. I., Sapkota, R., Karkee, M. &amp; Tselikas, N. D. Agentic AI with orchestrator-agent trust: a modular visual classification framework with trust-aware orchestration and rag-based reasoning. <italic toggle="yes">IEEE Access</italic><bold>14</bold>, 26965–26982 (2026).</mixed-citation></citation-alternatives></ref><ref id="CR32"><label>32.</label><mixed-citation publication-type="other">Sapkota, R., Roumeliotis, K. I. &amp; Karkee, M. UAVs meet agentic AI: a multidomain survey of autonomous aerial intelligence and agentic UAVs. Preprint at <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="https://arxiv.org/abs/2506.08045">https://arxiv.org/abs/2506.08045</ext-link> (2025).</mixed-citation></ref><ref id="CR33"><label>33.</label><citation-alternatives><element-citation id="ec-CR33" publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Benegas</surname><given-names>G</given-names></name><name name-style="western"><surname>Batra</surname><given-names>SS</given-names></name><name name-style="western"><surname>Song</surname><given-names>YS</given-names></name></person-group><article-title>DNA language models are powerful predictors of genome-wide variant effects</article-title><source>Proc. Natl. Acad. Sci. USA</source><year>2023</year><volume>120</volume><fpage>e2311219120</fpage><pub-id pub-id-type="doi">10.1073/pnas.2311219120</pub-id><pub-id pub-id-type="pmid">37883436</pub-id><pub-id pub-id-type="pmcid">PMC10622914</pub-id></element-citation><mixed-citation id="mc-CR33" publication-type="journal">Benegas, G., Batra, S. S. &amp; Song, Y. S. DNA language models are powerful predictors of genome-wide variant effects. <italic toggle="yes">Proc. Natl. Acad. Sci. USA</italic><bold>120</bold>, e2311219120 (2023).<pub-id pub-id-type="pmid">37883436</pub-id>
<pub-id pub-id-type="doi" assigning-authority="pmc">10.1073/pnas.2311219120</pub-id><pub-id pub-id-type="pmcid">PMC10622914</pub-id></mixed-citation></citation-alternatives></ref><ref id="CR34"><label>34.</label><citation-alternatives><element-citation id="ec-CR34" publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mendoza-Revilla</surname><given-names>J</given-names></name><etal/></person-group><article-title>A foundational large language model for edible plant genomes</article-title><source>Commun. Biol.</source><year>2024</year><volume>7</volume><fpage>835</fpage><pub-id pub-id-type="doi">10.1038/s42003-024-06465-2</pub-id><pub-id pub-id-type="pmid">38982288</pub-id><pub-id pub-id-type="pmcid">PMC11233511</pub-id></element-citation><mixed-citation id="mc-CR34" publication-type="journal">Mendoza-Revilla, J. et al. A foundational large language model for edible plant genomes. <italic toggle="yes">Commun. Biol.</italic><bold>7</bold>, 835 (2024).<pub-id pub-id-type="pmid">38982288</pub-id>
<pub-id pub-id-type="doi" assigning-authority="pmc">10.1038/s42003-024-06465-2</pub-id><pub-id pub-id-type="pmcid">PMC11233511</pub-id></mixed-citation></citation-alternatives></ref><ref id="CR35"><label>35.</label><citation-alternatives><element-citation id="ec-CR35" publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>R</given-names></name><etal/></person-group><article-title>PlantGPT: an Arabidopsis-based intelligent agent that answers questions about plant functional genomics</article-title><source>Adv. Sci.</source><year>2025</year><volume>12</volume><fpage>e03926</fpage><pub-id pub-id-type="doi">10.1002/advs.202503926</pub-id><pub-id pub-id-type="pmcid">PMC12376578</pub-id><pub-id pub-id-type="pmid">40397417</pub-id></element-citation><mixed-citation id="mc-CR35" publication-type="journal">Zhang, R. et al. PlantGPT: an Arabidopsis-based intelligent agent that answers questions about plant functional genomics. <italic toggle="yes">Adv. Sci.</italic><bold>12</bold>, e03926 (2025).<pub-id pub-id-type="doi" assigning-authority="pmc">10.1002/advs.202503926</pub-id><pub-id pub-id-type="pmcid">PMC12376578</pub-id><pub-id pub-id-type="pmid">40397417</pub-id></mixed-citation></citation-alternatives></ref><ref id="CR36"><label>36.</label><mixed-citation publication-type="other">Team, G. et al. Gemini: a family of highly capable multimodal models. Preprint at 10.48550/arXiv.2312.11805 (2025).</mixed-citation></ref><ref id="CR37"><label>37.</label><citation-alternatives><element-citation id="ec-CR37" publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>H</given-names></name><name name-style="western"><surname>Li</surname><given-names>C</given-names></name><name name-style="western"><surname>Wu</surname><given-names>Q</given-names></name><name name-style="western"><surname>Lee</surname><given-names>YJ</given-names></name></person-group><article-title>Visual instruction tuning</article-title><source>Adv. Neural Inf. Process. Syst.</source><year>2023</year><volume>36</volume><fpage>34892</fpage><lpage>34916</lpage><pub-id pub-id-type="doi">10.52202/075280-1516</pub-id></element-citation><mixed-citation id="mc-CR37" publication-type="journal">Liu, H., Li, C., Wu, Q. &amp; Lee, Y. J. Visual instruction tuning. <italic toggle="yes">Adv. Neural Inf. Process. Syst.</italic><bold>36</bold>, 34892–34916 (2023).</mixed-citation></citation-alternatives></ref><ref id="CR38"><label>38.</label><citation-alternatives><element-citation id="ec-CR38" publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Minervini</surname><given-names>M</given-names></name><name name-style="western"><surname>Giuffrida</surname><given-names>MV</given-names></name><name name-style="western"><surname>Perata</surname><given-names>P</given-names></name><name name-style="western"><surname>Tsaftaris</surname><given-names>SA</given-names></name></person-group><article-title>Phenotiki: an open software and hardware platform for affordable and easy image-based phenotyping of rosette-shaped plants</article-title><source> Plant J.</source><year>2017</year><volume>90</volume><fpage>204</fpage><lpage>216</lpage><pub-id pub-id-type="doi">10.1111/tpj.13472</pub-id><pub-id pub-id-type="pmid">28066963</pub-id></element-citation><mixed-citation id="mc-CR38" publication-type="journal">Minervini, M., Giuffrida, M. V., Perata, P. &amp; Tsaftaris, S. A. Phenotiki: an open software and hardware platform for affordable and easy image-based phenotyping of rosette-shaped plants. <italic toggle="yes"> Plant J.</italic><bold>90</bold>, 204–216 (2017).<pub-id pub-id-type="pmid">28066963</pub-id>
<pub-id pub-id-type="doi" assigning-authority="pmc">10.1111/tpj.13472</pub-id></mixed-citation></citation-alternatives></ref><ref id="CR39"><label>39.</label><citation-alternatives><element-citation id="ec-CR39" publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Williams</surname><given-names>D</given-names></name><name name-style="western"><surname>Macfarlane</surname><given-names>F</given-names></name><name name-style="western"><surname>Britten</surname><given-names>A</given-names></name></person-group><article-title>Leaf only SAM: a segment anything pipeline for zero-shot automated leaf segmentation</article-title><source>Smart Agric. Technol.</source><year>2024</year><volume>8</volume><fpage>100515</fpage><pub-id pub-id-type="doi">10.1016/j.atech.2024.100515</pub-id></element-citation><mixed-citation id="mc-CR39" publication-type="journal">Williams, D., Macfarlane, F. &amp; Britten, A. Leaf only SAM: a segment anything pipeline for zero-shot automated leaf segmentation. <italic toggle="yes">Smart Agric. Technol.</italic><bold>8</bold>, 100515 (2024).</mixed-citation></citation-alternatives></ref><ref id="CR40"><label>40.</label><mixed-citation publication-type="other">Hu, E. J. et al. Lora: low-rank adaptation of large language models. In <italic toggle="yes">Proc. International Conference on Learning Representations</italic> (OpenReview.net, 2022).</mixed-citation></ref><ref id="CR41"><label>41.</label><citation-alternatives><element-citation id="ec-CR41" publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yi</surname><given-names>J</given-names></name><etal/></person-group><article-title>Deep learning for non-invasive diagnosis of nutrient deficiencies in sugar beet using RGB images</article-title><source>Sensors</source><year>2020</year><volume>20</volume><fpage>5893</fpage><pub-id pub-id-type="doi">10.3390/s20205893</pub-id><pub-id pub-id-type="pmid">33080979</pub-id><pub-id pub-id-type="pmcid">PMC7589690</pub-id></element-citation><mixed-citation id="mc-CR41" publication-type="journal">Yi, J. et al. Deep learning for non-invasive diagnosis of nutrient deficiencies in sugar beet using RGB images. <italic toggle="yes">Sensors</italic><bold>20</bold>, 5893 (2020).<pub-id pub-id-type="pmid">33080979</pub-id>
<pub-id pub-id-type="doi" assigning-authority="pmc">10.3390/s20205893</pub-id><pub-id pub-id-type="pmcid">PMC7589690</pub-id></mixed-citation></citation-alternatives></ref><ref id="CR42"><label>42.</label><citation-alternatives><element-citation id="ec-CR42" publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yi</surname><given-names>J</given-names></name><etal/></person-group><article-title>Non-invasive diagnosis of nutrient deficiencies in winter wheat and winter rye using UAV-based RGB images</article-title><source>Comput. Electron. Agric.</source><year>2025</year><volume>239</volume><fpage>110865</fpage><pub-id pub-id-type="doi">10.1016/j.compag.2025.110865</pub-id></element-citation><mixed-citation id="mc-CR42" publication-type="journal">Yi, J. et al. Non-invasive diagnosis of nutrient deficiencies in winter wheat and winter rye using UAV-based RGB images. <italic toggle="yes">Comput. Electron. Agric.</italic><bold>239</bold>, 110865 (2025).</mixed-citation></citation-alternatives></ref><ref id="CR43"><label>43.</label><mixed-citation publication-type="other">DeepSeek-AI et al. DeepSeek LLM: scaling open-source language models with longtermism. Preprint at 10.48550/arXiv.2401.02954 (2024).</mixed-citation></ref><ref id="CR44"><label>44.</label><citation-alternatives><element-citation id="ec-CR44" publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Guo</surname><given-names>D</given-names></name><etal/></person-group><article-title>DeepSeek-R1 incentivizes reasoning in LLMs through reinforcement learning</article-title><source>Nature</source><year>2025</year><volume>645</volume><fpage>633</fpage><lpage>638</lpage><pub-id pub-id-type="doi">10.1038/s41586-025-09422-z</pub-id><pub-id pub-id-type="pmid">40962978</pub-id><pub-id pub-id-type="pmcid">PMC12443585</pub-id></element-citation><mixed-citation id="mc-CR44" publication-type="journal">Guo, D. et al. DeepSeek-R1 incentivizes reasoning in LLMs through reinforcement learning. <italic toggle="yes">Nature</italic><bold>645</bold>, 633–638 (2025).<pub-id pub-id-type="pmid">40962978</pub-id>
<pub-id pub-id-type="doi" assigning-authority="pmc">10.1038/s41586-025-09422-z</pub-id><pub-id pub-id-type="pmcid">PMC12443585</pub-id></mixed-citation></citation-alternatives></ref><ref id="CR45"><label>45.</label><mixed-citation publication-type="other">Gu, J. et al. A survey on LLM-as-a-Judge. <italic toggle="yes">The Innovation</italic> 101253 10.48550/arXiv.2411.15594 (2026).<pub-id pub-id-type="doi" assigning-authority="pmc">10.1016/j.xinn.2025.101253</pub-id><pub-id pub-id-type="pmcid">PMC13237853</pub-id><pub-id pub-id-type="pmid">42254963</pub-id></mixed-citation></ref><ref id="CR46"><label>46.</label><mixed-citation publication-type="other">Golechha, S. &amp; Garriga-Alonso, A. Among Us: a sandbox for measuring and detecting agentic deception. In <italic toggle="yes">Proc. the Thirty-Ninth Annual Conference on Neural Information Processing Systems</italic> (NeurIPS, 2025).</mixed-citation></ref><ref id="CR47"><label>47.</label><mixed-citation publication-type="other">Sapkota, R., Roumeliotis, K. I. &amp; Karkee, M. Vibe coding vs. agentic coding: fundamentals and practical implications of agentic AI. Preprint at <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="https://arxiv.org/abs/2505.19443">https://arxiv.org/abs/2505.19443</ext-link> (2025).</mixed-citation></ref><ref id="CR48"><label>48.</label><mixed-citation publication-type="other">Averly, R. &amp; Chao, W.-L. Unified out-of-distribution detection: a model-specific perspective. In <italic toggle="yes">Proc. 2023 IEEE/CVF International Conference on Computer Vision (ICCV)</italic> 1453–1463, 10.1109/ICCV51070.2023.00140 (2023).</mixed-citation></ref><ref id="CR49"><label>49.</label><mixed-citation publication-type="other">Miyai, A. et al. Generalized out-of-distribution detection and beyond in vision language model era: a survey. <italic toggle="yes">Transactions on Machine Learning Research</italic><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="https://openreview.net/forum?id=FO3IA4lUEY">https://openreview.net/forum?id=FO3IA4lUEY</ext-link> (2025).</mixed-citation></ref><ref id="CR50"><label>50.</label><mixed-citation publication-type="other">Tonmoy, M. R., Hossain, M. M., Dey, N. &amp; Mridha, M. MobilePlantViT: a mobile-friendly hybrid ViT for generalized plant disease image classification. Preprint at <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="https://arxiv.org/abs/2503.16628">https://arxiv.org/abs/2503.16628</ext-link> (2025).</mixed-citation></ref><ref id="CR51"><label>51.</label><mixed-citation publication-type="other">Han, B. et al. FoMo4Wheat: toward reliable crop vision foundation models with globally curated data. Preprint at 10.48550/arXiv.2509.06907 (2025).</mixed-citation></ref><ref id="CR52"><label>52.</label><mixed-citation publication-type="other">Hugging Face. Hugging Face–The AI community building the future <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="https://huggingface.co/">https://huggingface.co/</ext-link>.</mixed-citation></ref><ref id="CR53"><label>53.</label><mixed-citation publication-type="other">Kaggle. Kaggle: Your Machine Learning and Data Science Community <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="https://www.kaggle.com/">https://www.kaggle.com/</ext-link>.</mixed-citation></ref><ref id="CR54"><label>54.</label><mixed-citation publication-type="other">Model Context Protocol. What is the Model Context Protocol (MCP)? <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="https://modelcontextprotocol.io/docs/getting-started/intro">https://modelcontextprotocol.io/docs/getting-started/intro</ext-link>.</mixed-citation></ref><ref id="CR55"><label>55.</label><citation-alternatives><element-citation id="ec-CR55" publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Geerling</surname><given-names>W</given-names></name><name name-style="western"><surname>Mateer</surname><given-names>GD</given-names></name><name name-style="western"><surname>Wooten</surname><given-names>J</given-names></name><name name-style="western"><surname>Damodaran</surname><given-names>N</given-names></name></person-group><article-title>ChatGPT has aced the test of understanding in college economics: now what?</article-title><source> Am. Econ.</source><year>2023</year><volume>68</volume><fpage>233</fpage><lpage>245</lpage></element-citation><mixed-citation id="mc-CR55" publication-type="journal">Geerling, W., Mateer, G. D., Wooten, J. &amp; Damodaran, N. ChatGPT has aced the test of understanding in college economics: now what? <italic toggle="yes"> Am. Econ.</italic><bold>68</bold>, 233–245 (2023).</mixed-citation></citation-alternatives></ref><ref id="CR56"><label>56.</label><citation-alternatives><element-citation id="ec-CR56" publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>Y</given-names></name><name name-style="western"><surname>Liu</surname><given-names>TX</given-names></name><name name-style="western"><surname>Shan</surname><given-names>Y</given-names></name><name name-style="western"><surname>Zhong</surname><given-names>S</given-names></name></person-group><article-title>The emergence of economic rationality of GPT</article-title><source>Proc. Natl. Acad. Sci. USA</source><year>2023</year><volume>120</volume><fpage>e2316205120</fpage><pub-id pub-id-type="doi">10.1073/pnas.2316205120</pub-id><pub-id pub-id-type="pmid">38085780</pub-id><pub-id pub-id-type="pmcid">PMC10740389</pub-id></element-citation><mixed-citation id="mc-CR56" publication-type="journal">Chen, Y., Liu, T. X., Shan, Y. &amp; Zhong, S. The emergence of economic rationality of GPT. <italic toggle="yes">Proc. Natl. Acad. Sci. USA</italic><bold>120</bold>, e2316205120 (2023).<pub-id pub-id-type="pmid">38085780</pub-id>
<pub-id pub-id-type="doi" assigning-authority="pmc">10.1073/pnas.2316205120</pub-id><pub-id pub-id-type="pmcid">PMC10740389</pub-id></mixed-citation></citation-alternatives></ref><ref id="CR57"><label>57.</label><mixed-citation publication-type="other">Shah, D., Osiński, B., Ichter, B. &amp; Levine, S. LM-Nav: robotic navigation with large pre-trained models of language, vision, and action. In <italic toggle="yes">Proc. 6th Conference on Robot Learning</italic>, 492–504 (PMLR, 2023).</mixed-citation></ref><ref id="CR58"><label>58.</label><mixed-citation publication-type="other">Cui, C., Ma, Y., Cao, X., Ye, W. &amp; Wang, Z. Drive as you speak: enabling human-like interaction with large language models in autonomous vehicles. In <italic toggle="yes">Proc. 2024 IEEE/CVF Winter Conference on Applications of Computer Vision Workshops (WACVW)</italic> 902–909, 10.1109/WACVW60836.2024.00101 (2024).</mixed-citation></ref><ref id="CR59"><label>59.</label><citation-alternatives><element-citation id="ec-CR59" publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gao</surname><given-names>C</given-names></name><etal/></person-group><article-title>Large language models empowered agent-based modeling and simulation: a survey and perspectives</article-title><source>Humanit. Soc. Sci. Commun.</source><year>2024</year><volume>11</volume><fpage>1259</fpage><pub-id pub-id-type="doi">10.1057/s41599-024-03611-3</pub-id></element-citation><mixed-citation id="mc-CR59" publication-type="journal">Gao, C. et al. Large language models empowered agent-based modeling and simulation: a survey and perspectives. <italic toggle="yes">Humanit. Soc. Sci. Commun.</italic><bold>11</bold>, 1259 (2024).</mixed-citation></citation-alternatives></ref><ref id="CR60"><label>60.</label><mixed-citation publication-type="other">Han, K., Kuang, K., Zhao, Z., Ye, J. &amp; Wu, F. Causal agent based on large language model. Preprint at <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="https://arxiv.org/abs/2408.06849">https://arxiv.org/abs/2408.06849</ext-link> (2024).</mixed-citation></ref><ref id="CR61"><label>61.</label><mixed-citation publication-type="other">Wang, X. et al. Causal-copilot: an autonomous causal analysis agent. Preprint at <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="https://arxiv.org/abs/2504.13263">https://arxiv.org/abs/2504.13263</ext-link> (2025).</mixed-citation></ref><ref id="CR62"><label>62.</label><citation-alternatives><element-citation id="ec-CR62" publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chi</surname><given-names>H</given-names></name><etal/></person-group><article-title>Unveiling causal reasoning in large language models: reality or mirage?</article-title><source>Adv. Neural Inf. Process. Syst.</source><year>2024</year><volume>37</volume><fpage>96640</fpage><lpage>96670</lpage><pub-id pub-id-type="doi">10.52202/079017-3064</pub-id></element-citation><mixed-citation id="mc-CR62" publication-type="journal">Chi, H. et al. Unveiling causal reasoning in large language models: reality or mirage? <italic toggle="yes">Adv. Neural Inf. Process. Syst.</italic><bold>37</bold>, 96640–96670 (2024).</mixed-citation></citation-alternatives></ref><ref id="CR63"><label>63.</label><mixed-citation publication-type="other">Qian, C. et al. ChatDev: communicative agents for software development. In <italic toggle="yes">Proc. 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)</italic> (eds. Ku, L.-W., Martins, A. &amp; Srikumar, V.) 15174–15186 (Association for Computational Linguistics, 2024).</mixed-citation></ref><ref id="CR64"><label>64.</label><mixed-citation publication-type="other">Yang, Y. et al. AgentNet: decentralized evolutionary coordination for LLM-based multi-agent systems. In <italic toggle="yes">Proc. the Thirty-Ninth Annual Conference on Neural Information Processing Systems</italic> (NeurIPS, 2025).</mixed-citation></ref><ref id="CR65"><label>65.</label><mixed-citation publication-type="other">Wu, Q. et al. AutoGen: enabling next-gen LLM applications via multi-agent conversations. In <italic toggle="yes">Proc. First Conference on Language Modeling</italic> (OpenReview.net, 2024).</mixed-citation></ref><ref id="CR66"><label>66.</label><mixed-citation publication-type="other">CrewAI. CrewAI: the leading multi-agent platform <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="https://www.crewai.com/">https://www.crewai.com/</ext-link>.</mixed-citation></ref><ref id="CR67"><label>67.</label><mixed-citation publication-type="other">Li, H. et al. Advancing collaborative debates with role differentiation through multi-agent reinforcement learning. In <italic toggle="yes">Proc. 63rd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)</italic> (eds. Che, W., Nabende, J., Shutova, E. &amp; Pilehvar, M. T.) 22655–22666 (Association for Computational Linguistics, 2025).</mixed-citation></ref><ref id="CR68"><label>68.</label><mixed-citation publication-type="other">Lu, S., Shao, J., Luo, B. &amp; Lin, T. MorphAgent: empowering agents through self-evolving profiles and decentralized collaboration. Preprint at 10.48550/arXiv.2410.15048 (2025).</mixed-citation></ref><ref id="CR69"><label>69.</label><citation-alternatives><element-citation id="ec-CR69" publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Raza</surname><given-names>S</given-names></name><name name-style="western"><surname>Sapkota</surname><given-names>R</given-names></name><name name-style="western"><surname>Karkee</surname><given-names>M</given-names></name><name name-style="western"><surname>Emmanouilidis</surname><given-names>C</given-names></name></person-group><article-title>TRiSM for agentic AI: a review of trust, risk, and security management in LLM-based agentic multi-agent systems</article-title><source>AI Open</source><year>2026</year><volume>7</volume><fpage>71</fpage><lpage>95</lpage><pub-id pub-id-type="doi">10.1016/j.aiopen.2026.02.006</pub-id></element-citation><mixed-citation id="mc-CR69" publication-type="journal">Raza, S., Sapkota, R., Karkee, M. &amp; Emmanouilidis, C. TRiSM for agentic AI: a review of trust, risk, and security management in LLM-based agentic multi-agent systems. <italic toggle="yes">AI Open</italic><bold>7</bold>, 71–95 (2026).</mixed-citation></citation-alternatives></ref><ref id="CR70"><label>70.</label><mixed-citation publication-type="other">Google Cloud. Announcing the Agent2Agent Protocol (A2A) <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="https://developers.googleblog.com/en/a2a-a-new-era-of-agent-interoperability/">https://developers.googleblog.com/en/a2a-a-new-era-of-agent-interoperability/</ext-link>.</mixed-citation></ref><ref id="CR71"><label>71.</label><mixed-citation publication-type="other">Cheng, B., Misra, I., Schwing, A. G., Kirillov, A. &amp; Girdhar, R. Masked-attention mask transformer for universal image segmentation. In <italic toggle="yes">Proc. 2022 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</italic> 1280–1289, 10.1109/CVPR52688.2022.00135 (2022).</mixed-citation></ref><ref id="CR72"><label>72.</label><mixed-citation publication-type="other">Chen, F., Tsaftaris, S. A. &amp; Giuffrida, M. V. GMT: Guided mask transformer for leaf instance segmentation. In <italic toggle="yes">Proc. 2025 IEEE/CVF Winter Conference on Applications of Computer Vision (WACV)</italic> 1217–1226, 10.1109/WACV61041.2025.00126 (2025).</mixed-citation></ref><ref id="CR73"><label>73.</label><citation-alternatives><element-citation id="ec-CR73" publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Minervini</surname><given-names>M</given-names></name><name name-style="western"><surname>Fischbach</surname><given-names>A</given-names></name><name name-style="western"><surname>Scharr</surname><given-names>H</given-names></name><name name-style="western"><surname>Tsaftaris</surname><given-names>SA</given-names></name></person-group><article-title>Finely-grained annotated datasets for image-based plant phenotyping</article-title><source>Pattern Recognit. Lett.</source><year>2016</year><volume>81</volume><fpage>80</fpage><lpage>89</lpage><pub-id pub-id-type="doi">10.1016/j.patrec.2015.10.013</pub-id></element-citation><mixed-citation id="mc-CR73" publication-type="journal">Minervini, M., Fischbach, A., Scharr, H. &amp; Tsaftaris, S. A. Finely-grained annotated datasets for image-based plant phenotyping. <italic toggle="yes">Pattern Recognit. Lett.</italic><bold>81</bold>, 80–89 (2016).</mixed-citation></citation-alternatives></ref><ref id="CR74"><label>74.</label><mixed-citation publication-type="other">Chen, F., Giuffrida, M. V. &amp; Tsaftaris, S. A. Adapting vision foundation models for plant phenotyping. In <italic toggle="yes">Proc. 2023 IEEE/CVF International Conference on Computer Vision Workshops (ICCVW)</italic> 604–613, 10.1109/ICCVW60793.2023.00067 (2023).</mixed-citation></ref><ref id="CR75"><label>75.</label><mixed-citation publication-type="other">Oquab, M. et al. DINOv2: Learning Robust Visual Features without Supervision. <italic toggle="yes">Transactions on Machine Learning Research</italic><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="https://openreview.net/forum?id=a68SUt6zFt">https://openreview.net/forum?id=a68SUt6zFt</ext-link> (2024).</mixed-citation></ref><ref id="CR76"><label>76.</label><mixed-citation publication-type="other">Chen, F., Tsaftaris, S. A. &amp; Giuffrida, M. V. <italic toggle="yes">Arabidopsis thaliana</italic> data for a conversational multi-agent AI system for automated plant phenotyping [Data set]. 10.5281/zenodo.18940283 (2026).<pub-id pub-id-type="doi" assigning-authority="pmc">10.1038/s41467-026-71090-y</pub-id><pub-id pub-id-type="pmcid">PMC13376813</pub-id><pub-id pub-id-type="pmid">41932896</pub-id></mixed-citation></ref><ref id="CR77"><label>77.</label><mixed-citation publication-type="other">Williams, D., Macfarlane, F. &amp; Britten, A. Potato leaf data set [Data set]. 10.5281/zenodo.7938231 (2023).</mixed-citation></ref><ref id="CR78"><label>78.</label><mixed-citation publication-type="other">Chen, F. vios-s/PhenoAssistant: PhenoAssistant-nc-release (nc-release). 10.5281/zenodo.18334981 (2026).</mixed-citation></ref></ref-list></back></article>