
<!DOCTYPE article
  PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD with MathML3 v1.4 20241031//EN" "JATS-archivearticle1-4-mathml3.dtd">
<article article-type="research-article" xml:lang="en" dtd-version="1.4"><processing-meta base-tagset="archiving" mathml-version="3.0" table-model="xhtml" tagset-family="jats"><restricted-by>pmc</restricted-by></processing-meta><front><journal-meta><journal-id journal-id-type="nlm-ta">Nat Med</journal-id><journal-id journal-id-type="iso-abbrev">Nat Med</journal-id><journal-id journal-id-type="pmc-domain-id">981</journal-id><journal-id journal-id-type="pmc-domain">npgopen</journal-id><journal-id journal-id-type="nlm-id">9502015</journal-id><journal-title-group><journal-title>Nature Medicine</journal-title></journal-title-group><issn pub-type="ppub">1078-8956</issn><issn pub-type="epub">1546-170X</issn><?publisher_abbrev naturepg?><custom-meta-group><custom-meta><meta-name>pmc-is-collection-domain</meta-name><meta-value>yes</meta-value></custom-meta><custom-meta><meta-name>pmc-collection-title</meta-name><meta-value>Nature Portfolio</meta-value></custom-meta></custom-meta-group></journal-meta><article-meta><article-id pub-id-type="pmcid">PMC10719086</article-id><article-id pub-id-type="pmcid-ver">PMC10719086.1</article-id><article-id pub-id-type="pmcaid">10719086</article-id><article-id pub-id-type="pmcaiid">10719086</article-id><article-id pub-id-type="pmid">37973948</article-id><article-id pub-id-type="doi">10.1038/s41591-023-02625-9</article-id><article-id pub-id-type="publisher-id">2625</article-id><article-version article-version-type="pmc-version">1</article-version><article-categories><subj-group subj-group-type="heading"><subject>Article</subject></subj-group></article-categories><title-group><article-title>Prospective implementation of AI-assisted screen reading to improve early detection of breast cancer</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><contrib-id contrib-id-type="orcid" authenticated="false">http://orcid.org/0000-0002-0016-2275</contrib-id><name name-style="western"><surname>Ng</surname><given-names initials="AY">Annie Y.</given-names></name><address><email>annie@kheironmed.com</email></address><xref ref-type="aff" rid="Aff1">1</xref></contrib><contrib contrib-type="author"><contrib-id contrib-id-type="orcid" authenticated="false">http://orcid.org/0000-0003-0749-5117</contrib-id><name name-style="western"><surname>Oberije</surname><given-names initials="CJG">Cary J. G.</given-names></name><xref ref-type="aff" rid="Aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Ambrózay</surname><given-names initials="É">Éva</given-names></name><xref ref-type="aff" rid="Aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Szabó</surname><given-names initials="E">Endre</given-names></name><xref ref-type="aff" rid="Aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Serfőző</surname><given-names initials="O">Orsolya</given-names></name><xref ref-type="aff" rid="Aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Karpati</surname><given-names initials="E">Edit</given-names></name><xref ref-type="aff" rid="Aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Fox</surname><given-names initials="G">Georgia</given-names></name><xref ref-type="aff" rid="Aff1">1</xref></contrib><contrib contrib-type="author"><contrib-id contrib-id-type="orcid" authenticated="false">http://orcid.org/0000-0002-4897-9356</contrib-id><name name-style="western"><surname>Glocker</surname><given-names initials="B">Ben</given-names></name><xref ref-type="aff" rid="Aff1">1</xref><xref ref-type="aff" rid="Aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Morris</surname><given-names initials="EA">Elizabeth A.</given-names></name><xref ref-type="aff" rid="Aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Forrai</surname><given-names initials="G">Gábor</given-names></name><xref ref-type="aff" rid="Aff5">5</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Kecskemethy</surname><given-names initials="PD">Peter D.</given-names></name><xref ref-type="aff" rid="Aff1">1</xref></contrib><aff id="Aff1"><label>1</label><institution-wrap><institution-id institution-id-type="ROR">https://ror.org/01r3ct535</institution-id><institution-id institution-id-type="GRID">grid.500438.a</institution-id><institution>Kheiron Medical Technologies, </institution></institution-wrap>London, UK </aff><aff id="Aff2"><label>2</label>MaMMa Egészségügyi Zrt., Budapest, Hungary </aff><aff id="Aff3"><label>3</label><institution-wrap><institution-id institution-id-type="ROR">https://ror.org/041kmwe10</institution-id><institution-id institution-id-type="GRID">grid.7445.2</institution-id><institution-id institution-id-type="ISNI">0000 0001 2113 8111</institution-id><institution>Department of Computing, </institution><institution>Imperial College London, </institution></institution-wrap>London, UK </aff><aff id="Aff4"><label>4</label><institution-wrap><institution-id institution-id-type="ROR">https://ror.org/05t99sp05</institution-id><institution-id institution-id-type="GRID">grid.468726.9</institution-id><institution-id institution-id-type="ISNI">0000 0004 0486 2046</institution-id><institution>University of California, Davis, </institution></institution-wrap>Davis, CA USA </aff><aff id="Aff5"><label>5</label>Duna Medical Center, Budapest, Hungary </aff></contrib-group><pub-date pub-type="epub"><day>16</day><month>11</month><year>2023</year></pub-date><pub-date pub-type="ppub"><year>2023</year></pub-date><volume>29</volume><issue>12</issue><issue-id pub-id-type="pmc-issue-id">451012</issue-id><fpage>3044</fpage><lpage>3049</lpage><history><date date-type="received"><day>8</day><month>6</month><year>2022</year></date><date date-type="accepted"><day>4</day><month>10</month><year>2023</year></date></history><pub-history><event event-type="pmc-release"><date><day>16</day><month>11</month><year>2023</year></date></event><event event-type="pmc-live"><date><day>15</day><month>12</month><year>2023</year></date></event><event event-type="pmc-last-change"><date iso-8601-date="2026-02-07 21:25:16.233"><day>07</day><month>02</month><year>2026</year></date></event></pub-history><permissions><copyright-statement>© The Author(s) 2023</copyright-statement><license><ali:license_ref xmlns:ali="http://www.niso.org/schemas/ali/1.0/" specific-use="textmining" content-type="ccbylicense">https://creativecommons.org/licenses/by/4.0/</ali:license_ref><license-p><bold>Open Access</bold> This article is licensed under a Creative Commons Attribution 4.0 International License, which permits use, sharing, adaptation, distribution and reproduction in any medium or format, as long as you give appropriate credit to the original author(s) and the source, provide a link to the Creative Commons license, and indicate if changes were made. The images or other third party material in this article are included in the article’s Creative Commons license, unless indicated otherwise in a credit line to the material. If material is not included in the article’s Creative Commons license and your intended use is not permitted by statutory regulation or exceeds the permitted use, you will need to obtain permission directly from the copyright holder. To view a copy of this license, visit <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">http://creativecommons.org/licenses/by/4.0/</ext-link>.</license-p></license></permissions><self-uri xmlns:xlink="http://www.w3.org/1999/xlink" content-type="pmc-pdf" xlink:href="41591_2023_Article_2625.pdf"><?pdf-name 41591_2023_Article_2625.pdf?><?pdf-size 2942812?><?pdf-md5 894e70516c93a007b1b65214c6ac15d6?><?pdf-image-server-status NEVER_LOAD?><?pdf-cloudpmc-urn urn:app:80a2/10719086/894e70516c93/41591_2023_Article_2625.pdf?></self-uri><abstract id="Abs1"><p id="Par1">Artificial intelligence (AI) has the potential to improve breast cancer screening; however, prospective evidence of the safe implementation of AI into real clinical practice is limited. A commercially available AI system was implemented as an additional reader to standard double reading to flag cases for further arbitration review among screened women. Performance was assessed prospectively in three phases: a single-center pilot rollout, a wider multicenter pilot rollout and a full live rollout. The results showed that, compared to double reading, implementing the AI-assisted additional-reader process could achieve 0.7–1.6 additional cancer detection per 1,000 cases, with 0.16–0.30% additional recalls, 0–0.23% unnecessary recalls and a 0.1–1.9% increase in positive predictive value (PPV) after 7–11% additional human reads of AI-flagged cases (equating to 4–6% additional overall reading workload). The majority of cancerous cases detected by the AI-assisted additional-reader process were invasive (83.3%) and small-sized (≤10 mm, 47.0%). This evaluation suggests that using AI as an additional reader can improve the early detection of breast cancer with relevant prognostic features, with minimal to no unnecessary recalls. Although the AI-assisted additional-reader workflow requires additional reads, the higher PPV suggests that it can increase screening effectiveness.</p></abstract><abstract id="Abs2" abstract-type="web-summary"><p id="Par2">In a phased prospective rollout, the implementation of AI as an additional reader for mammography screening improved the real-world early detection of breast cancer compared to standard double reading involving two independent radiologists.</p></abstract><kwd-group kwd-group-type="npg-subject"><title>Subject terms</title><kwd>Breast cancer</kwd><kwd>Population screening</kwd><kwd>Radiography</kwd><kwd>Diagnosis</kwd></kwd-group><custom-meta-group><custom-meta><meta-name>pmc-status-qastatus</meta-name><meta-value>0</meta-value></custom-meta><custom-meta><meta-name>pmc-status-live</meta-name><meta-value>yes</meta-value></custom-meta><custom-meta><meta-name>pmc-status-embargo</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-status-released</meta-name><meta-value>yes</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-open-access</meta-name><meta-value>yes</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-olf</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-manuscript</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-legally-suppressed</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-has-pdf</meta-name><meta-value>yes</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-has-supplement</meta-name><meta-value>yes</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-pdf-only</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-suppress-copyright</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-is-real-version</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-is-scanned-article</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-preprint</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-in-epmc</meta-name><meta-value>yes</meta-value></custom-meta><custom-meta><meta-name>pmc-license-ref</meta-name><meta-value>CC BY</meta-value></custom-meta><custom-meta><meta-name>issue-copyright-statement</meta-name><meta-value>© Springer Nature America, Inc. 2023</meta-value></custom-meta></custom-meta-group></article-meta></front><body><sec id="Sec1"><title>Main</title><p id="Par3">Breast cancer screening detects cancer at earlier stages<sup><xref ref-type="bibr" rid="CR1">1</xref></sup>, leading to a meaningful reduction in breast cancer mortality<sup><xref ref-type="bibr" rid="CR2">2</xref></sup>. Moreover, early detection can lead to less aggressive treatments, reducing treatment toxicity. Although breast screening reduces overall mortality, it has limitations that result in failure to detect cancer in a considerable number of screened individuals. In these cases, cancer may be found later between screening rounds (interval cancer)<sup><xref ref-type="bibr" rid="CR3">3</xref></sup> or at the next screening round<sup><xref ref-type="bibr" rid="CR4">4</xref></sup>. Reported estimates for the rate of interval cancer detection vary widely between countries and screening programs with varying screening intervals, ranging from 0.7 to 4.9 per 1,000 screened women<sup><xref ref-type="bibr" rid="CR3">3</xref></sup>. Among them, the proportion of cancer cases that could be detected retrospectively at previous rounds is estimated to be 22%<sup><xref ref-type="bibr" rid="CR4">4</xref></sup>. In the past, computer-aided detection (CAD) systems were developed to improve cancer detection. However, the benefits of CAD found in experimental studies did not translate into real-world clinical benefits. The use of CAD resulted in increased recalls, more time needed to assess screens and more biopsies without improving cancer detection, ultimately conferring no screening benefit<sup><xref ref-type="bibr" rid="CR5">5</xref></sup>.</p><p id="Par4">Modern artificial intelligence (AI) based on deep learning is a different technology from past CAD systems and has demonstrated higher potential in supporting the quality of screening services and reducing workload, depending on its workflow integration<sup><xref ref-type="bibr" rid="CR6">6</xref>–<xref ref-type="bibr" rid="CR10">10</xref></sup>. AI has the highest performance risk for cases with less common characteristics; thus, it requires assessment in large-scale studies. As retrospective studies make large-scale evaluations possible, they are crucial to validate the safety and effectiveness of AI before prospective use. However, retrospective results can be expected to translate to real clinical practice only when appropriate study methods are used to ensure that the analyzed data are representative of what AI would process in real-world deployments. Otherwise, the usefulness of AI in clinical practice is not guaranteed<sup><xref ref-type="bibr" rid="CR4">4</xref>,<xref ref-type="bibr" rid="CR11">11</xref>,<xref ref-type="bibr" rid="CR12">12</xref></sup>. Prospective evaluations are needed to assess the real-world performance of AI integrated into live clinical workflows; however, these have been limited to date<sup><xref ref-type="bibr" rid="CR13">13</xref></sup>.</p><p id="Par5">This service evaluation presents results from using a commercially available AI system, Mia (Kheiron Medical Technologies), configured with regulatory-cleared predetermined sensitivity and specificity operating points in pilot implementations and live use in daily practice. The performance and generalizability of the AI system used were previously confirmed in a large-scale retrospective AI generalizability study<sup><xref ref-type="bibr" rid="CR8">8</xref>,<xref ref-type="bibr" rid="CR9">9</xref>,<xref ref-type="bibr" rid="CR14">14</xref></sup>. The current analysis used prospectively collected postmarket real-world data to assess the effectiveness of the AI system as an additional component to standard screening procedures and a quality-control safety net in the AI-assisted additional-reader workflow to support early cancer detection.</p></sec><sec id="Sec2" sec-type="results"><title>Results</title><p id="Par6">A three-phase approach was used to implement the AI system in an AI-assisted additional-reader workflow at four sites of MaMMa Egészségügyi Zrt. (MaMMa Klinika), a breast cancer screening institution that serves urban and rural populations in Hungary. The institution implements a 2-year screening interval and invites women aged 45–65 years to undergo screening. All institution sites also offer opportunistic screening, in which women who are not invited to screening but choose to participate are screened. These women undergo the same procedure as those participating in the population screening program. At the institution sites, full-field digital mammography images were obtained using the IMS Giotto Image 3DL and IMS Giotto Class systems, following the standard operating procedures at the four sites. All sites follow the standard double-reading workflow (with strictly no AI involvement) in which two radiologists review every case. When discordance arises, an arbitrator makes the decision to either recall or not recall a woman for further assessment. In the implemented AI-assisted additional-reader workflow, the AI system flagged cases for additional review among those classified by double reading as ‘no recall’. These positive discordant cases (that is, cases that AI flagged as ‘positive’ and human readers marked as ‘negative’) were additionally reviewed by a human arbitrator (additional arbitrator) to possibly recall additional cases and detect more cancerous cases at an early stage (Fig. <xref rid="Fig1" ref-type="fig">1</xref>). The additional arbitrator was provided with images containing AI-generated regions of interest highlighting areas suggestive of malignancy for their review.<fig id="Fig1" position="float" orientation="portrait"><label>Fig. 1</label><caption><title>AI as an additional reader.</title><p>The AI-assisted additional-reader workflow uses a standard double-reading process complemented by image assessment by AI. If double reading results in a ‘no recall’ decision but the AI system flags the case, the screen is assessed by an additional human arbitrator.</p></caption><graphic xmlns:xlink="http://www.w3.org/1999/xlink" id="d32e328" position="float" orientation="portrait" xlink:href="41591_2023_2625_Fig1_HTML.jpg"><?image-name 41591_2023_2625_Fig1_HTML.jpg?><?image-size 90795?><?image-md5 bfb28fd807546cbc53bed3e84ca7f68b?><?image-image-server-status LOAD_COMPLETED?><?image-original-height 1466?><?image-original-width 1057?><?image-scaled-height 976?><?image-scaled-width 704?><?image-cloudpmc-urn urn:cdn:blobs/80a2/10719086/bfb28fd80754/41591_2023_2625_Fig1_HTML.jpg?><?thumb-name 41591_2023_2625_Fig1_HTML.gif?><?thumb-size 4777?><?thumb-md5 eb88298a1289de7d9dc5e24bdef0d7ec?><?thumb-image-server-status NEVER_LOAD?><?thumb-scaled-height 139?><?thumb-scaled-width 100?><?thumb-cloudpmc-urn urn:cdn:blobs/80a2/10719086/eb88298a1289/41591_2023_2625_Fig1_HTML.gif?></graphic></fig></p><p id="Par7">The implementation of the AI system consisted of three phases to ensure the safe deployment of the AI-assisted additional-reader process into live use. The first phase aimed to demonstrate the clinical benefit of the AI-assisted additional-reader process in a limited pilot rollout in which only one senior radiologist reviewed the AI-flagged cases from a single site, with the original screening date between April 6 and September 28, 2021 inclusive. The second phase was launched as an extended multicenter pilot involving a wider rollout of the AI-assisted additional-reader process across four sites (including the initial pilot site) and three additional arbitrators (including the additional arbitrator from the first phase). In the second phase, the readers independently reviewed every case flagged by AI from April 6 through December 21, 2021, at the initial pilot site and from April 6 through June 30, 2021, at each of the other three sites. One of the additional arbitrators made the final decision on which cases to recall additionally based on the opinions of all three readers. The extended pilot also aimed to provide a training period for the three additional arbitrators before live use began.</p><p id="Par8">Finally, the third phase involved a full live rollout of the AI system as an official addition to the standard of care across the four sites from July 4, 2022. In this phase, the three additional arbitrators independently made recall decisions. The live rollout is ongoing, and the results presented here cover cases through January 31, 2023. Results were also simulated with a predetermined higher-specificity operating point to inform the sites on how the AI-assisted additional-reader process may be further optimized to suit their needs. The summary details of the dataset periods are provided in Table <xref rid="Tab1" ref-type="table">1</xref>. In live use, each AI-flagged case was independently reviewed by one of the three additional arbitrators who made the final recall decision on each case they reviewed. During the two pilot phases, additional recalls based on additional arbitration reviews were done after the screening participants had been informed of the double-reading decision. In the third phase involving implementation into daily practice, the screening participants were informed after the decision was finalized based on the additional arbitration reviews. All readers had specialist training and ≥14 years of screening mammography experience, with non-additional arbitrators reading approximately 12,000 screens per year and additional arbitrators reading 25,000 screens per year on average.<table-wrap id="Tab1" position="float" orientation="portrait"><label>Table 1</label><caption><p>Overview of screens per phase per site</p></caption><table frame="hsides" rules="groups"><thead><tr><th colspan="1" rowspan="1">Site</th><th colspan="1" rowspan="1">First month</th><th colspan="1" rowspan="1">Final month</th><th colspan="1" rowspan="1">Vendor</th><th colspan="1" rowspan="1">Equipment model</th><th colspan="1" rowspan="1">No. of available double-read screens</th><th colspan="1" rowspan="1">No. of processed screens</th><th colspan="1" rowspan="1">Percentage</th></tr></thead><tbody><tr><td colspan="8" rowspan="1">Phase 1, initial pilot (1 site, 1 additional arbitrator, additional arbitration cases were single read)</td></tr><tr><td colspan="1" rowspan="1"> Site 1</td><td colspan="1" rowspan="1">April 2021</td><td colspan="1" rowspan="1">September 2021</td><td colspan="1" rowspan="1">IMS</td><td colspan="1" rowspan="1">Giotto Class</td><td colspan="1" rowspan="1">3,817</td><td colspan="1" rowspan="1">3,746</td><td colspan="1" rowspan="1">98.1%</td></tr><tr><td colspan="8" rowspan="1">Phase 2, extended pilot (4 sites, 3 additional arbitrators, all additional arbitration cases were read by each additional arbitrator)</td></tr><tr><td colspan="1" rowspan="1"> Site 1</td><td colspan="1" rowspan="1">April 2021</td><td colspan="1" rowspan="1">December 2021</td><td colspan="1" rowspan="1">IMS</td><td colspan="1" rowspan="1">Giotto Class</td><td colspan="1" rowspan="1">5,859</td><td colspan="1" rowspan="1">5,758</td><td colspan="1" rowspan="1">98.3%</td></tr><tr><td colspan="1" rowspan="1"> Site 2</td><td colspan="1" rowspan="1">April 2021</td><td colspan="1" rowspan="1">June 2021</td><td colspan="1" rowspan="1">IMS</td><td colspan="1" rowspan="1">Giotto Class</td><td colspan="1" rowspan="1">1,187</td><td colspan="1" rowspan="1">1,172</td><td colspan="1" rowspan="1">98.7%</td></tr><tr><td colspan="1" rowspan="1"> Site 3</td><td colspan="1" rowspan="1">April 2021</td><td colspan="1" rowspan="1">June 2021</td><td colspan="1" rowspan="1">IMS</td><td colspan="1" rowspan="1">Giotto Image 3DL</td><td colspan="1" rowspan="1">918</td><td colspan="1" rowspan="1">911</td><td colspan="1" rowspan="1">99.2%</td></tr><tr><td colspan="1" rowspan="1"> Site 4</td><td colspan="1" rowspan="1">April 2021</td><td colspan="1" rowspan="1">June 2021</td><td colspan="1" rowspan="1">IMS</td><td colspan="1" rowspan="1">Giotto Image 3DL</td><td colspan="1" rowspan="1">1,302</td><td colspan="1" rowspan="1">1,271</td><td colspan="1" rowspan="1">97.6%</td></tr><tr><td colspan="1" rowspan="1">Total</td><td colspan="1" rowspan="1"/><td colspan="1" rowspan="1"/><td colspan="1" rowspan="1"/><td colspan="1" rowspan="1"/><td colspan="1" rowspan="1">9,266</td><td colspan="1" rowspan="1">9,112</td><td colspan="1" rowspan="1">98.3%</td></tr><tr><td colspan="8" rowspan="1">Phase 3, live use in standard clinical practice (4 sites, 3 additional arbitrators, additional arbitration cases were single read)</td></tr><tr><td colspan="1" rowspan="1"> Site 1</td><td colspan="1" rowspan="1">July 2022</td><td colspan="1" rowspan="1">January 2023</td><td colspan="1" rowspan="1">IMS</td><td colspan="1" rowspan="1">Giotto Class</td><td colspan="1" rowspan="1">4,818</td><td colspan="1" rowspan="1">4,711</td><td colspan="1" rowspan="1">97.8%</td></tr><tr><td colspan="1" rowspan="1"> Site 2</td><td colspan="1" rowspan="1">July 2022</td><td colspan="1" rowspan="1">January 2023</td><td colspan="1" rowspan="1">IMS</td><td colspan="1" rowspan="1">Giotto Class</td><td colspan="1" rowspan="1">4,605</td><td colspan="1" rowspan="1">4,537</td><td colspan="1" rowspan="1">98.5%</td></tr><tr><td colspan="1" rowspan="1"> Site 3</td><td colspan="1" rowspan="1">July 2022</td><td colspan="1" rowspan="1">January 2023</td><td colspan="1" rowspan="1">IMS</td><td colspan="1" rowspan="1">Giotto Image 3DL</td><td colspan="1" rowspan="1">2,925</td><td colspan="1" rowspan="1">2,903</td><td colspan="1" rowspan="1">99.2%</td></tr><tr><td colspan="1" rowspan="1"> Site 4</td><td colspan="1" rowspan="1">July 2022</td><td colspan="1" rowspan="1">January 2023</td><td colspan="1" rowspan="1">IMS</td><td colspan="1" rowspan="1">Giotto Image 3DL</td><td colspan="1" rowspan="1">3,908</td><td colspan="1" rowspan="1">3,802</td><td colspan="1" rowspan="1">97.3%</td></tr><tr><td colspan="1" rowspan="1">Total</td><td colspan="1" rowspan="1"/><td colspan="1" rowspan="1"/><td colspan="1" rowspan="1"/><td colspan="1" rowspan="1"/><td colspan="1" rowspan="1">16,256</td><td colspan="1" rowspan="1">15,953</td><td colspan="1" rowspan="1">98.1%</td></tr></tbody></table></table-wrap></p><sec id="Sec3"><title>Patient characteristics</title><p id="Par9">Table <xref rid="Tab2" ref-type="table">2</xref> shows the characteristics of participants in each phase. The initial pilot included 3,746 women with an average age of 58.2 (s.d. 11.0) years. Among them, 126 (3.4%) reported a family history of cancer and 479 (12.7%) had a Tabár parenchymal pattern classification of 4 or 5, correlating with high density. In the extended pilot (<italic toggle="yes">n</italic> = 9,112), the mean age was also 58.2 (s.d. 10.7) years. Tabár classification 4 or 5 was identified in 1,094 women (12.0%), and 274 women (3.0%) reported a family history of cancer. Finally, in the live-use phase, 15,953 women were included. The mean age was 58.6 (s.d. 10.5) years, with 615 women (3.9%) having reported a family history of cancer and 1,733 women (10.8%) having a Tabár classification of 4 or 5.<table-wrap id="Tab2" position="float" orientation="portrait"><label>Table 2</label><caption><p>Participant characteristics per phase</p></caption><table frame="hsides" rules="groups"><thead><tr><th colspan="1" rowspan="1">Variable</th><th colspan="1" rowspan="1">Initial pilot (<italic toggle="yes">n</italic> = 3,746)</th><th colspan="1" rowspan="1">Extended pilot (<italic toggle="yes">n</italic> = 9,112)</th><th colspan="1" rowspan="1">Live use (<italic toggle="yes">n</italic> = 15,953)</th></tr></thead><tbody><tr><td colspan="1" rowspan="1">Age (continuous, years), mean (s.d.)</td><td colspan="1" rowspan="1">58.2 (11.0)</td><td colspan="1" rowspan="1">58.2 (10.7)</td><td colspan="1" rowspan="1">58.6 (10.5)</td></tr><tr><td colspan="1" rowspan="1">Age group, <italic toggle="yes">n</italic> (%)</td><td colspan="1" rowspan="1"/><td colspan="1" rowspan="1"/><td colspan="1" rowspan="1"/></tr><tr><td colspan="1" rowspan="1">≤35 years</td><td colspan="1" rowspan="1">0 (0.0%)</td><td colspan="1" rowspan="1">0 (0.0%)</td><td colspan="1" rowspan="1">0 (0.0%)</td></tr><tr><td colspan="1" rowspan="1">36–45 years</td><td colspan="1" rowspan="1">518 (13.8%)</td><td colspan="1" rowspan="1">1,149 (12.6%)</td><td colspan="1" rowspan="1">1,583 (9.9%)</td></tr><tr><td colspan="1" rowspan="1">46–55 years</td><td colspan="1" rowspan="1">1,218 (32.5%)</td><td colspan="1" rowspan="1">2,998 (32.9%)</td><td colspan="1" rowspan="1">5,420 (34.0%)</td></tr><tr><td colspan="1" rowspan="1">56–65 years</td><td colspan="1" rowspan="1">940 (25.1%)</td><td colspan="1" rowspan="1">2,493 (27.4%)</td><td colspan="1" rowspan="1">4,699 (29.5%)</td></tr><tr><td colspan="1" rowspan="1">66–75 years</td><td colspan="1" rowspan="1">806 (21.5%)</td><td colspan="1" rowspan="1">1,902 (20.9%)</td><td colspan="1" rowspan="1">3,196 (20.0%)</td></tr><tr><td colspan="1" rowspan="1">&gt;75 years</td><td colspan="1" rowspan="1">264 (7.0%)</td><td colspan="1" rowspan="1">570 (6.3%)</td><td colspan="1" rowspan="1">1,055 (6.6%)</td></tr><tr><td colspan="1" rowspan="1">Family history<sup>a</sup>, <italic toggle="yes">n</italic> (%)</td><td colspan="1" rowspan="1"/><td colspan="1" rowspan="1"/><td colspan="1" rowspan="1"/></tr><tr><td colspan="1" rowspan="1">No</td><td colspan="1" rowspan="1">3,620 (96.6%)</td><td colspan="1" rowspan="1">8,838 (97.0%)</td><td colspan="1" rowspan="1">15,338 (96.1%)</td></tr><tr><td colspan="1" rowspan="1">Yes</td><td colspan="1" rowspan="1">126 (3.4%)</td><td colspan="1" rowspan="1">274 (3.0%)</td><td colspan="1" rowspan="1">615 (3.9%)</td></tr><tr><td colspan="1" rowspan="1">Tábar classification of parenchymal patterns<sup>b</sup>, <italic toggle="yes">n</italic> (%)</td><td colspan="1" rowspan="1"/><td colspan="1" rowspan="1"/><td colspan="1" rowspan="1"/></tr><tr><td colspan="1" rowspan="1">1</td><td colspan="1" rowspan="1">1,506 (40.2%)</td><td colspan="1" rowspan="1">3,950 (43.3%)</td><td colspan="1" rowspan="1">7,468 (46.8%)</td></tr><tr><td colspan="1" rowspan="1">2</td><td colspan="1" rowspan="1">729 (19.5%)</td><td colspan="1" rowspan="1">1,697 (18.6%)</td><td colspan="1" rowspan="1">2,921 (18.3%)</td></tr><tr><td colspan="1" rowspan="1">3</td><td colspan="1" rowspan="1">336 (9.0%)</td><td colspan="1" rowspan="1">679 (7.5%)</td><td colspan="1" rowspan="1">465 (2.9%)</td></tr><tr><td colspan="1" rowspan="1">4</td><td colspan="1" rowspan="1">365 (9.7%)</td><td colspan="1" rowspan="1">848 (9.3%)</td><td colspan="1" rowspan="1">1,423 (8.9%)</td></tr><tr><td colspan="1" rowspan="1">5</td><td colspan="1" rowspan="1">114 (3.0%)</td><td colspan="1" rowspan="1">246 (2.7%)</td><td colspan="1" rowspan="1">310 (1.9%)</td></tr><tr><td colspan="1" rowspan="1">Missing</td><td colspan="1" rowspan="1">696 (18.6%)</td><td colspan="1" rowspan="1">1,692 (18.6%)</td><td colspan="1" rowspan="1">3,366 (21.1%)</td></tr></tbody></table><table-wrap-foot><p><sup>a</sup>Family history of cancer = ‘yes’ if at least two first-degree female family members have been diagnosed with breast cancer.</p><p><sup>b</sup>A Tabár classification<sup><xref ref-type="bibr" rid="CR17">17</xref></sup> of 4 or 5 correlating with high density (BI-RADS (breast imaging and reporting data system) breast density class C or D).</p></table-wrap-foot></table-wrap></p></sec><sec id="Sec4"><title>Screening performance of the AI-assisted additional-reader workflow</title><p id="Par10">Across the three phases, the implementation of the AI-assisted additional-reader workflow resulted in 24 more cancer cases detected (7% relative increase in cancer detection rate (CDR)) and 70 more women recalled (0.28% increase in absolute recall rate), at a positive predictive value (PPV) for screening of 20.0% (3% relative increase) (Table <xref rid="Tab3" ref-type="table">3</xref>). The initial pilot, extended pilot and live-use assessments included 3,746 of 3,817 (98.1%), 9,112 of 9,266 (98.3%) and 15,953 of 16,256 (98.1%) double-read cases that the AI could process, respectively (Table <xref rid="Tab1" ref-type="table">1</xref>). Table <xref rid="Tab3" ref-type="table">3</xref> shows the outcome metrics for each phase and reports the results of the McNemar test for sensitivity and CDR. In summary, standard double reading resulted in recall rates of 6.7% (initial pilot), 7.0% (extended pilot) and 7.7% (live use) and CDRs of 12.8 per 1,000 cases (initial pilot), 13.8 per 1,000 cases (extended pilot) and 14.9 per 1,000 cases (live use). For the initial and extended pilots, AI flagged for review 10.6% (396/3,746) and 11.2% (1,024/9,112) of cases, respectively. Before launching the AI system into live use, its decision threshold was adjusted to a more specific predetermined operating point to accommodate the site’s workload capacity, resulting in a smaller proportion of cases (7.4%, 1,186/15,953) flagged for additional review in live use. The additional arbitration reviews resulted in six (initial pilot), 22 (extended pilot) and 48 (live use) additional recalled cases, increasing the recall rate by 0.16% (initial pilot), 0.23% (extended pilot) and 0.25% (live use), respectively. From the additional recalls, six (initial pilot), 13 (extended pilot) and 11 (live use) additional cancer cases were found, increasing the CDR by 1.6 per 1,000 cases (a 13% relative increase), 1.4 per 1,000 cases (a 10% relative increase) and 0.7 per 1,000 cases (a 5% relative increase) for the initial pilot, extended pilot and live-use phases, respectively (all statistically significant with <italic toggle="yes">P</italic> &lt; 0.05) (Table <xref rid="Tab3" ref-type="table">3</xref>). Of the additional cancer cases, four (66.7%) in the initial pilot, ten (76.9%) in the extended pilot and five (45.5%) in the live-use phase were confirmed to be invasive. In addition, one case (16.7%) in the initial pilot, one case (7.7%) in the extended pilot and two cases (18.2%) in live use were in situ cancer. Meanwhile, one case (16.7%) in the initial pilot, two cases (15.4%) in the extended pilot and four cases (36.4%) in live use had missing invasiveness information. Of the additional cancer cases found with available data on either pathological or radiological tumor size, 50.0% (two of four) in the initial pilot, 40% (four of ten) in the extended pilot and 57.1% (four of seven) in live use were ≤10 mm. Overall, the screening performance of double reading plus the AI-assisted additional-reader workflow resulted in recall rates of 6.8% (initial pilot), 7.3% (extended pilot) and 8.0% (live use); arbitration rates of 13.6% (initial pilot), 14.2% (extended pilot) and 10.8% (live use); and CDRs of 14.4 per 1,000 cases (initial pilot), 15.3 per 1,000 cases (extended pilot) and 15.6 per 1,000 cases (live use).<table-wrap id="Tab3" position="float" orientation="portrait"><label>Table 3</label><caption><p>Outcome metrics for standard double reading versus double reading plus the AI-assisted additional-reader workflow</p></caption><table frame="hsides" rules="groups"><thead><tr><th rowspan="2" colspan="1">Variable</th><th colspan="2" rowspan="1">Double reading</th><th colspan="2" rowspan="1">Double reading plus the AI-assisted additional-reader workflow</th><th rowspan="2" colspan="1">Difference</th></tr><tr><th colspan="1" rowspan="1">Num/Denom</th><th colspan="1" rowspan="1">Value (95% CI)</th><th colspan="1" rowspan="1">Num/Denom</th><th colspan="1" rowspan="1">Value (95% CI)</th></tr></thead><tbody><tr><td colspan="6" rowspan="1">Results of phase 1, pilot rollout (1 site, 1 additional arbitrator, additional arbitration cases were single read), <italic toggle="yes">n</italic> = 3,746 screens</td></tr><tr><td colspan="1" rowspan="1"> CDR (per 1,000 cases)</td><td colspan="1" rowspan="1">48/3,746</td><td colspan="1" rowspan="1">12.8 (9.7–16.9)</td><td colspan="1" rowspan="1">54/3,746</td><td colspan="1" rowspan="1">14.4 (11.1–18.8)</td><td colspan="1" rowspan="1">1.6<sup>a</sup></td></tr><tr><td colspan="1" rowspan="1"> RR (%)</td><td colspan="1" rowspan="1">250/3,746</td><td colspan="1" rowspan="1">6.7 (5.9–7.5)</td><td colspan="1" rowspan="1">256/3,746</td><td colspan="1" rowspan="1">6.8 (6.1–7.7)</td><td colspan="1" rowspan="1">0.2</td></tr><tr><td colspan="1" rowspan="1"> Sen (%)</td><td colspan="1" rowspan="1">48/58</td><td colspan="1" rowspan="1">82.8 (71.7–90.4)</td><td colspan="1" rowspan="1">54/58</td><td colspan="1" rowspan="1">93.1 (83.6–97.3)</td><td colspan="1" rowspan="1">10.3<sup>a</sup></td></tr><tr><td colspan="1" rowspan="1"> Spec (%)</td><td colspan="1" rowspan="1">3,486/3,688</td><td colspan="1" rowspan="1">94.5 (93.7–95.2)</td><td colspan="1" rowspan="1">3,486/3,688</td><td colspan="1" rowspan="1">94.5 (93.7–95.2)</td><td colspan="1" rowspan="1">0.0</td></tr><tr><td colspan="1" rowspan="1"> PPV (%)</td><td colspan="1" rowspan="1">48/250</td><td colspan="1" rowspan="1">19.2 (14.8–24.5)</td><td colspan="1" rowspan="1">54/256</td><td colspan="1" rowspan="1">21.1 (16.5–26.5)</td><td colspan="1" rowspan="1">1.9</td></tr><tr><td colspan="1" rowspan="1"> Arbitration rate (%)</td><td colspan="1" rowspan="1">114/3,746</td><td colspan="1" rowspan="1">3.0 (2.5–3.6)</td><td colspan="1" rowspan="1">510/3,746</td><td colspan="1" rowspan="1">13.6 (12.6–14.8)</td><td colspan="1" rowspan="1">10.6</td></tr><tr><td colspan="1" rowspan="1"> Positive discordance rate (%)</td><td colspan="1" rowspan="1">–</td><td colspan="1" rowspan="1">–</td><td colspan="1" rowspan="1">396/3,746</td><td colspan="1" rowspan="1">10.6 (9.6–11.6)</td><td colspan="1" rowspan="1">–</td></tr><tr><td colspan="1" rowspan="1"> RR of additional arbitration (%)</td><td colspan="1" rowspan="1">–</td><td colspan="1" rowspan="1">–</td><td colspan="1" rowspan="1">6/396</td><td colspan="1" rowspan="1">1.5 (0.7–3.3)</td><td colspan="1" rowspan="1">–</td></tr><tr><td colspan="1" rowspan="1"> PPV of additional arbitration (%)</td><td colspan="1" rowspan="1">–</td><td colspan="1" rowspan="1">–</td><td colspan="1" rowspan="1">6/6</td><td colspan="1" rowspan="1">100 (61.0–100)</td><td colspan="1" rowspan="1">–</td></tr><tr><td colspan="6" rowspan="1">Results of phase 2, extended pilot (4 sites, 3 additional arbitrators, all additional arbitration cases were read by each additional reader), <italic toggle="yes">n</italic> = 9,112 screens</td></tr><tr><td colspan="1" rowspan="1"> CDR (per 1,000 cases)</td><td colspan="1" rowspan="1">126/9,112</td><td colspan="1" rowspan="1">13.8 (11.6–16.4)</td><td colspan="1" rowspan="1">139/9,112</td><td colspan="1" rowspan="1">15.3 (12.9–18.0)</td><td colspan="1" rowspan="1">1.4<sup>a</sup></td></tr><tr><td colspan="1" rowspan="1"> RR (%)</td><td colspan="1" rowspan="1">639/9,112</td><td colspan="1" rowspan="1">7.0 (6.5–7.6)</td><td colspan="1" rowspan="1">661/9,112</td><td colspan="1" rowspan="1">7.3 (6.7–7.8)</td><td colspan="1" rowspan="1">0.2</td></tr><tr><td colspan="1" rowspan="1"> Sen (%)</td><td colspan="1" rowspan="1">126/145</td><td colspan="1" rowspan="1">86.9 (80.4–91.4)</td><td colspan="1" rowspan="1">139/145</td><td colspan="1" rowspan="1">95.9 (91.3–98.1)</td><td colspan="1" rowspan="1">9.0<sup>a</sup></td></tr><tr><td colspan="1" rowspan="1"> Spec (%)</td><td colspan="1" rowspan="1">8,454/8,967</td><td colspan="1" rowspan="1">94.3 (93.8–94.7)</td><td colspan="1" rowspan="1">8,445/8,967</td><td colspan="1" rowspan="1">94.2 (93.7–94.6)</td><td colspan="1" rowspan="1">−0.1</td></tr><tr><td colspan="1" rowspan="1"> PPV (%)</td><td colspan="1" rowspan="1">126/639</td><td colspan="1" rowspan="1">19.7 (16.8–23.0)</td><td colspan="1" rowspan="1">139/661</td><td colspan="1" rowspan="1">21.0 (18.1–24.3)</td><td colspan="1" rowspan="1">1.3</td></tr><tr><td colspan="1" rowspan="1"> Arbitration rate (%)</td><td colspan="1" rowspan="1">270/9,112</td><td colspan="1" rowspan="1">3.0 (2.6–3.3)</td><td colspan="1" rowspan="1">1,294/9,112</td><td colspan="1" rowspan="1">14.2 (13.5–14.9)</td><td colspan="1" rowspan="1">11.2</td></tr><tr><td colspan="1" rowspan="1"> Positive discordance rate (%)</td><td colspan="1" rowspan="1">–</td><td colspan="1" rowspan="1">–</td><td colspan="1" rowspan="1">1,024/9,112</td><td colspan="1" rowspan="1">11.2 (10.6–11.9)</td><td colspan="1" rowspan="1">–</td></tr><tr><td colspan="1" rowspan="1"> RR of additional arbitration (%)</td><td colspan="1" rowspan="1">–</td><td colspan="1" rowspan="1">–</td><td colspan="1" rowspan="1">22/1,024</td><td colspan="1" rowspan="1">2.1 (1.4–3.2)</td><td colspan="1" rowspan="1">–</td></tr><tr><td colspan="1" rowspan="1"> PPV of additional arbitration (%)</td><td colspan="1" rowspan="1">–</td><td colspan="1" rowspan="1">–</td><td colspan="1" rowspan="1">13/22</td><td colspan="1" rowspan="1">59.1 (38.7–76.7)</td><td colspan="1" rowspan="1">–</td></tr><tr><td colspan="6" rowspan="1">Results of phase 3, live use in standard clinical practice (4 sites, 3 additional arbitrators, additional arbitration cases were single read), <italic toggle="yes">n</italic> = 15,953 screens</td></tr><tr><td colspan="1" rowspan="1"> CDR (per 1,000 cases)</td><td colspan="1" rowspan="1">238/15,953</td><td colspan="1" rowspan="1">14.9 (13.2–16.9)</td><td colspan="1" rowspan="1">249/15,953</td><td colspan="1" rowspan="1">15.6 (13.8–17.7)</td><td colspan="1" rowspan="1">0.7<sup>a</sup></td></tr><tr><td colspan="1" rowspan="1"> RR (%)</td><td colspan="1" rowspan="1">1,228/15,953</td><td colspan="1" rowspan="1">7.7 (7.3–8.1)</td><td colspan="1" rowspan="1">1,276/15,953</td><td colspan="1" rowspan="1">8.0 (7.6–8.4)</td><td colspan="1" rowspan="1">0.3</td></tr><tr><td colspan="1" rowspan="1"> Sen (%)</td><td colspan="1" rowspan="1">238/253</td><td colspan="1" rowspan="1">94.1 (90.4–96.4)</td><td colspan="1" rowspan="1">249/253</td><td colspan="1" rowspan="1">98.4 (96.0–99.4)</td><td colspan="1" rowspan="1">4.3<sup>a</sup></td></tr><tr><td colspan="1" rowspan="1"> Spec (%)</td><td colspan="1" rowspan="1">14,710/15,700</td><td colspan="1" rowspan="1">93.7 (93.3–94.1)</td><td colspan="1" rowspan="1">14,673/15,700</td><td colspan="1" rowspan="1">93.5 (93.1–93.8)</td><td colspan="1" rowspan="1">−0.2</td></tr><tr><td colspan="1" rowspan="1"> PPV (%)</td><td colspan="1" rowspan="1">238/1,228</td><td colspan="1" rowspan="1">19.4 (17.3–21.7)</td><td colspan="1" rowspan="1">249/1,276</td><td colspan="1" rowspan="1">19.5 (17.4–21.8)</td><td colspan="1" rowspan="1">0.1</td></tr><tr><td colspan="1" rowspan="1"> Arbitration rate (%)</td><td colspan="1" rowspan="1">529/15,953</td><td colspan="1" rowspan="1">3.3 (3.0–3.6)</td><td colspan="1" rowspan="1">1,715/15,953</td><td colspan="1" rowspan="1">10.8 (10.3–11.2)</td><td colspan="1" rowspan="1">7.4</td></tr><tr><td colspan="1" rowspan="1"> Positive discordance rate (%)</td><td colspan="1" rowspan="1">–</td><td colspan="1" rowspan="1">–</td><td colspan="1" rowspan="1">1,186/15,953</td><td colspan="1" rowspan="1">7.4 (7.0–7.9)</td><td colspan="1" rowspan="1">–</td></tr><tr><td colspan="1" rowspan="1"> RR of additional arbitration (%)</td><td colspan="1" rowspan="1">–</td><td colspan="1" rowspan="1">–</td><td colspan="1" rowspan="1">48/1,186</td><td colspan="1" rowspan="1">4.0 (3.1–5.3)</td><td colspan="1" rowspan="1">–</td></tr><tr><td colspan="1" rowspan="1"> PPV of additional arbitration (%)</td><td colspan="1" rowspan="1">–</td><td colspan="1" rowspan="1">–</td><td colspan="1" rowspan="1">11/48</td><td colspan="1" rowspan="1">22.9 (13.3–36.5)</td><td colspan="1" rowspan="1">–</td></tr></tbody></table><table-wrap-foot><p>Num, numerator; Denom, denominator; CI, confidence interval; Sen, sensitivity; Spec, specificity; RR, recall rate; see metric definitions in <xref rid="Sec7" ref-type="sec">Methods</xref>.</p><p><sup>a</sup>The two-sided McNemar test to assess CDR and Sen differences between double reading and double reading plus the AI-assisted additional-reader workflow resulted in <italic toggle="yes">P</italic> values of 0.0031, 0.0002 and 0.001 for phases 1, 2 and 3, respectively. The McNemar test is based on the binomial distribution. Continuity correction was applied.</p></table-wrap-foot></table-wrap></p></sec><sec id="Sec5"><title>Performance at a simulated higher-specificity operating point</title><p id="Par11">When the performance of the AI system was evaluated at a predetermined higher-specificity operating point through simulations, the AI-assisted additional-reader workflow substantially reduced the proportion of cases requiring additional review to 2.4% (89/3,746), 3.0% (274/9,112) and 2.9% (457/15,953) for the initial pilot, extended pilot and live-use phases, respectively, while still detecting 5 of the 6 (1.3/1,000, a 10% relative increase) additional cancer cases found in the initial pilot, 11 of the 13 (1.2/1,000, a 9% relative increase) additional cancer cases found in the extended pilot and 10 of the 11 (0.6/1,000, a 4% relative increase) additional cancer cases found in live use (Table <xref rid="Tab4" ref-type="table">4</xref>). Of the additional cancer cases, four (80.0%) in the initial pilot, nine (81.1%) in the extended pilot and five (50.0%) in live use were confirmed to be invasive; zero (0.0%) in the initial pilot, one (9.1%) in the extended pilot and two (20.0%) in live use were confirmed to be in situ cancer; and one (20.0%) in the initial pilot, one (9.1%) in the extended pilot and three (30.0%) in live use had missing invasiveness information.<table-wrap id="Tab4" position="float" orientation="portrait"><label>Table 4</label><caption><p>Outcome metrics for standard double reading versus double reading plus the AI-assisted additional-reader workflow at a higher-specificity operating point</p></caption><table frame="hsides" rules="groups"><thead><tr><th rowspan="2" colspan="1">Variable</th><th colspan="2" rowspan="1">Double reading</th><th colspan="2" rowspan="1">Double reading plus the AI-assisted additional-reader workflow</th><th rowspan="2" colspan="1">Difference</th></tr><tr><th colspan="1" rowspan="1">Num/Denom</th><th colspan="1" rowspan="1">Value (95% CI)</th><th colspan="1" rowspan="1">Num/Denom</th><th colspan="1" rowspan="1">Value (95% CI)</th></tr></thead><tbody><tr><td colspan="6" rowspan="1">Results of phase 1, pilot rollout (1 site, 1 additional arbitrator, additional arbitration cases were single read), <italic toggle="yes">n</italic> = 3,746 screens</td></tr><tr><td colspan="1" rowspan="1"> CDR (per 1,000 cases)</td><td colspan="1" rowspan="1">48/3,746</td><td colspan="1" rowspan="1">12.8 (9.7–16.9)</td><td colspan="1" rowspan="1">53/3,746</td><td colspan="1" rowspan="1">14.1 (10.8–18.5)</td><td colspan="1" rowspan="1">1.3<sup>a</sup></td></tr><tr><td colspan="1" rowspan="1"> RR (%)</td><td colspan="1" rowspan="1">250/3,746</td><td colspan="1" rowspan="1">6.7 (5.9–7.5)</td><td colspan="1" rowspan="1">255/3,746</td><td colspan="1" rowspan="1">6.8 (6.0–7.7)</td><td colspan="1" rowspan="1">0.1</td></tr><tr><td colspan="1" rowspan="1"> Sen (%)</td><td colspan="1" rowspan="1">48/57</td><td colspan="1" rowspan="1">82.8 (71.7–90.4)</td><td colspan="1" rowspan="1">53/57</td><td colspan="1" rowspan="1">93.0 (83.3–97.2)</td><td colspan="1" rowspan="1">8.8<sup>a</sup></td></tr><tr><td colspan="1" rowspan="1"> Spec (%)</td><td colspan="1" rowspan="1">3,487/3,689</td><td colspan="1" rowspan="1">94.5 (93.7–95.2)</td><td colspan="1" rowspan="1">3,487/3,689</td><td colspan="1" rowspan="1">94.5 (93.7–95.2)</td><td colspan="1" rowspan="1">0.0</td></tr><tr><td colspan="1" rowspan="1"> PPV (%)</td><td colspan="1" rowspan="1">48/250</td><td colspan="1" rowspan="1">19.2 (14.8–24.5)</td><td colspan="1" rowspan="1">53/255</td><td colspan="1" rowspan="1">20.8 (16.3–26.2)</td><td colspan="1" rowspan="1">1.6</td></tr><tr><td colspan="1" rowspan="1"> Arbitration rate (%)</td><td colspan="1" rowspan="1">114/3,746</td><td colspan="1" rowspan="1">3.0 (2.5–3.6)</td><td colspan="1" rowspan="1">203/3,746</td><td colspan="1" rowspan="1">5.4 (4.7–6.2)</td><td colspan="1" rowspan="1">2.4</td></tr><tr><td colspan="1" rowspan="1"> Positive discordance rate (%)</td><td colspan="1" rowspan="1">–</td><td colspan="1" rowspan="1">–</td><td colspan="1" rowspan="1">89/3,746</td><td colspan="1" rowspan="1">2.4 (1.9–2.9)</td><td colspan="1" rowspan="1">–</td></tr><tr><td colspan="1" rowspan="1"> RR of additional arbitration (%)</td><td colspan="1" rowspan="1">–</td><td colspan="1" rowspan="1">–</td><td colspan="1" rowspan="1">5/89</td><td colspan="1" rowspan="1">5.6 (2.4–12.5)</td><td colspan="1" rowspan="1">–</td></tr><tr><td colspan="1" rowspan="1"> PPV of additional arbitration (%)</td><td colspan="1" rowspan="1">–</td><td colspan="1" rowspan="1">–</td><td colspan="1" rowspan="1">5/5</td><td colspan="1" rowspan="1">100 (56.6–100)</td><td colspan="1" rowspan="1">–</td></tr><tr><td colspan="6" rowspan="1">Results of phase 2, extended pilot (4 sites, 3 additional arbitrators, all additional arbitration cases were read by each additional arbitrator), <italic toggle="yes">n</italic> = 9,112 screens</td></tr><tr><td colspan="1" rowspan="1"> CDR (per 1,000 cases)</td><td colspan="1" rowspan="1">126/9,112</td><td colspan="1" rowspan="1">13.8 (11.6–16.4)</td><td colspan="1" rowspan="1">137/9,112</td><td colspan="1" rowspan="1">15.0 (12.7–17.7)</td><td colspan="1" rowspan="1">1.2<sup>a</sup></td></tr><tr><td colspan="1" rowspan="1"> RR (%)</td><td colspan="1" rowspan="1">639/9,112</td><td colspan="1" rowspan="1">7.0 (6.5–7.6)</td><td colspan="1" rowspan="1">653/9,112</td><td colspan="1" rowspan="1">7.2 (6.7–7.7)</td><td colspan="1" rowspan="1">0.2</td></tr><tr><td colspan="1" rowspan="1"> Sen (%)</td><td colspan="1" rowspan="1">126/142</td><td colspan="1" rowspan="1">86.9 (80.4–91.4)</td><td colspan="1" rowspan="1">137/142</td><td colspan="1" rowspan="1">96.5 (92.0–98.5)</td><td colspan="1" rowspan="1">7.7<sup>a</sup></td></tr><tr><td colspan="1" rowspan="1"> Spec (%)</td><td colspan="1" rowspan="1">8,457/8,970</td><td colspan="1" rowspan="1">94.3 (93.8–94.7)</td><td colspan="1" rowspan="1">8,454/8,970</td><td colspan="1" rowspan="1">94.2 (93.7–94.7)</td><td colspan="1" rowspan="1">0.0</td></tr><tr><td colspan="1" rowspan="1"> PPV (%)</td><td colspan="1" rowspan="1">126/639</td><td colspan="1" rowspan="1">19.7 (16.8–23.0)</td><td colspan="1" rowspan="1">137/653</td><td colspan="1" rowspan="1">21.0 (18.0–24.3)</td><td colspan="1" rowspan="1">1.3</td></tr><tr><td colspan="1" rowspan="1"> Arbitration rate (%)</td><td colspan="1" rowspan="1">270/9,112</td><td colspan="1" rowspan="1">3.0 (2.6–3.3)</td><td colspan="1" rowspan="1">544/9,112</td><td colspan="1" rowspan="1">6.0 (5.5–6.5)</td><td colspan="1" rowspan="1">3.0</td></tr><tr><td colspan="1" rowspan="1"> Positive discordance rate (%)</td><td colspan="1" rowspan="1">–</td><td colspan="1" rowspan="1">–</td><td colspan="1" rowspan="1">274/9,112</td><td colspan="1" rowspan="1">3.0 (2.7–3.4)</td><td colspan="1" rowspan="1">–</td></tr><tr><td colspan="1" rowspan="1"> RR of additional arbitration (%)</td><td colspan="1" rowspan="1">–</td><td colspan="1" rowspan="1">–</td><td colspan="1" rowspan="1">14/274</td><td colspan="1" rowspan="1">5.1 (3.1–8.4)</td><td colspan="1" rowspan="1">–</td></tr><tr><td colspan="1" rowspan="1"> PPV of additional arbitration (%)</td><td colspan="1" rowspan="1">–</td><td colspan="1" rowspan="1">–</td><td colspan="1" rowspan="1">11/14</td><td colspan="1" rowspan="1">78.6 (52.4–92.4)</td><td colspan="1" rowspan="1">–</td></tr><tr><td colspan="6" rowspan="1">Results of phase 3, live use in standard clinical practice (4 sites, 3 additional arbitrators, additional arbitration cases were single read), <italic toggle="yes">n</italic> = 15,953 screens</td></tr><tr><td colspan="1" rowspan="1"> CDR (per 1,000 cases)</td><td colspan="1" rowspan="1">238/15,953</td><td colspan="1" rowspan="1">14.9 (13.2–16.9)</td><td colspan="1" rowspan="1">248/15,953</td><td colspan="1" rowspan="1">15.5 (13.7–17.6)</td><td colspan="1" rowspan="1">0.6<sup>a</sup></td></tr><tr><td colspan="1" rowspan="1"> RR (%)</td><td colspan="1" rowspan="1">1,228/15,953</td><td colspan="1" rowspan="1">7.7 (7.3–8.1)</td><td colspan="1" rowspan="1">1,252/15,953</td><td colspan="1" rowspan="1">7.8 (7.4–8.3)</td><td colspan="1" rowspan="1">0.2</td></tr><tr><td colspan="1" rowspan="1"> Sen (%)</td><td colspan="1" rowspan="1">238/251</td><td colspan="1" rowspan="1">94.1 (90.4–96.4)</td><td colspan="1" rowspan="1">248/251</td><td colspan="1" rowspan="1">98.8 (96.5–99.6)</td><td colspan="1" rowspan="1">4.0<sup>a</sup></td></tr><tr><td colspan="1" rowspan="1"> Spec (%)</td><td colspan="1" rowspan="1">14,712/15,702</td><td colspan="1" rowspan="1">93.7 (93.3–94.1)</td><td colspan="1" rowspan="1">14,698/15,702</td><td colspan="1" rowspan="1">93.6 (93.2–94.0)</td><td colspan="1" rowspan="1">−0.1</td></tr><tr><td colspan="1" rowspan="1"> PPV (%)</td><td colspan="1" rowspan="1">238/1,228</td><td colspan="1" rowspan="1">19.4 (17.3–21.7)</td><td colspan="1" rowspan="1">248/1,252</td><td colspan="1" rowspan="1">19.8 (17.7–22.1)</td><td colspan="1" rowspan="1">0.4</td></tr><tr><td colspan="1" rowspan="1"> Arbitration rate (%)</td><td colspan="1" rowspan="1">529/15,953</td><td colspan="1" rowspan="1">3.3 (3.0–3.6)</td><td colspan="1" rowspan="1">986/15,953</td><td colspan="1" rowspan="1">6.2 (5.8–6.6)</td><td colspan="1" rowspan="1">2.9</td></tr><tr><td colspan="1" rowspan="1"> Positive discordance rate (%)</td><td colspan="1" rowspan="1">–</td><td colspan="1" rowspan="1">–</td><td colspan="1" rowspan="1">457/15,953</td><td colspan="1" rowspan="1">2.9 (2.6–3.1)</td><td colspan="1" rowspan="1">–</td></tr><tr><td colspan="1" rowspan="1"> RR of additional arbitration (%)</td><td colspan="1" rowspan="1">–</td><td colspan="1" rowspan="1">–</td><td colspan="1" rowspan="1">24/457</td><td colspan="1" rowspan="1">5.3 (3.6–7.7)</td><td colspan="1" rowspan="1">–</td></tr><tr><td colspan="1" rowspan="1"> PPV of additional arbitration (%)</td><td colspan="1" rowspan="1">–</td><td colspan="1" rowspan="1">–</td><td colspan="1" rowspan="1">10/24</td><td colspan="1" rowspan="1">41.7 (24.5–61.2)</td><td colspan="1" rowspan="1">–</td></tr></tbody></table><table-wrap-foot><p>See metric definitions in <xref rid="Sec7" ref-type="sec">Methods</xref>.</p><p><sup>a</sup>The two-sided McNemar test to assess CDR and Sen differences between double reading and double reading plus the AI-assisted additional-reader workflow resulted in <italic toggle="yes">P</italic> values of 0.063, 0.001 and &lt;0.001 for phases 1, 2 and 3, respectively. The McNemar test is based on the binomial distribution. Continuity correction was applied.</p></table-wrap-foot></table-wrap></p></sec></sec><sec id="Sec6" sec-type="discussion"><title>Discussion</title><p id="Par12">This analysis of prospective real-world usage data provides evidence that using AI in clinical practice results in a measurable increase in breast cancer detection. We analyzed the effects of the AI-assisted additional-reader workflow in two pilot phases and found that the results were maintained when AI was used in daily screening practice. Moreover, the observed clinical benefit (a significant 5–13% increase in the rate of early detection of mostly invasive and small cancerous tumors) had minimal impact on recall rates, thereby demonstrating the possibility of increasing cancer detection with no false-positive additional recalls. Although the double-reading recall rate (6.7–7.7%) in this evaluation is in line with previous results published in the UK and Europe<sup><xref ref-type="bibr" rid="CR9">9</xref>,<xref ref-type="bibr" rid="CR15">15</xref></sup>, the double-reading CDR is higher (14/1,000) than previously reported<sup><xref ref-type="bibr" rid="CR9">9</xref></sup>—possibly resulting from the resumption of breast cancer screening programs after the coronavirus disease pandemic. Nevertheless, the AI-assisted additional-reader workflow supported the screening service by further increasing the rate of early cancer detection. It also can potentially reduce the proportion of cases requiring additional arbitration review to &lt;3% of cases while still achieving increased cancer detection by 0.5–1.3 per 1,000 cases, corresponding to a 4–10% relative increase in cancer detection using a higher-specificity operating point. Future work investigating the implementation of a variety of operating points would be needed to confirm the extent of achievable improvement in early cancer detection in the context of sites with different needs, capacities and screening population characteristics.</p><p id="Par13">Implementing AI into the diagnostic workflow requires careful monitoring of continued performance over time<sup><xref ref-type="bibr" rid="CR16">16</xref></sup>. For the AI-assisted additional-reader workflow, the effectiveness of downstream clinical assessments of recalled positive discordant cases should be examined to ensure that potential cancer cases are found. Moreover, the AI-assisted additional-reader workflow could be combined with workflows focused on workload savings, such as using AI as an independent second reader. Large-scale retrospective studies of the same AI system used in this assessment have demonstrated that AI as an independent second reader can offer up to 45% workload savings<sup><xref ref-type="bibr" rid="CR8">8</xref>,<xref ref-type="bibr" rid="CR9">9</xref></sup>, offsetting the 3–11% additional arbitration reads (1–6% additional overall reading workload) for the AI-assisted additional-reader workflow while providing the benefit of increased cancer detection.</p><p id="Par14">The AI-assisted additional-reader workflow was designed to flag high-priority cases not recalled by standard double reading, likely making the flagged set of cases a more difficult or complex set to read. We believe that this would be helpful in the training of mammogram readers. The spectrum of disease detected with the AI-assisted additional-reader workflow will be assessed in future work covering features such as invasiveness, tumor size, grade and lymph node status.</p><p id="Par15">Several limitations need to be considered when interpreting the presented results. First, data were collected from only one breast cancer screening institution (with four sites) in one country. As screening programs vary between clinical sites and countries, future studies must confirm the benefit of the AI-assisted additional-reader workflow in other settings and screening populations. Furthermore, as only one commercial AI system was evaluated, the results may not be representative of other commercially available systems. Additionally, given that the follow-up period in this prospective assessment ranged only from 2 to 9 months, no information is yet available about possible interval cancer cases in the studied population. A longer follow-up analysis is required for a more accurate assessment of AI’s potential for improving cancer detection in the context of interval cancer occurrence. Moreover, the impact of inter-reader variation on the AI-assisted additional-reader workflow’s screening outcomes remains unclear and needs to be assessed in follow-up work.</p><p id="Par16">Despite the many challenges in developing, validating, deploying and monitoring AI to ensure patient safety, this evaluation shows that a commercially available AI system can be effectively deployed, with its previously predicted benefits realized in a prospective real-world assessment of a live clinical workflow. We believe that the findings highlight opportunities for using AI in breast screening while demonstrating concrete steps for its safe deployment. The phased prospective approach underlines the potential for various AI adoption pathways.</p></sec><sec id="Sec7"><title>Methods</title><sec id="Sec8"><title>Datasets for analysis</title><p id="Par17">This study is an analysis of postmarket data collected at MaMMa Klinika, a large breast cancer screening institution in Hungary. Structured query language was used to collect data. Custom code using Python software version 3.8.8 and open-source Python packages, including pandas version 1.2.4, NumPy version 1.20.1, sklearn version 0.24.1 and statsmodels version 0.12.2, were used for data analysis. The analysis complied with all relevant ethical regulations. External ethical review was not required as the AI system was used as part of the standard of care in the screening service at each implementation phase of this service evaluation. Ethical considerations were reviewed internally by the screening service provider, MaMMa Klinika. The evaluation used deidentified data and presented results in aggregate without listing data of individual screening participants to protect their anonymity. As a consequence, the evaluation also did not require patient consent.</p></sec><sec id="Sec9"><title>Metrics</title><p id="Par18">Standard breast screening metrics, CDR and recall rate were primarily used to assess the effects of the AI-assisted additional-reader workflow compared to standard double reading without AI. CDR was calculated as the number of screen-detected cancer cases detected divided by the number of all screening cases. Recall rate was calculated as the number of cases recalled divided by the number of all cases; this should not be confused with the term ‘recall’ often used as a metric for sensitivity in machine learning. Arbitration rate was calculated as the number of arbitrations conducted divided by the number of all cases, with the double-reading arbitration rate including only double-reading arbitrations and the total arbitration rate including double-reading and additional-reader arbitrations. PPV was calculated as the number of screen-detected cancer cases divided by the number of recalled screens. Sensitivity was calculated as the number of screen-detected cancer cases divided by the number of all known positive screens. Specificity was calculated as the number of non-recalled screens divided by the number of all non-positive screens. Positive discordance rate was calculated as the number of AI-flagged positive discordant cases divided by the number of all cases. As the AI-assisted additional-reader workflow occurs subsequently to the double-reading workflow on the same cases, paired comparisons between the AI-assisted additional-reader and double-reading workflows were possible, with an exact measurement of the impact of AI in terms of additional recalls and cancer cases found. All detected cancer cases were confirmed with biopsy or histopathological examination within 12 months of the original screen or judged to be cancer by the patient tumor board (multidisciplinary team).</p></sec><sec id="Sec10"><title>Statistical analysis</title><p id="Par19">No statistical method was used to predetermine sample sizes. No data were excluded from the analyses. Blinding was not required as randomization was not applied. The standard double-reading process did not involve the AI system, and readers were blinded to the AI system’s output during the double-reading process. The Wilson score method was used to calculate 95% CIs. The statistical significance of CDR differences was assessed using the McNemar test. A <italic toggle="yes">P</italic> value of &lt;0.05 was defined as statistically significant.</p></sec><sec id="Sec11"><title>AI system</title><p id="Par20">This evaluation used a commercially available AI system (Mia version 2.0, Kheiron Medical Technologies). The AI system is intended to process only cases from female participants and works with standard DICOM (Digital Imaging and Communications in Medicine) cases as inputs. The AI system analyzes four images with two standard full-field digital mammography views (craniocaudal and mediolateral oblique) per breast. The AI system’s primary output per case is a single binary recommendation of ‘recall’ (for further assessment based on findings suggestive of malignancy) or ‘no recall’ (no further assessment until the next screening interval). The AI system can provide binary recall recommendations for six predetermined operating points, ranging from having a balanced trade-off between sensitivity and specificity to having trade-offs that emphasize either sensitivity or specificity. The AI system’s balanced sensitivity/specificity and higher-specificity operating points are most relevant when the AI system is used in the AI-assisted additional-reader workflow. The set of cases flagged by the AI system’s higher-specificity operating point in the AI-assisted additional-reader workflow is always a subset of the cases flagged by the AI system’s balanced sensitivity/specificity operating point. Therefore, results at the higher-specificity operating point can be precisely simulated based on the balanced operating point results. The optionality between the different operating point trade-offs makes a significant difference for practical applicability at sites with differing workforces. Additionally, the AI system provides regions of interest indicating image locations showing characteristics most suggestive of malignancy. Depending on the clinical workflow and exact integration of the AI system, the AI’s recommendation may be used independently or combined with human reader assessment.</p><p id="Par21">The underlying technology of the AI system is based on deep convolutional neural networks (CNNs), which are state-of-the-art machine learning tools for image classification. The AI system is a combination (also known as an ensemble) of multiple models with a diverse set of different CNN architectures. Each model was trained for malignancy detection. The final prediction of the ensemble is obtained by aggregating individual model outputs, with a subsequent threshold applied to the malignancy detection score to generate a binary recommendation of ‘recall’ or ‘no recall’. The thresholds relate to one of the AI system’s six predetermined, clinically meaningful operating points according to desired sensitivity/specificity trade-offs.</p><p id="Par22">The AI system was trained on a heterogeneous, large-scale collection of more than 1 million images from real-world screening programs across different countries, multiple sites and equipment from different vendors over a period of &gt;10 years. Positive cases were defined as pathology-proven malignancies confirmed by fine-needle aspiration cytology, core needle biopsy, vacuum-assisted core biopsy and/or histological analysis of surgical specimens. Negative cases were confirmed through multiple years of follow-up.</p><p id="Par23">The AI software version and operating points used in the present evaluation were fixed before each phase. None of the evaluation data were used in any aspect of algorithm development.</p><p id="Par24">The AI system’s performance, generalizability and clinical utility were previously confirmed in a large-scale retrospective AI generalizability study<sup><xref ref-type="bibr" rid="CR8">8</xref>,<xref ref-type="bibr" rid="CR9">9</xref>,<xref ref-type="bibr" rid="CR14">14</xref></sup>. The study demonstrated that double reading with the AI system, compared to human double reading, resulted in at least noninferior recall rate, CDR, sensitivity, specificity and PPV for each mammography vendor and site, with superior recall rate, specificity and PPV observed for some mammography vendors and sites<sup><xref ref-type="bibr" rid="CR9">9</xref></sup>. The double-reading simulation with the AI system indicated that using AI as an independent reader (in all cases it could process) can result in a 3.3–12.3% increase in the arbitration rate<sup><xref ref-type="bibr" rid="CR9">9</xref></sup> but can reduce human workload by 30.0–44.8%. AI as a supporting reader (used as a second reader only when it agrees with the first human reader) was found to be superior or noninferior on all screening metrics compared to human double reading while nearly halving the number of arbitrations (from 3.4% to 1.8%) and reducing the number of cases requiring second human reading (by up to 87%)<sup><xref ref-type="bibr" rid="CR8">8</xref></sup>. Additionally, no differences in prognostic features (invasiveness, grade, tumor size and lymph node status) were found between the cancer cases detected by the AI system and those detected by human readers<sup><xref ref-type="bibr" rid="CR14">14</xref></sup>. These findings imply that cancer cases detected by the AI system and human readers are likely to have similar clinical courses and outcomes, with limited or no downstream effects on screening programs, supporting the potential role of AI as a reader in the double-reading workflow.</p></sec><sec id="Sec12"><title>Reporting summary</title><p id="Par25">Further information on research design is available in the <xref rid="MOESM1" ref-type="media">Nature Portfolio Reporting Summary</xref> linked to this article.</p></sec></sec><sec id="Sec13" sec-type="materials|methods"><title>Online content</title><p id="Par26">Any methods, additional references, Nature Portfolio reporting summaries, source data, extended data, supplementary information, acknowledgements, peer review information; details of author contributions and competing interests; and statements of data and code availability are available at 10.1038/s41591-023-02625-9.</p></sec><sec sec-type="supplementary-material"><sec id="Sec14"><title>Supplementary information</title><p>
<supplementary-material content-type="local-data" id="MOESM1" position="float" orientation="portrait"><media xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="41591_2023_2625_MOESM1_ESM.pdf" position="float" orientation="portrait"><?suppdata-name 41591_2023_2625_MOESM1_ESM.pdf?><?suppdata-size 1831840?><?suppdata-md5 78fb11977c95c3be2b7c92a81c671960?><?suppdata-image-server-status NEVER_LOAD?><?suppdata-mime-type application?><?suppdata-mime-sub-type pdf?><?suppdata-cloudpmc-urn urn:app:80a2/10719086/78fb11977c95/41591_2023_2625_MOESM1_ESM.pdf?><caption><p>Reporting Summary</p></caption></media></supplementary-material>
</p></sec></sec></body><back><fn-group><fn><p><bold>Publisher’s note</bold> Springer Nature remains neutral with regard to jurisdictional claims in published maps and institutional affiliations.</p></fn></fn-group><sec><title>Supplementary information</title><p>The online version contains supplementary material available at 10.1038/s41591-023-02625-9.</p></sec><ack><title>Acknowledgements</title><p>Annie Y. Ng, Cary J.G. Oberije and Éva Ambrózay contributed equally to this article and share first authorship. We thank MaMMa Egészségügyi Zrt. (MaMMa Klinika), Béker-Soft Informatika Kft., A. Vadászy, D. Visi, R. Kovács, C. Gadóczi, T. Rijken, J. Yearsley and S. Kerruish for supporting the collection of data and execution of the evaluation.</p></ack><notes notes-type="author-contribution"><title>Author contributions</title><p>É.A., P.D.K., E.K. and A.Y.N. contributed to the design of the work. É.A., E.S. and O.S. contributed to clinical data collection. C.J.G.O., A.Y.N., G. Fox and P.D.K. contributed to data analysis. C.J.G.O., A.Y.N., P.D.K., G. Fox, E.A.M. and G. Forrai contributed to data interpretation. A.Y.N., C.J.G.O., B.G. and P.D.K. contributed to manuscript drafting. A.Y.N., C.J.G.O., B.G., P.D.K., E.A.M. and G. Forrai contributed to manuscript revision. All authors read and approved the manuscript.</p></notes><notes notes-type="peer-review"><title>Peer review</title><sec id="FPar1"><title>Peer review information</title><p id="Par27"><italic toggle="yes">Nature Medicine</italic> thanks Ritse Mann and the other, anonymous, reviewer(s) for their contribution to the peer review of this work. Primary Handling Editor: Ming Yang, in collaboration with the <italic toggle="yes">Nature Medicine</italic> team.</p></sec></notes><notes notes-type="data-availability"><title>Data availability</title><p>Access to patient-level data and supporting clinical information can be made available upon request, contingent on patient privacy and confidentiality obligations and subject to information governance at MaMMa Klinika (Hungary). Data access requests can be made to the corresponding author by email at annie@kheironmed.com and will be processed within 4 weeks.</p></notes><notes notes-type="data-availability"><title>Code availability</title><p>The code used for training and deploying the evaluated AI system has many dependencies on internal tooling, proprietary components, infrastructure and hardware. Therefore, full code release is not feasible. We provide a technical description of the AI system in the online <xref rid="Sec7" ref-type="sec">Methods</xref>, together with a code repository to facilitate the reproducibility of research involving deep learning models for breast cancer detection using digital mammography. The code provided at <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="https://github.com/Kheiron-Medical/mammo-net">https://github.com/Kheiron-Medical/mammo-net</ext-link> demonstrates the training and testing of state-of-the-art CNNs that build the core component of most commercially available AI systems for breast cancer detection.</p></notes><notes id="FPar2" notes-type="COI-statement"><title>Competing interests</title><p id="Par28">This postmarket analysis was funded by Kheiron Medical Technologies Ltd. (‘Kheiron’). C.J.G.O., E.K., A.Y.N., G. Fox, B.G. and P.D.K. are employees of Kheiron and hold stock options as part of the standard compensation package. E.A.M. holds an advisory board member position and stock options at Kheiron Medical Technologies. G. Forrai is a paid consultant for Kheiron Medical Technologies. É.A., E.S. and O.S. declare no competing interests.</p></notes><ref-list id="Bib1"><title>References</title><ref id="CR1"><label>1.</label><element-citation publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Duffy</surname><given-names>SW</given-names></name><etal/></person-group><article-title>Mammography screening reduces rates of advanced and fatal breast cancers: results in 549,091 women</article-title><source>Cancer</source><year>2020</year><volume>126</volume><fpage>2971</fpage><lpage>2979</lpage><pub-id pub-id-type="doi">10.1002/cncr.32859</pub-id><pub-id pub-id-type="pmid">32390151</pub-id><pub-id pub-id-type="pmcid">PMC7318598</pub-id></element-citation></ref><ref id="CR2"><label>2.</label><element-citation publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zielonke</surname><given-names>N</given-names></name><etal/></person-group><article-title>Evidence for reducing cancer-specific mortality due to screening for breast cancer in Europe: a systematic review</article-title><source>Eur. J. Cancer</source><year>2020</year><volume>127</volume><fpage>191</fpage><lpage>206</lpage><pub-id pub-id-type="doi">10.1016/j.ejca.2019.12.010</pub-id><pub-id pub-id-type="pmid">31932175</pub-id></element-citation></ref><ref id="CR3"><label>3.</label><element-citation publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Houssami</surname><given-names>N</given-names></name><name name-style="western"><surname>Hunter</surname><given-names>K</given-names></name></person-group><article-title>The epidemiology, radiology and biological characteristics of interval breast cancers in population mammography screening</article-title><source>NPJ Breast Cancer</source><year>2017</year><volume>3</volume><fpage>12</fpage><pub-id pub-id-type="doi">10.1038/s41523-017-0014-x</pub-id><pub-id pub-id-type="pmid">28649652</pub-id><pub-id pub-id-type="pmcid">PMC5460204</pub-id></element-citation></ref><ref id="CR4"><label>4.</label><element-citation publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hovda</surname><given-names>T</given-names></name><name name-style="western"><surname>Tsuruda</surname><given-names>K</given-names></name><name name-style="western"><surname>Hoff</surname><given-names>SR</given-names></name><name name-style="western"><surname>Sahlberg</surname><given-names>KK</given-names></name><name name-style="western"><surname>Hofvind</surname><given-names>S</given-names></name></person-group><article-title>Radiological review of prior screening mammograms of screen-detected breast cancer</article-title><source>Eur. Radiol.</source><year>2021</year><volume>31</volume><fpage>2568</fpage><lpage>2579</lpage><pub-id pub-id-type="doi">10.1007/s00330-020-07130-y</pub-id><pub-id pub-id-type="pmid">33001307</pub-id><pub-id pub-id-type="pmcid">PMC7979605</pub-id></element-citation></ref><ref id="CR5"><label>5.</label><element-citation publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lehman</surname><given-names>CD</given-names></name><etal/></person-group><article-title>Diagnostic accuracy of digital screening mammography with and without computer-aided detection</article-title><source>JAMA Intern. Med.</source><year>2015</year><volume>175</volume><fpage>1828</fpage><lpage>1837</lpage><pub-id pub-id-type="doi">10.1001/jamainternmed.2015.5231</pub-id><pub-id pub-id-type="pmid">26414882</pub-id><pub-id pub-id-type="pmcid">PMC4836172</pub-id></element-citation></ref><ref id="CR6"><label>6.</label><element-citation publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>McKinney</surname><given-names>SM</given-names></name><etal/></person-group><article-title>International evaluation of an AI system for breast cancer screening</article-title><source>Nature</source><year>2020</year><volume>577</volume><fpage>89</fpage><lpage>94</lpage><pub-id pub-id-type="doi">10.1038/s41586-019-1799-6</pub-id><pub-id pub-id-type="pmid">31894144</pub-id></element-citation></ref><ref id="CR7"><label>7.</label><element-citation publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Leibig</surname><given-names>C</given-names></name><etal/></person-group><article-title>Combining the strengths of radiologists and AI for breast cancer screening: a retrospective analysis</article-title><source>Lancet Digit. Health</source><year>2022</year><volume>4</volume><fpage>e507</fpage><lpage>e519</lpage><pub-id pub-id-type="doi">10.1016/S2589-7500(22)00070-X</pub-id><pub-id pub-id-type="pmid">35750400</pub-id><pub-id pub-id-type="pmcid">PMC9839981</pub-id></element-citation></ref><ref id="CR8"><label>8.</label><mixed-citation publication-type="other">Ng, A. Y. et al. Artificial intelligence as supporting reader in breast screening: a novel workflow to preserve quality and reduce workload. <italic toggle="yes">J. Breast Imaging</italic>10.1093/jbi/wbad010 (2023).<pub-id pub-id-type="doi" assigning-authority="pmc">10.1093/jbi/wbad010</pub-id><pub-id pub-id-type="pmid">38416889</pub-id></mixed-citation></ref><ref id="CR9"><label>9.</label><element-citation publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sharma</surname><given-names>N</given-names></name><etal/></person-group><article-title>Multi-vendor evaluation of artificial intelligence as an independent reader for double reading in breast cancer screening on 275,900 mammograms</article-title><source>BMC Cancer</source><year>2023</year><volume>23</volume><fpage>460</fpage><pub-id pub-id-type="doi">10.1186/s12885-023-10890-7</pub-id><pub-id pub-id-type="pmid">37208717</pub-id><pub-id pub-id-type="pmcid">PMC10197505</pub-id></element-citation></ref><ref id="CR10"><label>10.</label><mixed-citation publication-type="other">Koch, H. W., Larsen, M., Bartsch, H., Kurz, K. D. &amp; Hofvind, S. Artificial intelligence in BreastScreen Norway: a retrospective analysis of a cancer-enriched sample including 1254 breast cancer cases. <italic toggle="yes">Eur. Radiol.</italic>10.1007/s00330-023-09461-y (2023).<pub-id pub-id-type="doi" assigning-authority="pmc">10.1007/s00330-023-09461-y</pub-id><pub-id pub-id-type="pmcid">PMC10121532</pub-id><pub-id pub-id-type="pmid">36917260</pub-id></mixed-citation></ref><ref id="CR11"><label>11.</label><mixed-citation publication-type="other">Kim, C. et al. Multicentre external validation of a commercial artificial intelligence software to analyse chest radiographs in health screening environments with low disease prevalence. <italic toggle="yes">Eur. Radiol.</italic>10.1007/s00330-022-09315-z (2023).<pub-id pub-id-type="doi" assigning-authority="pmc">10.1007/s00330-022-09315-z</pub-id><pub-id pub-id-type="pmid">36624227</pub-id></mixed-citation></ref><ref id="CR12"><label>12.</label><element-citation publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Marinovich</surname><given-names>ML</given-names></name><etal/></person-group><article-title>Artificial intelligence (AI) for breast cancer screening: BreastScreen population-based cohort study of cancer detection</article-title><source>EBioMedicine</source><year>2023</year><volume>90</volume><fpage>104498</fpage><pub-id pub-id-type="doi">10.1016/j.ebiom.2023.104498</pub-id><pub-id pub-id-type="pmid">36863255</pub-id><pub-id pub-id-type="pmcid">PMC9996220</pub-id></element-citation></ref><ref id="CR13"><label>13.</label><element-citation publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Freeman</surname><given-names>K</given-names></name><etal/></person-group><article-title>Use of artificial intelligence for image analysis in breast cancer screening programmes: systematic review of test accuracy</article-title><source>BMJ</source><year>2021</year><volume>374</volume><fpage>n1872</fpage><pub-id pub-id-type="doi">10.1136/bmj.n1872</pub-id><pub-id pub-id-type="pmid">34470740</pub-id><pub-id pub-id-type="pmcid">PMC8409323</pub-id></element-citation></ref><ref id="CR14"><label>14.</label><element-citation publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Oberije</surname><given-names>CJG</given-names></name><etal/></person-group><article-title>Comparing prognostic factors of cancers identified by artificial intelligence (AI) and human readers in breast cancer screening</article-title><source>Cancers</source><year>2023</year><volume>15</volume><fpage>3069</fpage><pub-id pub-id-type="doi">10.3390/cancers15123069</pub-id><pub-id pub-id-type="pmid">37370680</pub-id><pub-id pub-id-type="pmcid">PMC10296295</pub-id></element-citation></ref><ref id="CR15"><label>15.</label><element-citation publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Peintinger</surname><given-names>F</given-names></name></person-group><article-title>National breast screening programs across Europe</article-title><source>Breast Care</source><year>2019</year><volume>14</volume><fpage>354</fpage><lpage>358</lpage><pub-id pub-id-type="doi">10.1159/000503715</pub-id><pub-id pub-id-type="pmid">31933580</pub-id><pub-id pub-id-type="pmcid">PMC6940461</pub-id></element-citation></ref><ref id="CR16"><label>16.</label><mixed-citation publication-type="other">Sahiner, B., Chen, W., Samala, R. K. &amp; Petrick, N. Data drift in medical machine learning: implications and potential remedies. <italic toggle="yes">Br. J. Radiol</italic>. 10.1259/bjr.20220878 (2023).<pub-id pub-id-type="doi" assigning-authority="pmc">10.1259/bjr.20220878</pub-id><pub-id pub-id-type="pmcid">PMC10546450</pub-id><pub-id pub-id-type="pmid">36971405</pub-id></mixed-citation></ref><ref id="CR17"><label>17.</label><element-citation publication-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gram</surname><given-names>IT</given-names></name><name name-style="western"><surname>Funkhouser</surname><given-names>E</given-names></name><name name-style="western"><surname>Tabár</surname><given-names>L</given-names></name></person-group><article-title>The Tabár classification of mammographic parenchymal patterns</article-title><source>Eur. J. Radiol.</source><year>1997</year><volume>24</volume><fpage>131</fpage><lpage>136</lpage><pub-id pub-id-type="doi">10.1016/S0720-048X(96)01138-2</pub-id><pub-id pub-id-type="pmid">9097055</pub-id></element-citation></ref></ref-list></back></article>