<?xml version='1.0' encoding='UTF-8'?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.4 20241031//EN" "JATS-journalpublishing1-4.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:mml="http://www.w3.org/1998/Math/MathML" article-type="research-article" dtd-version="1.4" xml:lang="en">
  <front>
    <journal-meta>
      <journal-id journal-id-type="publisher-id">aside-oc</journal-id>
      <journal-title-group>
        <journal-title>ASIDE Oncology</journal-title>
      </journal-title-group>
      <issn pub-type="ppub">3069-9959</issn>
      <issn pub-type="epub">3069-9967</issn>
      <publisher>
        <publisher-name>PubPorta Publishing LLC</publisher-name>
      </publisher>
    </journal-meta>
    <article-meta>
      <article-id pub-id-type="doi">10.71079/ASIDE.Onc.030926535</article-id>
      <article-categories>
        <subj-group subj-group-type="heading">
          <subject>Original Article</subject>
        </subj-group>
      </article-categories>
      <title-group>
        <article-title>Federated Hybrid CNN Vision Transformer Framework for Breast Cancer Classification Under Simulated Non-IID Settings</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="author" corresp="yes" id="contrib-735e7b388050">
          <contrib-id contrib-id-type="orcid">https://orcid.org/0009-0002-0988-0131</contrib-id>
          <name>
            <surname>Panda</surname>
            <given-names>Bandhan</given-names>
          </name>
          <role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
          <role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
          <role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
          <role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing – Original Draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing – Original Draft</role>
          <role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing – Review &amp; Editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing – Review &amp; Editing</role>
          <xref ref-type="aff" rid="aff1"/>
          <xref ref-type="corresp" rid="cor1"/>
          <email>bandhan.panda@nist.edu</email>
        </contrib>
        <contrib contrib-type="author" id="contrib-a50325675603">
          <contrib-id contrib-id-type="orcid">https://orcid.org/0009-0009-0924-8188</contrib-id>
          <name>
            <surname>Patro</surname>
            <given-names>Bibek Kumar</given-names>
          </name>
          <role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Software" vocab-term-identifier="https://credit.niso.org/contributor-roles/software/">Software</role>
          <role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
          <role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Data Curation" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/">Data Curation</role>
          <role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Formal Analysis" vocab-term-identifier="https://credit.niso.org/contributor-roles/formal-analysis/">Formal Analysis</role>
          <role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
          <role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing – Original Draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing – Original Draft</role>
          <xref ref-type="aff" rid="aff1"/>
        </contrib>
        <contrib contrib-type="author" id="contrib-1140a63cdc06">
          <contrib-id contrib-id-type="orcid">https://orcid.org/0009-0000-9571-779X</contrib-id>
          <name>
            <surname>Das</surname>
            <given-names>Siba Sundar</given-names>
          </name>
          <role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Software" vocab-term-identifier="https://credit.niso.org/contributor-roles/software/">Software</role>
          <role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
          <role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Data Curation" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/">Data Curation</role>
          <role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
          <role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
          <role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing – Original Draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing – Original Draft</role>
          <xref ref-type="aff" rid="aff1"/>
        </contrib>
        <contrib contrib-type="author" id="contrib-e358b296dfec">
          <contrib-id contrib-id-type="orcid">https://orcid.org/0009-0008-1900-4542</contrib-id>
          <name>
            <surname>Kar</surname>
            <given-names>Santosh Kumar</given-names>
          </name>
          <role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
          <role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
          <role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
          <role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing – Original Draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing – Original Draft</role>
          <role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing – Review &amp; Editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing – Review &amp; Editing</role>
          <xref ref-type="aff" rid="aff1"/>
        </contrib>
      </contrib-group>
      <aff id="aff1">
        <institution>Department of Computer Science &amp; Engineering, NIST University, Berhampur, Odisha</institution>
        <country>India</country>
      </aff>
      <author-notes>
        <corresp id="cor1">Corresponding author. E-mail: <email>bandhan.panda@nist.edu</email></corresp>
        <fn fn-type="coi-statement">
          <p>The authors declare no competing interests that could have influenced the objectivity or outcome of this research.</p>
        </fn>
      </author-notes>
      <pub-date publication-format="electronic" date-type="pub" iso-8601-date="2026-03-09">
        <day>09</day>
        <month>03</month>
        <year>2026</year>
      </pub-date>
      <pub-date publication-format="electronic" date-type="collection" iso-8601-date="2026">
        <year>2026</year>
      </pub-date>
      <volume>1</volume>
      <issue>1</issue>
      <fpage>28</fpage>
      <lpage>37</lpage>
      <history>
        <date date-type="received" iso-8601-date="2026-01-23">
          <day>23</day>
          <month>01</month>
          <year>2026</year>
        </date>
        <date date-type="rev-recd" iso-8601-date="2026-03-01">
          <day>01</day>
          <month>03</month>
          <year>2026</year>
        </date>
        <date date-type="accepted" iso-8601-date="2026-03-05">
          <day>05</day>
          <month>03</month>
          <year>2026</year>
        </date>
      </history>
      <permissions>
        <copyright-year>2026</copyright-year>
        <copyright-holder>Bandhan Panda, Bibek Kumar Patro, Siba Sundar Das, Santosh Kumar Kar</copyright-holder>
        <license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0">
          <license-p>This is an open-access article.</license-p>
        </license>
      </permissions>
      <abstract>
        <p>Background: Breast cancer diagnosis increasingly relies on data-driven learning from heterogeneous medical imaging sources. However, centralized deep learning approaches face major limitations, including privacy risks, institutional data silos, and limited generalization across imaging modalities.</p>
        <p>Methods: This study proposes a privacy-aware federated learning framework integrating a hybrid CNN–Vision Transformer architecture for breast cancer classification under simulated non-identically distributed (non-IID) conditions. Public datasets representing different imaging modalities, BreakHis (histopathology), INbreast and CBIS-DDSM (mammography), and BUSI (ultrasound) are treated as independent federated clients to emulate multi-institutional collaboration. Each client trains a local model, and only the model parameters are aggregated via Federated Averaging, thereby preserving data locality. The hybrid architecture combines a ResNet-18 convolutional branch for local feature extraction with a Vision Transformer branch for global contextual representation.</p>
        <p>Results: Across ten federated communication rounds, the global model demonstrates stable convergence under heterogeneous client distributions. The final global validation performance reaches 72.43% accuracy, AUC 0.7475, and F1-score 0.718. Evaluation on the pooled test cohort achieves approximately 84% overall accuracy with a weighted F1-score of 0.82, while malignant recall approaches 96%, prioritizing clinically critical cancer detection. Qualitative explainability analysis indicates that the model focuses on diagnostically relevant tissue regions.</p>
        <p>Conclusion: The results demonstrate the feasibility of hybrid CNN–Vision Transformer training in a federated setting for privacy-aware breast cancer classification across heterogeneous imaging domains.</p>
      </abstract>
      <kwd-group>
        <kwd>Breast cancer diagnosis</kwd>
        <kwd>Federated learning</kwd>
        <kwd>Vision transformer</kwd>
        <kwd>Explainable artificial intelligence</kwd>
        <kwd>Privacy-aware federated learning</kwd>
      </kwd-group>
      <funding-group>
        <funding-statement>The authors declare that no specific grant or funding was received for this research from any public, commercial, or not-for-profit funding agency.</funding-statement>
      </funding-group>
    </article-meta>
  </front>
  <body>
    <sec id="sec-cfdf73e00824">
      <title>Introduction</title>
      <p id="blk-4ed4cede590a">Breast cancer has been among the most common causes of death as a result of cancer in diseases worldwide, hence the need to employ effective and prompt diagnostic steps capable of aiding in clinical decision-making. The recent developments of deep learning have shown great potential for automated detection of breast cancer in various imaging modalities such as histopathology, mammography, and ultrasound images. Although there have been these improvements, the bulk of the current practices is dependent on centralized training paradigms, whereby sensitive patient information has to be accumulated at a point of centralization. These practices are associated with significant questions of patient privacy, compliance with regulations, and institutional management of data, especially in the cases of healthcare settings with harsh data protection policies [<sup><xref ref-type="bibr" rid="ref-6f67085b37d6">1</xref></sup>,<sup><xref ref-type="bibr" rid="ref-707ad2879e3f">2</xref></sup>].</p>
      <p id="blk-e485fc01a268">Federated learning has become one of the most promising alternatives, which allows collaborative model training across distributed data sources without requiring the exchange of raw patient data [<sup><xref ref-type="bibr" rid="ref-8aa40c20886c">3</xref></sup>,<sup><xref ref-type="bibr" rid="ref-707ad2879e3f">2</xref></sup>]. The paradigm is particularly suitable for medical imaging, where data are inherently partitioned across hospitals, laboratories, and diagnostic facilities. Nonetheless, federated learning introduces technical challenges, including statistical heterogeneity, communication inefficiency, and instability under non-identically distributed (non-IID) data conditions [<sup><xref ref-type="bibr" rid="ref-8aa40c20886c">3</xref></sup>,<sup><xref ref-type="bibr" rid="ref-3495eb21e423">4</xref></sup>]. Disparities in imaging characteristics across acquisition protocols and modalities are especially pronounced in breast cancer diagnosis, where histopathology, mammography, and ultrasound images exhibit substantial structural and intensity differences. In this study, federated learning is implemented as a simulation framework in which publicly available datasets are treated as independent federated clients to emulate multi-institutional collaboration. No explicit privacy threat model, such as membership inference, gradient leakage, or model inversion attacks, is evaluated, and no formal differential privacy or secure aggregation guarantees are experimentally validated. Accordingly, the proposed framework should be interpreted as privacy-aware and privacy-motivated rather than privacy-guaranteeing.</p>
      <p id="blk-4d08b775287a">Parallel to this, Vision Transformers (ViTs) have received growing recognition regarding their capabilities to compute long-range spatial coupling, as well as global contextual associations, and frequently perform better than traditional convolutional neural networks on image recognition tasks [<sup><xref ref-type="bibr" rid="ref-1b499b2a31ac">5</xref></sup>,<sup><xref ref-type="bibr" rid="ref-e4debb87f107">6</xref></sup>]. ViTs have demonstrated positive results in medical imaging tasks that utilize holistic feature representations. However, their incorporation in federated learning systems to diagnose breast cancer is comparatively low, especially in convergence behaviour in heterogeneous settings, modular architectural design, and interpretability [<sup><xref ref-type="bibr" rid="ref-6f67085b37d6">1</xref></sup>,<sup><xref ref-type="bibr" rid="ref-b2ab91f7172f">7</xref></sup>]. It should also be noted that the current research paper uses a federated learning framework of simulation, where publicly available datasets are considered to be independent federated clients. This design provides a surrogate of multi-institutional learning, where it is possible to theoretically analyse non-IID behaviour based on non-IID data sources, without privacy and access limitations of actual clinical partnerships. Although this proxy describes important statistical issues, such as domain shift, class imbalance, and feature distribution in mode, it lacks the realistic modelling of real-world institutional issues, such as communication latency, governance policy, or site-specific calibration procedures.</p>
      <p id="blk-c76c0cf6c0c2">In addition, the proposed framework uses a single binary classification head that cuts across all domains of imaging, directly exposing the model to cross-domain shift in images. Accordingly, assertions about imaging cross-modality generalization are made in the framework of this single decision boundary and its recognition that there is no modality-specific calibration or threshold optimization to date, which is another key avenue to future clinical translation. This work is motivated by recent studies that have highlighted the need to establish robust deep learning architectures and explainable decision-making in medical image analysis [<sup><xref ref-type="bibr" rid="ref-c27bab27efc4">8</xref></sup>,<sup><xref ref-type="bibr" rid="ref-08a1a08f747f">9</xref></sup>,<sup><xref ref-type="bibr" rid="ref-262cc96f9f56">10</xref></sup>], which motivates the current study to propose a federated hybrid CNN-Vision Transformer framework that would be able to work under simulated multi-source non-IID conditions at high diagnostic sensitivity and interpretation rates.</p>
    </sec>
    <sec id="sec-2538c1bf5aa9">
      <title>Related work</title>
      <sec id="sec-a68fe15d32c8">
        <title>Federated Learning in Medical Imaging</title>
        <p id="blk-36eccb3df62a">The initial concepts of federated learning were proposed to train a model over decentralized data collaboratively, and data privacy was maintained [<sup><xref ref-type="bibr" rid="ref-8aa40c20886c">3</xref></sup>,<sup><xref ref-type="bibr" rid="ref-707ad2879e3f">2</xref></sup>]. Its applicability to medical imaging has been well acknowledged, since medical information is naturally spread out across medical institutions and is highly regulated due to its sensitive nature. Several works have investigated the use of federated learning in clinical practice, showing that it is possible in various clinical settings, including radiology, pathology, and disease classification [<sup><xref ref-type="bibr" rid="ref-1bae84594c98">11</xref></sup>,<sup><xref ref-type="bibr" rid="ref-08a1a08f747f">9</xref></sup>]. Nevertheless, heterogeneity optimization federated by nonuniform data sets is a basic issue. Previous literature has revealed that non-IID data among clients may result in the slow convergence and deterioration of performance, especially when client data vary greatly in size and modality [<sup><xref ref-type="bibr" rid="ref-3495eb21e423">4</xref></sup>]. Such constraints require thorough architectural planning and assessment of a medical real-life setting.</p>
      </sec>
      <sec id="sec-76923eebd55c">
        <title>Medical Image Analysis Vision Transformers</title>
        <p id="blk-53e7683ab12f">A new category of self-attention–based architectures, including Vision Transformers (ViTs), has emerged as a powerful alternative to convolutional neural networks for modelling global contextual relationships in images [<sup><xref ref-type="bibr" rid="ref-1b499b2a31ac">5</xref></sup>], while earlier deep generative models such as Deep Boltzmann Machines laid important foundations for hierarchical representation learning in deep neural architectures [<sup><xref ref-type="bibr" rid="ref-af0e8126d13d">12</xref></sup>]. Later improvements, including attention-based distillation and training data efficiency, have made them more useful when dealing with limited-data regimes typical of medical imaging [<sup><xref ref-type="bibr" rid="ref-e4debb87f107">6</xref></sup>]. Recent surveys and empirical experiments have indicated positive performances of ViTs at different medical image analysis tasks, such as breast cancer classification [<sup><xref ref-type="bibr" rid="ref-c27bab27efc4">8</xref></sup>,<sup><xref ref-type="bibr" rid="ref-262cc96f9f56">10</xref></sup>]. In spite of these developments, the majority of approaches based on transformers are trained in centralized environments, and their behaviour when applied to federated learning has not been adequately studied.</p>
      </sec>
      <sec id="sec-dec78f8ce96d">
        <title>Explainability and Clinical Interpretability</title>
        <p id="blk-35eb79ee7b6a">Explainable artificial intelligence (XAI) is a requirement for applying deep learning models to clinical practice, where transparency and trust are mandatory [<sup><xref ref-type="bibr" rid="ref-425f6742a018">13</xref></sup>]. Grad-CAM has become a popular gradient-based visualization method to offer spatial explanations to deep neural network predictions [<sup><xref ref-type="bibr" rid="ref-ed6be83c48b1">14</xref></sup>]. Recent papers have highlighted the importance of explainability in medical decision support systems, and it has been shown that explainable models can enhance clinician confidence and error analysis [<sup><xref ref-type="bibr" rid="ref-10222a7c52e0">15</xref></sup>]. However, explainability in a federated environment creates further complexity, since explanations need to be consistent when compared with the distribution of heterogeneous client data.</p>
      </sec>
      <sec id="sec-418700074e3f">
        <title>Breast Cancer Diagnosis and Generalization with other domains</title>
        <p id="blk-790b4165fab0">Detection of breast cancer has been studied widely using deep learning methods in both imaging modalities, with models produced showing high accuracy in a controlled setup [<sup><xref ref-type="bibr" rid="ref-c27bab27efc4">8</xref></sup>,<sup><xref ref-type="bibr" rid="ref-ddf859ae6406">16</xref></sup>]. Nevertheless, cross-domain generalization is also an ongoing problem as a result of changes in the imaging instructions and in the number of patients involved [<sup><xref ref-type="bibr" rid="ref-aa580804fd60">17</xref></sup>]. The recent reviews have noted the need to have federated and privacy-enabling structures that can be robust generalizations across the institutions without loss of diagnostic reliability [<sup><xref ref-type="bibr" rid="ref-648f94c2eb2c">18</xref></sup>,<sup><xref ref-type="bibr" rid="ref-74346807e08e">19</xref></sup>]. These findings encourage the construction of federated transformer-based methods that clearly focus on the issues of modality-based heterogeneity and clinical interpretability.</p>
      </sec>
    </sec>
    <sec id="sec-a61b953c6def">
      <title>Dataset description and preprocessing</title>
      <p id="blk-acbd339d7233">Experiments to evaluate the proposed federated learning framework under multi-institutional conditions were conducted using four publicly available breast imaging datasets: BreakHis, INbreast, CBIS-DDSM, and BUSI. These datasets represent heterogeneous imaging domains, including histopathology, mammography, and ultrasound, and are widely used benchmarks in breast cancer research [<sup><xref ref-type="bibr" rid="ref-6f67085b37d6">1</xref></sup>,<sup><xref ref-type="bibr" rid="ref-c27bab27efc4">8</xref></sup>,<sup><xref ref-type="bibr" rid="ref-aa580804fd60">17</xref></sup>].</p>
      <p id="blk-6385c7e74b25">The simulated federated learning environment represents data silos frequently seen in healthcare institutions by treating data as individual federated clients. This formulation causes a natural non-identically distributed (non- IID) data format, allowing the examination of the heterogeneity impacts on federated optimization to be controlled. In order to facilitate reproducibility and transparency, essential qualities of datasets, such as sample sizes, class distributions, labelling schemes, and units of data splitting, are summarized in <xref ref-type="table" rid="tbl-1"/>.</p>
      <p id="blk-ca6cf3e02993">When using the mammography data sets (INbreast and CBIS-DDSM), the image files in pre-processed ROI format that were provided publicly by the dataset maintainers were used directly. In this study, no actual raw DICOM-level windowing, bit-depth manipulation, or intensity clipping was done. The selected option will guarantee reproducibility, and yet note that pipeline optimization of mammographic preprocessing is out of the scope of the corresponding work.</p>
      <p id="blk-546d2412455a">All images across datasets were converted to a three-channel RGB format and resized to 224 × 224 pixels to ensure compatibility with the shared CNN – Vision Transformer backbone. Input normalization was performed using ImageNet statistics to facilitate transfer learning from pretrained weights and to maintain a unified preprocessing pipeline across heterogeneous imaging domains. While this normalization strategy simplifies cross-domain training, it may not fully preserve modality-specific intensity semantics, particularly for mammography and ultrasound, and is therefore acknowledged as a methodological limitation.</p>
      <p id="blk-8f28901e13fe">For the BreakHis histopathology dataset, no explicit stain normalization was applied. Although stain variability can influence generalization in histopathological analysis, this decision was made to preserve preprocessing consistency across datasets and to avoid introducing modality-specific transformations that could confound cross-domain federated learning behaviour. The absence of stain normalization is treated as a limitation of the current study.</p>
      <p id="blk-b0972513bb7a">For INBreast, the total dataset comprises 410 full mammographic images corresponding to 115 patients. However, for model training and evaluation, lesion-level or ROI-based patch extraction was performed, resulting in an expanded number of evaluation instances. Accordingly, the reported test sample count in <xref ref-type="table" rid="tbl-3"/> reflects patch-level evaluation units rather than original full-image counts. All references to “samples” in performance tables hence denote model-level input instances (patches) rather than raw full images.</p>
      <p id="blk-e26744e12996">During federated training, no raw data were exchanged between clients, and only model parameters were shared with the central server, in accordance with federated learning principles [<sup><xref ref-type="bibr" rid="ref-707ad2879e3f">2</xref></sup>,<sup><xref ref-type="bibr" rid="ref-08a1a08f747f">9</xref></sup>]. Model evaluation was performed on a pooled test set constructed from client-specific held-out splits to assess global diagnostic performance across heterogeneous imaging sources. The overall data preparation and preprocessing workflow is illustrated in <xref ref-type="fig" rid="fig-1"/>.</p>
      <p id="blk-e216f9c0d409">Beyond class imbalance, inter-client feature distribution heterogeneity was qualitatively assessed through embedding-space visualization of penultimate-layer representations using t-SNE. Distinct clustering patterns across datasets indicate measurable feature-space divergence between clients, supporting the characterization of the setting as statistically non-identically distributed beyond simple label skew.</p>
      <p id="blk-1e5b180323a3">As shown in <xref ref-type="table" rid="tbl-2"/>, malignant prevalence varies from 37.56% to 50.06% across federated clients. The BUSI dataset exhibits a pronounced benign skew (majority/minority ratio = 1.662), whereas the remaining datasets are nearly balanced. This quantified label heterogeneity provides empirical justification for modelling the federated setting as non-identically distributed (non-IID).</p>
    </sec>
    <sec id="sec-7dcfe02ae0fa">
      <title>Proposed Federated Vision Transformer Framework</title>
      <p id="blk-67f260672b9b"><xref ref-type="fig" rid="fig-2"/> illustrates the proposed federated hybrid CNN – Vision Transformer (CNN – ViT) framework, which is employed consistently in both centralized and federated learning settings. The architecture follows a dual-branch design, where a convolutional neural network (ResNet-18) extracts localized spatial features, and a Vision Transformer (ViT) branch captures global contextual representations. Features from both branches are concatenated and passed to a shared classification head, enabling complementary modelling of fine-grained and holistic tissue characteristics.</p>
      <p id="blk-512d0d42bd15">In the federated learning simulation, each imaging dataset is treated as an independent client, reflecting realistic data silos across institutions or acquisition sources. Model training is performed locally at each client, while collaboration is achieved through parameter aggregation without exchanging raw imaging data, thereby adhering to federated learning principles.</p>
      <sec id="sec-36a8370bc628">
        <title>Federated Learning Formulation</title>
        <p id="blk-440a946d3adb">Let K=1,2,…,Kdenote the set of federated clients, where K=4in this study. Each client corresponds to one independent publicly available dataset: BreakHis (histopathology), INbreast (mammography), CBIS-DDSM (mammography), and BUSI (ultrasound). Clients are therefore defined at the dataset level rather than strictly at the imaging modality level. Although imaging modalities partially overlap, both INbreast and CBIS-DDSM are mammography datasets; they are treated as separate clients due to differences in acquisition protocols, case distributions, annotation procedures, and preprocessing characteristics. Therefore, clients are defined at the dataset level rather than strictly at the modality level.</p>
        <p id="blk-b5930e00e16d">Each client <inline-formula><alternatives><tex-math id="tm-1">\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym}
\usepackage{amsfonts}
\usepackage{amssymb}
\usepackage{amsbsy}
\usepackage{mathrsfs}
\usepackage{upgreek}
\setlength{\oddsidemargin}{-69pt}
\begin{document}$k \in K$\end{document}</tex-math><mml:math display="inline" id="mml-1"><mml:mrow><mml:mi>k</mml:mi><mml:mo>∈</mml:mo><mml:mi>K</mml:mi></mml:mrow></mml:math></alternatives></inline-formula> possesses a local dataset <inline-formula><alternatives><tex-math id="tm-2">\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym}
\usepackage{amsfonts}
\usepackage{amssymb}
\usepackage{amsbsy}
\usepackage{mathrsfs}
\usepackage{upgreek}
\setlength{\oddsidemargin}{-69pt}
\begin{document}$D_k$\end{document}</tex-math><mml:math display="inline" id="mml-2"><mml:mrow><mml:msub><mml:mi>D</mml:mi><mml:mi>k</mml:mi></mml:msub></mml:mrow></mml:math></alternatives></inline-formula>, which differs in size, imaging modality, class distribution, and feature characteristics from other clients. These variations introduce statistical heterogeneity and label skew across clients, resulting in a non-identically distributed (non-IID) federated learning setting [<sup><xref ref-type="bibr" rid="ref-8aa40c20886c">3</xref></sup>,<sup><xref ref-type="bibr" rid="ref-3495eb21e423">4</xref></sup>,<sup><xref ref-type="bibr" rid="ref-707ad2879e3f">2</xref></sup>].</p>
        <p id="blk-bca6f3b2cc6d">Local model training is performed independently at each client using only its private dataset D_k. During federated optimization, no raw imaging data are transmitted between clients or to the central server. Instead, only model parameter updates are communicated for aggregation. This design preserves dataset isolation within the simulated federated environment while enabling collaborative model learning.</p>
      </sec>
      <sec id="sec-b0d16b5e7173">
        <title>Hybrid Vision Transformer Backbone</title>
        <p id="blk-bfda9e0a370f">Local model: A local model of the same hybrid CNN-ViT architecture is used by all clients. The trained CNN branch (ResNet-18) that is pre-trained using ImageNet weights captures local texture and morphological features, based on patterns of breast tissue. Simultaneously, the Vision Transformer arm works on resized images 224 x 224RGB and splits them into fixed-size patches, then serializing and linearly embedding them before processing through stacked self- attention layers [<sup><xref ref-type="bibr" rid="ref-1b499b2a31ac">5</xref></sup>,<sup><xref ref-type="bibr" rid="ref-e4debb87f107">6</xref></sup>].</p>
        <p id="blk-f221c79efedd">ViT has a branch, referred to as ViT, that allows the long-range spatial dependency and global contextual relations to be modelled, and this aspect is especially significant in heterogeneous breast cancer imaging. The CNN and ViT branches’ outputs are combined with the feature level and launched through a simple fully connected classifier that makes binary decisions (benign vs. malignant).</p>
      </sec>
      <sec id="sec-299f832eff13">
        <title>Extrinsic Optimization and Aggregation</title>
        <p id="blk-e2554cb5883e">Training is done through several communication rounds. This involves each client updating its local model parameters suitably using stochastic gradient descent with its own private dataset in every round. The model updates are transmitted to the central server after local training, and then there is data-size-weighted federated averaging (FedAvg) to calculate the global model update [<sup><xref ref-type="bibr" rid="ref-707ad2879e3f">2</xref></sup>]. In particular, the contribution of each client is apportioned equally to the size of the local dataset in its possession, i.e., in the case of heterogeneous client counts, there is no bias in aggregate.</p>
        <p id="blk-7de87cd2b5cd">The new model developed via globalization is then reissued to all customers in the face of the second wave of local training. The procedure is iterative and allows collaborative learning without interference between datasets in the simulated federated structure.</p>
      </sec>
      <sec id="sec-d4e7bdbaec3a">
        <title>Integration of Explainability</title>
        <p id="blk-c24c2a9e96b7">To improve clarification and medical interest, clarification procedures are incorporated into the suggested environment. In the case of the CNN branch, class-discriminative areas of activation are visualized through the application of Grad-CAM. In the case of the Vision Transformer branch, attention-based attribution (attention rollout) is the mechanism that checks the spatial regions with the highest contribution to the model predictions.</p>
        <p id="blk-ba747ccc02be">These attribution maps are produced after post hoc and normed to be visualized, and a qualitative evaluation can be done regarding the compatibility of model attention and clinically relevant tissue structures. Explainability is used uniformly across clients and assessed on held-out test samples to aid in the transparent analysis of model behaviour [<sup><xref ref-type="bibr" rid="ref-425f6742a018">13</xref></sup>,<sup><xref ref-type="bibr" rid="ref-ed6be83c48b1">14</xref></sup>].</p>
        <p id="blk-7d6241e54ff7"><bold>Input:</bold> Federated client datasets <inline-formula><alternatives><tex-math id="tm-3">\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym}
\usepackage{amsfonts}
\usepackage{amssymb}
\usepackage{amsbsy}
\usepackage{mathrsfs}
\usepackage{upgreek}
\setlength{\oddsidemargin}{-69pt}
\begin{document}$\{D_k\}_{k=1}^{K},$\end{document}</tex-math><mml:math display="inline" id="mml-3"><mml:mrow><mml:mo stretchy="false">{</mml:mo><mml:msub><mml:mi>D</mml:mi><mml:mi>k</mml:mi></mml:msub><mml:msubsup><mml:mo stretchy="false">}</mml:mo><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>K</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo></mml:mrow></mml:math></alternatives></inline-formula> where each client corresponds to one independent dataset (BreakHis, INbreast, CBIS-DDSM, BUSI), representing dataset-level data silos rather than modality-exclusive partitions. Number of communications rounds <inline-formula><alternatives><tex-math id="tm-4">\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym}
\usepackage{amsfonts}
\usepackage{amssymb}
\usepackage{amsbsy}
\usepackage{mathrsfs}
\usepackage{upgreek}
\setlength{\oddsidemargin}{-69pt}
\begin{document}$R$\end{document}</tex-math><mml:math display="inline" id="mml-4"><mml:mrow><mml:mi>R</mml:mi></mml:mrow></mml:math></alternatives></inline-formula> Local batch size <inline-formula><alternatives><tex-math id="tm-5">\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym}
\usepackage{amsfonts}
\usepackage{amssymb}
\usepackage{amsbsy}
\usepackage{mathrsfs}
\usepackage{upgreek}
\setlength{\oddsidemargin}{-69pt}
\begin{document}$B$\end{document}</tex-math><mml:math display="inline" id="mml-5"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:math></alternatives></inline-formula> Vision Transformer model <inline-formula><alternatives><tex-math id="tm-6">\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym}
\usepackage{amsfonts}
\usepackage{amssymb}
\usepackage{amsbsy}
\usepackage{mathrsfs}
\usepackage{upgreek}
\setlength{\oddsidemargin}{-69pt}
\begin{document}$f_{\theta}$\end{document}</tex-math><mml:math display="inline" id="mml-6"><mml:mrow><mml:msub><mml:mi>f</mml:mi><mml:mrow><mml:mi>θ</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:math></alternatives></inline-formula> <bold>Output:</bold> Trained a global model <inline-formula><alternatives><tex-math id="tm-7">\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym}
\usepackage{amsfonts}
\usepackage{amssymb}
\usepackage{amsbsy}
\usepackage{mathrsfs}
\usepackage{upgreek}
\setlength{\oddsidemargin}{-69pt}
\begin{document}$f_{\theta^*}$\end{document}</tex-math><mml:math display="inline" id="mml-7"><mml:mrow><mml:msub><mml:mi>f</mml:mi><mml:mrow><mml:msup><mml:mi>θ</mml:mi><mml:mo>*</mml:mo></mml:msup></mml:mrow></mml:msub></mml:mrow></mml:math></alternatives></inline-formula>for breast cancer classification <bold>Step 1:</bold> Initialize global model parameters <inline-formula><alternatives><tex-math id="tm-8">\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym}
\usepackage{amsfonts}
\usepackage{amssymb}
\usepackage{amsbsy}
\usepackage{mathrsfs}
\usepackage{upgreek}
\setlength{\oddsidemargin}{-69pt}
\begin{document}$\theta^{(0)}$\end{document}</tex-math><mml:math display="inline" id="mml-8"><mml:mrow><mml:msup><mml:mi>θ</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mn>0</mml:mn><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msup></mml:mrow></mml:math></alternatives></inline-formula>at the central server. <bold>Step 2:</bold> For each communication round <inline-formula><alternatives><tex-math id="tm-9">\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym}
\usepackage{amsfonts}
\usepackage{amssymb}
\usepackage{amsbsy}
\usepackage{mathrsfs}
\usepackage{upgreek}
\setlength{\oddsidemargin}{-69pt}
\begin{document}$r=1,2,\ldots,R$\end{document}</tex-math><mml:math display="inline" id="mml-9"><mml:mrow><mml:mi>r</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mi>…</mml:mi><mml:mo>,</mml:mo><mml:mi>R</mml:mi></mml:mrow></mml:math></alternatives></inline-formula>, perform: Client Selection: All available clients <inline-formula><alternatives><tex-math id="tm-10">\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym}
\usepackage{amsfonts}
\usepackage{amssymb}
\usepackage{amsbsy}
\usepackage{mathrsfs}
\usepackage{upgreek}
\setlength{\oddsidemargin}{-69pt}
\begin{document}$k\in\{1,\ldots,K\}$\end{document}</tex-math><mml:math display="inline" id="mml-10"><mml:mrow><mml:mi>k</mml:mi><mml:mo>∈</mml:mo><mml:mo stretchy="false">{</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mi>…</mml:mi><mml:mo>,</mml:mo><mml:mi>K</mml:mi><mml:mo stretchy="false">}</mml:mo></mml:mrow></mml:math></alternatives></inline-formula>participate in the current round. Local Model Update: For each client <inline-formula><alternatives><tex-math id="tm-11">\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym}
\usepackage{amsfonts}
\usepackage{amssymb}
\usepackage{amsbsy}
\usepackage{mathrsfs}
\usepackage{upgreek}
\setlength{\oddsidemargin}{-69pt}
\begin{document}$k$\end{document}</tex-math><mml:math display="inline" id="mml-11"><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:math></alternatives></inline-formula>: Receive global parameters <inline-formula><alternatives><tex-math id="tm-12">\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym}
\usepackage{amsfonts}
\usepackage{amssymb}
\usepackage{amsbsy}
\usepackage{mathrsfs}
\usepackage{upgreek}
\setlength{\oddsidemargin}{-69pt}
\begin{document}$\theta^{(r-1)}$\end{document}</tex-math><mml:math display="inline" id="mml-12"><mml:mrow><mml:msup><mml:mi>θ</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>r</mml:mi><mml:mo>−</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msup></mml:mrow></mml:math></alternatives></inline-formula>. Initialize local model <inline-formula><alternatives><tex-math id="tm-13">\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym}
\usepackage{amsfonts}
\usepackage{amssymb}
\usepackage{amsbsy}
\usepackage{mathrsfs}
\usepackage{upgreek}
\setlength{\oddsidemargin}{-69pt}
\begin{document}$f_{\theta_k^{(r)}}\leftarrow f_{\theta^{(r-1)}}$\end{document}</tex-math><mml:math display="inline" id="mml-13"><mml:mrow><mml:msub><mml:mi>f</mml:mi><mml:mrow><mml:msubsup><mml:mi>θ</mml:mi><mml:mi>k</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>r</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msubsup></mml:mrow></mml:msub><mml:mo>←</mml:mo><mml:msub><mml:mi>f</mml:mi><mml:mrow><mml:msup><mml:mi>θ</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>r</mml:mi><mml:mo>−</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msup></mml:mrow></mml:msub></mml:mrow></mml:math></alternatives></inline-formula>. Train <inline-formula><alternatives><tex-math id="tm-14">\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym}
\usepackage{amsfonts}
\usepackage{amssymb}
\usepackage{amsbsy}
\usepackage{mathrsfs}
\usepackage{upgreek}
\setlength{\oddsidemargin}{-69pt}
\begin{document}$f_{\theta_k^{(r)}}$\end{document}</tex-math><mml:math display="inline" id="mml-14"><mml:mrow><mml:msub><mml:mi>f</mml:mi><mml:mrow><mml:msubsup><mml:mi>θ</mml:mi><mml:mi>k</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>r</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msubsup></mml:mrow></mml:msub></mml:mrow></mml:math></alternatives></inline-formula>on the local dataset <inline-formula><alternatives><tex-math id="tm-15">\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym}
\usepackage{amsfonts}
\usepackage{amssymb}
\usepackage{amsbsy}
\usepackage{mathrsfs}
\usepackage{upgreek}
\setlength{\oddsidemargin}{-69pt}
\begin{document}$D_k$\end{document}</tex-math><mml:math display="inline" id="mml-15"><mml:mrow><mml:msub><mml:mi>D</mml:mi><mml:mi>k</mml:mi></mml:msub></mml:mrow></mml:math></alternatives></inline-formula>for one epoch using cross-entropy loss and stochastic gradient descent. Transmit updated parameters <inline-formula><alternatives><tex-math id="tm-16">\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym}
\usepackage{amsfonts}
\usepackage{amssymb}
\usepackage{amsbsy}
\usepackage{mathrsfs}
\usepackage{upgreek}
\setlength{\oddsidemargin}{-69pt}
\begin{document}$\theta_k^{(r)}$\end{document}</tex-math><mml:math display="inline" id="mml-16"><mml:mrow><mml:msubsup><mml:mi>θ</mml:mi><mml:mi>k</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>r</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msubsup></mml:mrow></mml:math></alternatives></inline-formula>to the server. Model Aggregation (FedAvg): Aggregate local updates to obtain the global model: <inline-formula><alternatives><tex-math id="tm-17">\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym}
\usepackage{amsfonts}
\usepackage{amssymb}
\usepackage{amsbsy}
\usepackage{mathrsfs}
\usepackage{upgreek}
\setlength{\oddsidemargin}{-69pt}
\begin{document}$\theta^{(r)}\leftarrow \frac{1}{K}\sum_{k=1}^{K}\theta_k^{(r)}$\end{document}</tex-math><mml:math display="inline" id="mml-17"><mml:mrow><mml:msup><mml:mi>θ</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>r</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msup><mml:mo>←</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>K</mml:mi></mml:mrow></mml:mfrac><mml:msubsup><mml:mo>∑</mml:mo><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>K</mml:mi></mml:mrow></mml:msubsup><mml:msubsup><mml:mi>θ</mml:mi><mml:mi>k</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>r</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msubsup></mml:mrow></mml:math></alternatives></inline-formula> Global Evaluation: Evaluate the aggregated model <inline-formula><alternatives><tex-math id="tm-18">\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym}
\usepackage{amsfonts}
\usepackage{amssymb}
\usepackage{amsbsy}
\usepackage{mathrsfs}
\usepackage{upgreek}
\setlength{\oddsidemargin}{-69pt}
\begin{document}$f_{\theta^{(r)}}$\end{document}</tex-math><mml:math display="inline" id="mml-18"><mml:mrow><mml:msub><mml:mi>f</mml:mi><mml:mrow><mml:msup><mml:mi>θ</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>r</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msup></mml:mrow></mml:msub></mml:mrow></mml:math></alternatives></inline-formula>on the global validation set and record performance metrics (Accuracy, AUC, F1-score). <bold>Step 3:</bold> After completing <inline-formula><alternatives><tex-math id="tm-19">\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym}
\usepackage{amsfonts}
\usepackage{amssymb}
\usepackage{amsbsy}
\usepackage{mathrsfs}
\usepackage{upgreek}
\setlength{\oddsidemargin}{-69pt}
\begin{document}$R$\end{document}</tex-math><mml:math display="inline" id="mml-19"><mml:mrow><mml:mi>R</mml:mi></mml:mrow></mml:math></alternatives></inline-formula>rounds, select the final global model <inline-formula><alternatives><tex-math id="tm-20">\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym}
\usepackage{amsfonts}
\usepackage{amssymb}
\usepackage{amsbsy}
\usepackage{mathrsfs}
\usepackage{upgreek}
\setlength{\oddsidemargin}{-69pt}
\begin{document}$f_{\theta^*}=f_{\theta^{(R)}}$\end{document}</tex-math><mml:math display="inline" id="mml-20"><mml:mrow><mml:msub><mml:mi>f</mml:mi><mml:mrow><mml:msup><mml:mi>θ</mml:mi><mml:mo>*</mml:mo></mml:msup></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mi>f</mml:mi><mml:mrow><mml:msup><mml:mi>θ</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>R</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msup></mml:mrow></mml:msub></mml:mrow></mml:math></alternatives></inline-formula>. <bold>Step 4:</bold> Perform final clinical evaluation on the aggregated test set and generate explainability maps using gradient-based attribution. <bold>Return:</bold> Final trained model <inline-formula><alternatives><tex-math id="tm-21">\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym}
\usepackage{amsfonts}
\usepackage{amssymb}
\usepackage{amsbsy}
\usepackage{mathrsfs}
\usepackage{upgreek}
\setlength{\oddsidemargin}{-69pt}
\begin{document}$f_{\theta^*}$\end{document}</tex-math><mml:math display="inline" id="mml-21"><mml:mrow><mml:msub><mml:mi>f</mml:mi><mml:mrow><mml:msup><mml:mi>θ</mml:mi><mml:mo>*</mml:mo></mml:msup></mml:mrow></mml:msub></mml:mrow></mml:math></alternatives></inline-formula>and associated evaluation metrics.</p>
      </sec>
      <sec id="sec-b594035a435d">
        <title>Experimental Setup / Training Configuration</title>
        <p id="blk-30a301623c93">The hybrid CNN – Vision Transformer architecture comprises approximately 36 million trainable parameters based on the implemented configuration. Federated optimization was conducted over 10 communication rounds, with 5 local training epochs per client per round using data-size-weighted Federated Averaging (FedAvg). The initial learning rate was set to <inline-formula><alternatives><tex-math id="tm-22">\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym}
\usepackage{amsfonts}
\usepackage{amssymb}
\usepackage{amsbsy}
\usepackage{mathrsfs}
\usepackage{upgreek}
\setlength{\oddsidemargin}{-69pt}
\begin{document}$2 \times 10^{-4}$\end{document}</tex-math><mml:math display="inline" id="mml-22"><mml:mrow><mml:mn>2</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mn>10</mml:mn><mml:mrow><mml:mo>−</mml:mo><mml:mn>4</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:math></alternatives></inline-formula> using the Adam optimizer with a batch size of 32. No learning-rate scheduling or hyperparameter search was performed.</p>
        <p id="blk-443bf81e7aac">The serialized global model size is approximately 140 MB in FP32 precision, representing the upload and download communication cost per client per round. No model compression, pruning, quantization, or communication-efficient strategies were applied; therefore, the reported communication overhead reflects a standard uncompressed federated setup.</p>
        <p id="blk-566cfc8031c9">Wall-clock training time was not systematically recorded in the present study. Given the fixed hardware configuration (NVIDIA Tesla T4, 16 GB VRAM), the reported results should therefore be interpreted primarily in terms of algorithmic feasibility rather than deployment-level computational benchmarking. Detailed runtime profiling and communication-latency analysis remain important directions for future work.</p>
        <p id="blk-2ec63c946e35">All experiments were conducted using Python 3.11, PyTorch 2.3, torchvision 0.18, CUDA 12.0, and cuDNN 8.9 under Ubuntu 22.04 LTS. A fixed random seed (42) was used for data splitting, weight initialization, and optimization to ensure deterministic reproducibility. These details are reported to ensure methodological transparency and reproducibility of the simulated federated learning setup.</p>
      </sec>
    </sec>
    <sec id="sec-89f32112cf45">
      <title>Experimental Results and Analysis</title>
      <p id="blk-cde3fb52be67">This section presents a structured evaluation of the proposed federated hybrid CNN – Vision Transformer framework under a simulated multi-institutional, non-identically distributed (non-IID) learning setting. Four heterogeneous breast imaging dataset BreakHis, INbreast, CBIS-DDSM, and BUSI, were treated as independent dataset-level federated clients, as described in Section III.</p>
      <sec id="sec-026f00c7bc61">
        <title>Training Configuration and Data Processing</title>
        <p id="blk-85ffbb6970d6">All input images were resized to 224 × 224 pixels and normalized using ImageNet statistics to maintain compatibility with pretrained backbones. The ResNet-18 branch was initialized with ImageNet weights, while the Vision Transformer branch employed patch-based embeddings with learnable positional encodings. Data augmentation was limited to random horizontal flips and minor intensity normalization to reduce overfitting while preserving diagnostic characteristics.</p>
        <p id="blk-a4c9232492fa">Each client trained its local model using the Adam optimizer with an initial learning rate of <inline-formula><alternatives><tex-math id="tm-23">\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym}
\usepackage{amsfonts}
\usepackage{amssymb}
\usepackage{amsbsy}
\usepackage{mathrsfs}
\usepackage{upgreek}
\setlength{\oddsidemargin}{-69pt}
\begin{document}$2 \times 10^{-4}$\end{document}</tex-math><mml:math display="inline" id="mml-23"><mml:mrow><mml:mn>2</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mn>10</mml:mn><mml:mrow><mml:mo>−</mml:mo><mml:mn>4</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:math></alternatives></inline-formula> and batch size 32. Binary cross-entropy loss was applied for classification. No client-specific hyperparameter tuning was performed, ensuring consistency across heterogeneous datasets.</p>
        <p id="blk-455a08c7f853">Federated training consisted of 10 global communication rounds, each containing 5 local epochs per participating client. Data-size-weighted Federated Averaging (FedAvg) was used for model aggregation, proportionally weighting client updates based on local dataset size.</p>
        <p id="blk-de4b8111a6ab">Raw imaging data remained local to each client throughout training. Only model parameters were transmitted to the central server. Evaluation was performed on client-specific held-out test sets, followed by pooled analysis across clients to estimate aggregated performance. This pooled evaluation was conducted solely for experimental benchmarking and does not reflect a real-world federated deployment scenario.</p>
      </sec>
      <sec id="sec-31dab17ad7df">
        <title>Federated Training Performance and Convergence Analysis</title>
        <p id="blk-d0aee204444b">The federated hybrid CNN-Vision Transformer network was trained in ten communicative rounds in a simulated non-identically distribu­ted(non-IID) scenario. <xref ref-type="fig" rid="fig-3"/> demonstrates how the accuracy, F1-score, and ROC-AUC of the world change with the communication round. The global model is showing a steady performance improvement pattern, whereby consistency in finding accuracy rises in each round of performance by starting with 57.9 percent, then 72.4 percent in the first and last round, respectively, accompanied by a rise in ROC-AUC between 0.71 and 0.75.</p>
        <p id="blk-cc5501d4fed0">Intermediate fluctuations are used, which are attributes of optimization of federated learning with heterogeneous client distributions and indicate the variability of local updates instead of instability in the learning process. Notwithstanding such fluctuations, the world model continues to recover again and again and become better, suggesting that there is solid convergence behavior of cross-modality heterogeneity. These findings confirm that the proposed framework could utilize the knowledge of heterogeneous sources of imaging without centralized access to this data, even when the distribution of clients is non-IID.All reported metrics represent single-seed estimates; variance across multiple random initializations was not evaluated.</p>
      </sec>
      <sec id="sec-a4d050ffcd5f">
        <title>Global Clinical Classification Performance</title>
        <p id="blk-5e697bd0e63c">The final global model was evaluated on the aggregated test cohort constructed from client-specific held-out splits. The confusion matrix results are presented in <xref ref-type="fig" rid="fig-4"/>. The model achieved an overall classification accuracy of approximately 82.5%, with a weighted F1-score of 0.81 and a macro F1-score of 0.78. Notably, the framework demonstrated a high malignant recall rate of approximately 96.4%, with only 65 false-negative cases among malignant samples. This high sensitivity is clinically significant, as early breast cancer screening prioritizes minimizing missed malignant detections.</p>
        <p id="blk-e05d238453c2">Conversely, the benign recall is relatively less (64.8%), and this denotes a sensitivity-specificity trade-off under a constant decision level. Although this action leads to more false-positive referrals, it gives more emphasis to malignant detection, which is usually desired in early screening settings. No threshold optimization or cost-sensitive calibration was applied in this study; therefore, this trade-off is reported as an empirical outcome rather than a designed diagnostic bias.</p>
      </sec>
      <sec id="sec-91302fd01a3e">
        <title>Per-Client Performance Analysis</title>
        <p id="blk-fa3d44a191fd">To evaluate robustness under heterogeneous federated conditions, the final global model was additionally assessed on each client’s held-out test set separately. Dataset-level performance metrics, including Accuracy, AUC, F1-score, Sensitivity, and Specificity, are summarized in <xref ref-type="table" rid="tbl-3"/>, while a visual comparison of client-wise model performance across heterogeneous datasets is presented in <xref ref-type="fig" rid="fig-5"/>. Reporting per-client performance provides a more transparent assessment of cross-modality generalization under non-IID conditions, beyond pooled global metrics.</p>
        <p id="blk-d260c5f2c3a2">As shown in <xref ref-type="table" rid="tbl-3"/>, substantial performance variability is observed across clients, reflecting the challenges of cross-domain heterogeneity in federated learning. The model achieved excellent diagnostic performance for the BreakHis and INbreast datasets, with AUC values of 0.987 and 0.997, respectively, indicating strong generalization across histopathological and mammographic imaging domains. In contrast, comparatively lower performance was observed for the CBIS-DDSM and BUSI datasets, where AUC values were approximately 0.585 and 0.546. This discrepancy can be attributed to modality-specific distribution shifts, limited sample sizes, and differences in imaging acquisition characteristics. Importantly, malignant sensitivity remained relatively high across all clients, demonstrating the framework’s ability to prioritize cancer detection under heterogeneous non-IID conditions.</p>
        <p id="blk-5eaac61cba6a">These findings highlight that federated learning models trained across highly heterogeneous medical imaging domains may exhibit modality-dependent performance variations, emphasizing the need for domain-aware calibration strategies in future work.</p>
      </sec>
      <sec id="sec-4dd3fbb9a932">
        <title>Explainability and Error Analysis</title>
        <p id="blk-a0436c1c349d">To assess interpretability, qualitative explainability maps were generated for representative test samples, as shown in <xref ref-type="fig" rid="fig-6"/>. The visualization includes examples of true positive, false positive, and false negative predictions.</p>
        <p id="blk-6bb5d6ec67f6">This is because, in cases of rightfully categorized malignant cases, the model predominantly focuses on the dense cellular areas and abnormal tissue formations, which are known histopathological signs of cancer. The patterns of attention in false-negative and false-positive cases are diffuse or unclear and draw greater focus to visually subtle areas that can make diagnostic decisions difficult.</p>
        <p id="blk-3929c3887818">These explainability results present a qualitative interpretation that the model is using meaningful image parts to operate its predictions and not spurious artifacts. But there was no quantitative faithfulness/consistency test (e.g., deletion/insertion tests or a clinician validation). In this sense, the explainability analysis will be used to contribute to the interpretability on an illustrative level, and systematic validation will become a valuable direction to work in the future.</p>
      </sec>
    </sec>
    <sec id="sec-7eb98fcceb07">
      <title>Comparative analysis</title>
      <p id="blk-be4cbe9c709d">The proposed framework is conceptually informed by the hybrid explainable federated Vision Transformer approach introduced by Al-Hejri et al. [<sup><xref ref-type="bibr" rid="ref-6f67085b37d6">1</xref></sup>], which demonstrated the viability of integrating transformer-based architectures within a federated learning paradigm for breast cancer classification. However, the scope, architectural composition, and evaluation objectives of the two studies differ substantially.</p>
      <p id="blk-5a213c5df214">While Al-Hejri et al. [<sup><xref ref-type="bibr" rid="ref-6f67085b37d6">1</xref></sup>] primarily focused on establishing the feasibility of federated transformer learning with integrated explainability under privacy-aware settings, the present study explicitly investigates cross-modality heterogeneity by treating distinct imaging datasets (histopathology, mammography, and ultrasound) as independent federated clients. Notably, even within the mammography domain, INbreast and CBIS-DDSM are modelled as separate clients due to differences in acquisition protocols, case distributions, and preprocessing characteristics. This formulation produces a more pronounced non-identically distributed (non-IID) setting, enabling analysis of convergence dynamics under modality-driven distribution shifts.</p>
      <p id="blk-41e18efa68e0">In contrast to prior work that predominantly reports aggregate performance metrics, the present study additionally provides round-wise convergence trajectories, offering insight into optimization stability and recovery behaviour across communication rounds under heterogeneous client conditions. Furthermore, error analysis highlights strong malignant sensitivity, a clinically relevant characteristic in screening-oriented diagnostic systems.</p>
      <p id="blk-1349543290c5">Importantly, this comparison remains qualitative in nature. No direct head-to-head baseline experiments, such as centralized CNN, centralized ViT, or alternative federated optimization methods (e.g., FedProx or SCAFFOLD) were conducted using identical dataset splits. Therefore, the proposed framework should be interpreted as demonstrating the feasibility of federated hybrid CNN – ViT training under simulated multi-domain non-IID conditions, rather than establishing performance superiority over existing methods. Comprehensive benchmark comparisons under standardized experimental settings are identified as future work.</p>
    </sec>
    <sec id="sec-7ff385ea9355">
      <title>Discussion</title>
      <p id="blk-e1bc107c9250">In this paper, a federated micro-hybrid CNN-Vision Transformer model is tested in a multi-source non- identically distributed (non-IID) simulation-based environment by using publicly accessible datasets of breast imaging data. The data sets are treated as independent federated clients, which allows for gaining controlled insight into the cross-domain heterogeneity and federated convergence behaviour. Nonetheless, this experimental design fails to completely reflect the federated learning constraints of the real world, such as site-specific governance policies, secure aggregation protocols, asynchronous client participation, client dropouts, and communication variability. The findings published, therefore, should be viewed as test evidence of methodological support and not clinical implementation readiness. Though heterogeneous imaging domains -histopathology, mammography, and ultrasound are modelled jointly with the use of a common binary classification head, such pooled modelling brings significant threats to validity.</p>
      <p id="blk-5cc33485b10c">Acquisitional variabilities that may cause domain shift, label harmonisation ambiguity, and disparate operating points across modalities may arise because of differences in acquisition protocols, contrast mechanisms, and annotation conventions. In practice in clinical settings, these difficulties may necessitate the use of modality-sensitive calibration policies, domain-relevant decision thresholds, distinct classification heads, or explicit domain adaptation policies so that there can be consistency in the diagnostic behaviour of imaging sources. Demographic metadata (e.g., age, ethnicity, or acquisition site information) were not consistently available across all publicly accessible datasets; therefore, subgroup fairness analysis was not conducted. Potential performance disparities across demographic or acquisition-based subgroups cannot be excluded. Future validation in multi-center settings with structured demographic reporting would be necessary to evaluate the fairness, equity, and generalizability of the proposed framework. The federated learning application to this paper must be regarded as the privacy-conscious architecture paradigm and not the officially confirmed privacy-saving algorithm. Raw data do not leave the client during training, and actually, no explicit privacy threat model, e.g., membership inference or gradient leaks, is analyzed, and no privacy-ensuring mechanisms (e.g., secure aggregation or differential privacy) are enforced.</p>
      <p id="blk-4e66228d5196">Moreover, model assessment is done on merged test data to be used in performance benchmarking, but in practice, real-world federated deployments are based on local evaluation or trusted third-party evaluation, and this may impact the reported performance. All experiments were conducted using a fixed random seed (42) to ensure deterministic reproducibility of data splits, weight initialization, and optimization trajectories. Variability across multiple random initializations was not evaluated in the present study. Consequently, reported performance metrics should be interpreted as single-seed estimates rather than variance-adjusted confidence intervals. Multi-seed stability analysis is recommended as an important future validation step to strengthen robustness claims. Analysis of explanations offers the qualitative depictions of the attentions of models to correctly and incorrectly classified cases, indicating that the framework is attentive to diagnostically significant tissue areas. Nonetheless, these results cannot be sustained by quantitative measures of faithfulness, systematic expert validation, and cross-client consistency measures. Interpretability results must therefore be considered to be illustrative, and systematic validation is a field where future research should focus. Furthermore, leave-one-client-out cross-domain generalization experiments, in which the federated model is trained on three clients and evaluated on a held-out client, were not conducted in the present study and remain an important direction for future validation of cross-institutional robustness.</p>
    </sec>
    <sec id="sec-899d3e0f6874">
      <title>Conclusion and future work</title>
      <p id="blk-bbf8bc7cde27">This paper introduces a federated hybrid CNN-Vision Transformer system for breast cancer classification subjected to simulated non-IID and multi-source experiments. The study proves that it is possible to collaboratively learn across domains without sharing data centrally by considering heterogeneous imaging datasets as autonomous federated customers. The results obtained with the experiments indicate consistent convergence under the evaluated configuration and high malignant sensitivity in the aggregated test cohort, with performance variability observed across individual clients. In addition to predictive performance, qualitative explainability studies depict the manner in which the model serves diagnostically significant image areas, which promotes interpretability at the level of exploration.</p>
      <p id="blk-f927ca840478">Nevertheless, the results are to be viewed as a proof-of-concept instead of a clinically ready one, due to the utilization of publicly available datasets, centralized analysis, and the lack of a formal privacy or calibration analysis. Future directions will involve research in validation of the framework in realistic multi-center systems, including modality-aware optimization of the system, modality-aware aggregation, running federated optimizers, and doing systematic fairness and interpretability across oneself and other deeper optimization frameworks. These findings provide a structured experimental foundation for subsequent multi-center validation, statistical robustness analysis, and privacy-certified deployment studies required for clinical translation. Cross-domain robustness under leave-one-client-out evaluation was not assessed and remains an important direction for future validation.</p>
    </sec>
  </body>
  <back>
    <ack>
      <title>Acknowledgments</title>
      <p>None.</p>
    </ack>
    <sec sec-type="ethics-statement">
      <title>Institutional Review Board (IRB)</title>
      <p>This study involves secondary analysis of publicly available breast imaging datasets and does not include human subject recruitment, intervention, or access to identifiable patient information. All datasets were used in accordance with their respective licensing agreements and publicly documented usage policies. Therefore, institutional review board (IRB) approval was not required.</p>
    </sec>
    <sec sec-type="ai-statement">
      <title>Large Language Model</title>
      <p>Generative artificial intelligence tools were used in a limited and supportive role during manuscript preparation. Literature searches to identify relevant background articles were performed using OpenEvidence (accessed December 2024 and January 2025). ChatGPT (OpenAI; accessed December 2024 and January 2025) was used to assist with language refinement, grammar, clarity, and formatting of the manuscript. Neither tool was used to generate original scientific content, interpret data, perform analyses, draw conclusions, or make clinical judgments. All content was reviewed, verified, and edited by the authors, who take full responsibility for the accuracy, integrity, and originality of the manuscript. No AI tool is listed as an author, in accordance with the journal’s authorship and contributorship policies.</p>
    </sec>
    <sec sec-type="author-contributions">
      <title>Authors Contribution</title>
      <p>BP contributed to conceptualization, study design, supervision, writing the original draft, and writing review and editing. SKK contributed to conceptualization, research planning, supervision, writing the original draft, and writing review and editing. BKP contributed to software, methodology, data curation, formal analysis, validation, and writing the original draft. SSD contributed to software, methodology, data curation, experimental investigation, validation, and writing the original draft. All authors have read and approved the final version of the manuscript.</p>
    </sec>
    <sec sec-type="data-availability">
      <title>Data Availability</title>
      <p>BreakHis is publicly accessible for research use upon request from the dataset maintainers. INbreast requires institutional registration and adherence to usage conditions specified by the hosting repository. CBIS-DDSM is available through The Cancer Imaging Archive (TCIA), subject to TCIA data usage policies. BUSI is publicly available for academic research purposes. Researchers must comply with the respective licensing and citation requirements of each dataset provider.</p>
    </sec>
    <ref-list>
      <ref id="ref-6f67085b37d6">
        <element-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Al-Hejri</surname>
              <given-names>A. M.</given-names>
            </name>
            <name>
              <surname>Sable</surname>
              <given-names>A. H.</given-names>
            </name>
            <name>
              <surname>Al-Tam</surname>
              <given-names>R. M.</given-names>
            </name>
            <name>
              <surname>Al-Antari</surname>
              <given-names>M. A.</given-names>
            </name>
            <name>
              <surname>Alshamrani</surname>
              <given-names>S. S.</given-names>
            </name>
            <name>
              <surname>Alshmrany</surname>
              <given-names>K. M.</given-names>
            </name>
            <name>
              <surname>Alatebi</surname>
              <given-names>W.</given-names>
            </name>
          </person-group>
          <article-title>A hybrid explainable federated-based vision transformer framework for breast cancer prediction via risk factors</article-title>
          <source>Sci Rep</source>
          <year>2025</year>
          <volume>15</volume>
          <issue>1</issue>
          <fpage>18453</fpage>
          <pub-id pub-id-type="doi">10.1038/s41598-025-96527-0</pub-id>
          <pub-id pub-id-type="pmid">40419634</pub-id>
          <pub-id pub-id-type="pmcid">PMC12106662</pub-id>
          <ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/pubmed/40419634">https://www.ncbi.nlm.nih.gov/pubmed/40419634</ext-link>
        </element-citation>
      </ref>
      <ref id="ref-707ad2879e3f">
        <element-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>McMahan</surname>
              <given-names>Brendan</given-names>
            </name>
            <name>
              <surname>Moore</surname>
              <given-names>Eider</given-names>
            </name>
            <name>
              <surname>Ramage</surname>
              <given-names>Daniel</given-names>
            </name>
            <name>
              <surname>Hampson</surname>
              <given-names>Seth</given-names>
            </name>
            <name>
              <surname>y Arcas</surname>
              <given-names>Blaise Aguera</given-names>
            </name>
          </person-group>
          <article-title>Communication-efficient learning of deep networks from decentralized data</article-title>
          <source>Artificial intelligence and statistics</source>
          <year>2017</year>
          <fpage>1273</fpage>
          <lpage>1282</lpage>
        </element-citation>
      </ref>
      <ref id="ref-8aa40c20886c">
        <element-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Konečný</surname>
              <given-names>Jakub</given-names>
            </name>
            <name>
              <surname>McMahan</surname>
              <given-names>H Brendan</given-names>
            </name>
            <name>
              <surname>Ramage</surname>
              <given-names>Daniel</given-names>
            </name>
            <name>
              <surname>Richtárik</surname>
              <given-names>Peter</given-names>
            </name>
          </person-group>
          <article-title>Federated optimization: Distributed machine learning for on-device intelligence</article-title>
          <source>arXiv preprint arXiv:1610.02527</source>
          <year>2016</year>
          <pub-id pub-id-type="doi">10.48550/arXiv.1610.02527</pub-id>
        </element-citation>
      </ref>
      <ref id="ref-3495eb21e423">
        <element-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Li</surname>
              <given-names>Tian</given-names>
            </name>
            <name>
              <surname>Sahu</surname>
              <given-names>Anit Kumar</given-names>
            </name>
            <name>
              <surname>Zaheer</surname>
              <given-names>Manzil</given-names>
            </name>
            <name>
              <surname>Sanjabi</surname>
              <given-names>Maziar</given-names>
            </name>
            <name>
              <surname>Talwalkar</surname>
              <given-names>Ameet</given-names>
            </name>
            <name>
              <surname>Smith</surname>
              <given-names>Virginia</given-names>
            </name>
          </person-group>
          <article-title>Federated optimization in heterogeneous networks</article-title>
          <source>Proceedings of Machine learning and systems</source>
          <year>2020</year>
          <volume>2</volume>
          <fpage>429</fpage>
          <lpage>450</lpage>
        </element-citation>
      </ref>
      <ref id="ref-1b499b2a31ac">
        <element-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Dosovitskiy</surname>
              <given-names>Alexey</given-names>
            </name>
            <name>
              <surname>Beyer</surname>
              <given-names>Lucas</given-names>
            </name>
            <name>
              <surname>Kolesnikov</surname>
              <given-names>Alexander</given-names>
            </name>
            <name>
              <surname>Weissenborn</surname>
              <given-names>Dirk</given-names>
            </name>
            <name>
              <surname>Zhai</surname>
              <given-names>Xiaohua</given-names>
            </name>
            <name>
              <surname>Unterthiner</surname>
              <given-names>Thomas</given-names>
            </name>
            <name>
              <surname>Dehghani</surname>
              <given-names>Mostafa</given-names>
            </name>
            <name>
              <surname>Minderer</surname>
              <given-names>Matthias</given-names>
            </name>
            <name>
              <surname>Heigold</surname>
              <given-names>Georg</given-names>
            </name>
            <name>
              <surname>Gelly</surname>
              <given-names>Sylvain</given-names>
            </name>
          </person-group>
          <article-title>An image is worth 16x16 words: Transformers for image recognition at scale</article-title>
          <source>arXiv preprint arXiv:2010.11929</source>
          <year>2020</year>
          <pub-id pub-id-type="doi">10.48550/arXiv.2010.11929</pub-id>
        </element-citation>
      </ref>
      <ref id="ref-e4debb87f107">
        <element-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Touvron</surname>
              <given-names>Hugo</given-names>
            </name>
            <name>
              <surname>Cord</surname>
              <given-names>Matthieu</given-names>
            </name>
            <name>
              <surname>Douze</surname>
              <given-names>Matthijs</given-names>
            </name>
            <name>
              <surname>Massa</surname>
              <given-names>Francisco</given-names>
            </name>
            <name>
              <surname>Sablayrolles</surname>
              <given-names>Alexandre</given-names>
            </name>
            <name>
              <surname>Jégou</surname>
              <given-names>Hervé</given-names>
            </name>
          </person-group>
          <article-title>Training data-efficient image transformers &amp; distillation through attention</article-title>
          <source>International conference on machine learning</source>
          <year>2021</year>
          <fpage>10347</fpage>
          <lpage>10357</lpage>
        </element-citation>
      </ref>
      <ref id="ref-b2ab91f7172f">
        <element-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Li</surname>
              <given-names>Qinbin</given-names>
            </name>
            <name>
              <surname>He</surname>
              <given-names>Bingsheng</given-names>
            </name>
            <name>
              <surname>Song</surname>
              <given-names>Dawn</given-names>
            </name>
          </person-group>
          <article-title>Model-contrastive federated learning</article-title>
          <source>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</source>
          <year>2021</year>
          <fpage>10713</fpage>
          <lpage>10722</lpage>
          <pub-id pub-id-type="doi">10.1109/CVPR46437.2021.01057</pub-id>
        </element-citation>
      </ref>
      <ref id="ref-c27bab27efc4">
        <element-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Alotaibi</surname>
              <given-names>M.</given-names>
            </name>
            <name>
              <surname>Aljouie</surname>
              <given-names>A.</given-names>
            </name>
            <name>
              <surname>Alluhaidan</surname>
              <given-names>N.</given-names>
            </name>
            <name>
              <surname>Qureshi</surname>
              <given-names>W.</given-names>
            </name>
            <name>
              <surname>Almatar</surname>
              <given-names>H.</given-names>
            </name>
            <name>
              <surname>Alduhayan</surname>
              <given-names>R.</given-names>
            </name>
            <name>
              <surname>Alsomaie</surname>
              <given-names>B.</given-names>
            </name>
            <name>
              <surname>Almazroa</surname>
              <given-names>A.</given-names>
            </name>
          </person-group>
          <article-title>Breast cancer classification based on convolutional neural network and image fusion approaches using ultrasound images</article-title>
          <source>Heliyon</source>
          <year>2023</year>
          <volume>9</volume>
          <issue>11</issue>
          <fpage>e22406</fpage>
          <pub-id pub-id-type="doi">10.1016/j.heliyon.2023.e22406</pub-id>
          <pub-id pub-id-type="pmid">38074874</pub-id>
          <pub-id pub-id-type="pmcid">PMC10700613</pub-id>
          <ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/pubmed/38074874">https://www.ncbi.nlm.nih.gov/pubmed/38074874</ext-link>
        </element-citation>
      </ref>
      <ref id="ref-08a1a08f747f">
        <element-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Rieke</surname>
              <given-names>N.</given-names>
            </name>
            <name>
              <surname>Hancox</surname>
              <given-names>J.</given-names>
            </name>
            <name>
              <surname>Li</surname>
              <given-names>W.</given-names>
            </name>
            <name>
              <surname>Milletari</surname>
              <given-names>F.</given-names>
            </name>
            <name>
              <surname>Roth</surname>
              <given-names>H. R.</given-names>
            </name>
            <name>
              <surname>Albarqouni</surname>
              <given-names>S.</given-names>
            </name>
            <name>
              <surname>Bakas</surname>
              <given-names>S.</given-names>
            </name>
            <name>
              <surname>Galtier</surname>
              <given-names>M. N.</given-names>
            </name>
            <name>
              <surname>Landman</surname>
              <given-names>B. A.</given-names>
            </name>
            <name>
              <surname>Maier-Hein</surname>
              <given-names>K.</given-names>
            </name>
            <name>
              <surname>Ourselin</surname>
              <given-names>S.</given-names>
            </name>
            <name>
              <surname>Sheller</surname>
              <given-names>M.</given-names>
            </name>
            <name>
              <surname>Summers</surname>
              <given-names>R. M.</given-names>
            </name>
            <name>
              <surname>Trask</surname>
              <given-names>A.</given-names>
            </name>
            <name>
              <surname>Xu</surname>
              <given-names>D.</given-names>
            </name>
            <name>
              <surname>Baust</surname>
              <given-names>M.</given-names>
            </name>
            <name>
              <surname>Cardoso</surname>
              <given-names>M. J.</given-names>
            </name>
          </person-group>
          <article-title>The future of digital health with federated learning</article-title>
          <source>NPJ Digit Med</source>
          <year>2020</year>
          <volume>3</volume>
          <fpage>119</fpage>
          <pub-id pub-id-type="doi">10.1038/s41746-020-00323-1</pub-id>
          <pub-id pub-id-type="pmid">33015372</pub-id>
          <pub-id pub-id-type="pmcid">PMC7490367</pub-id>
          <ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/pubmed/33015372">https://www.ncbi.nlm.nih.gov/pubmed/33015372</ext-link>
        </element-citation>
      </ref>
      <ref id="ref-262cc96f9f56">
        <element-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Shen</surname>
              <given-names>D.</given-names>
            </name>
            <name>
              <surname>Wu</surname>
              <given-names>G.</given-names>
            </name>
            <name>
              <surname>Suk</surname>
              <given-names>H. I.</given-names>
            </name>
          </person-group>
          <article-title>Deep Learning in Medical Image Analysis</article-title>
          <source>Annu Rev Biomed Eng</source>
          <year>2017</year>
          <volume>19</volume>
          <fpage>221</fpage>
          <lpage>248</lpage>
          <pub-id pub-id-type="doi">10.1146/annurev-bioeng-071516-044442</pub-id>
          <pub-id pub-id-type="pmid">28301734</pub-id>
          <pub-id pub-id-type="pmcid">PMC5479722</pub-id>
          <ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/pubmed/28301734">https://www.ncbi.nlm.nih.gov/pubmed/28301734</ext-link>
        </element-citation>
      </ref>
      <ref id="ref-1bae84594c98">
        <element-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Abdulrahman</surname>
              <given-names>Sawsan</given-names>
            </name>
            <name>
              <surname>Tout</surname>
              <given-names>Hanine</given-names>
            </name>
            <name>
              <surname>Ould-Slimane</surname>
              <given-names>Hakima</given-names>
            </name>
            <name>
              <surname>Mourad</surname>
              <given-names>Azzam</given-names>
            </name>
            <name>
              <surname>Talhi</surname>
              <given-names>Chamseddine</given-names>
            </name>
            <name>
              <surname>Guizani</surname>
              <given-names>Mohsen</given-names>
            </name>
          </person-group>
          <article-title>A Survey on Federated Learning: The Journey From Centralized to Distributed On-Site Learning and Beyond</article-title>
          <source>IEEE Internet of Things Journal</source>
          <year>2021</year>
          <volume>8</volume>
          <issue>7</issue>
          <fpage>5476</fpage>
          <lpage>5497</lpage>
          <pub-id pub-id-type="doi">10.1109/jiot.2020.3030072</pub-id>
        </element-citation>
      </ref>
      <ref id="ref-af0e8126d13d">
        <element-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Salakhutdinov</surname>
              <given-names>Ruslan</given-names>
            </name>
            <name>
              <surname>Hinton</surname>
              <given-names>Geoffrey</given-names>
            </name>
          </person-group>
          <article-title>Deep boltzmann machines</article-title>
          <source>Artificial intelligence and statistics</source>
          <year>2009</year>
          <fpage>448</fpage>
          <lpage>455</lpage>
        </element-citation>
      </ref>
      <ref id="ref-425f6742a018">
        <element-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Holzinger</surname>
              <given-names>Andreas</given-names>
            </name>
            <name>
              <surname>Biemann</surname>
              <given-names>Chris</given-names>
            </name>
            <name>
              <surname>Pattichis</surname>
              <given-names>Constantinos S</given-names>
            </name>
            <name>
              <surname>Kell</surname>
              <given-names>Douglas B</given-names>
            </name>
          </person-group>
          <article-title>What do we need to build explainable AI systems for the medical domain?</article-title>
          <source>arXiv preprint arXiv:1712.09923</source>
          <year>2017</year>
          <pub-id pub-id-type="doi">10.48550/arXiv.1712.09923</pub-id>
        </element-citation>
      </ref>
      <ref id="ref-ed6be83c48b1">
        <element-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Selvaraju</surname>
              <given-names>Ramprasaath R</given-names>
            </name>
            <name>
              <surname>Cogswell</surname>
              <given-names>Michael</given-names>
            </name>
            <name>
              <surname>Das</surname>
              <given-names>Abhishek</given-names>
            </name>
            <name>
              <surname>Vedantam</surname>
              <given-names>Ramakrishna</given-names>
            </name>
            <name>
              <surname>Parikh</surname>
              <given-names>Devi</given-names>
            </name>
            <name>
              <surname>Batra</surname>
              <given-names>Dhruv</given-names>
            </name>
          </person-group>
          <article-title>Grad-cam: Visual explanations from deep networks via gradient-based localization</article-title>
          <source>Proceedings of the IEEE international conference on computer vision</source>
          <year>2017</year>
          <fpage>618</fpage>
          <lpage>626</lpage>
        </element-citation>
      </ref>
      <ref id="ref-10222a7c52e0">
        <element-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Brunese</surname>
              <given-names>L.</given-names>
            </name>
            <name>
              <surname>Mercaldo</surname>
              <given-names>F.</given-names>
            </name>
            <name>
              <surname>Reginelli</surname>
              <given-names>A.</given-names>
            </name>
            <name>
              <surname>Santone</surname>
              <given-names>A.</given-names>
            </name>
          </person-group>
          <article-title>Explainable Deep Learning for Pulmonary Disease and Coronavirus COVID-19 Detection from X-rays</article-title>
          <source>Comput Methods Programs Biomed</source>
          <year>2020</year>
          <volume>196</volume>
          <fpage>105608</fpage>
          <pub-id pub-id-type="doi">10.1016/j.cmpb.2020.105608</pub-id>
          <pub-id pub-id-type="pmid">32599338</pub-id>
          <pub-id pub-id-type="pmcid">PMC7831868</pub-id>
          <ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/pubmed/32599338">https://www.ncbi.nlm.nih.gov/pubmed/32599338</ext-link>
        </element-citation>
      </ref>
      <ref id="ref-ddf859ae6406">
        <element-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Perez</surname>
              <given-names>Luis</given-names>
            </name>
            <name>
              <surname>Wang</surname>
              <given-names>Jason</given-names>
            </name>
          </person-group>
          <article-title>The effectiveness of data augmentation in image classification using deep learning</article-title>
          <source>arXiv preprint arXiv:1712.04621</source>
          <year>2017</year>
          <pub-id pub-id-type="doi">10.48550/arXiv.1712.04621</pub-id>
        </element-citation>
      </ref>
      <ref id="ref-aa580804fd60">
        <element-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Garrucho</surname>
              <given-names>L.</given-names>
            </name>
            <name>
              <surname>Kushibar</surname>
              <given-names>K.</given-names>
            </name>
            <name>
              <surname>Jouide</surname>
              <given-names>S.</given-names>
            </name>
            <name>
              <surname>Diaz</surname>
              <given-names>O.</given-names>
            </name>
            <name>
              <surname>Igual</surname>
              <given-names>L.</given-names>
            </name>
            <name>
              <surname>Lekadir</surname>
              <given-names>K.</given-names>
            </name>
          </person-group>
          <article-title>Domain generalization in deep learning based mass detection in mammography: A large-scale multi-center study</article-title>
          <source>Artif Intell Med</source>
          <year>2022</year>
          <volume>132</volume>
          <fpage>102386</fpage>
          <pub-id pub-id-type="doi">10.1016/j.artmed.2022.102386</pub-id>
          <pub-id pub-id-type="pmid">36207090</pub-id>
          <ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/pubmed/36207090">https://www.ncbi.nlm.nih.gov/pubmed/36207090</ext-link>
        </element-citation>
      </ref>
      <ref id="ref-648f94c2eb2c">
        <element-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Guan</surname>
              <given-names>H.</given-names>
            </name>
            <name>
              <surname>Yap</surname>
              <given-names>P. T.</given-names>
            </name>
            <name>
              <surname>Bozoki</surname>
              <given-names>A.</given-names>
            </name>
            <name>
              <surname>Liu</surname>
              <given-names>M.</given-names>
            </name>
          </person-group>
          <article-title>Federated learning for medical image analysis: A survey</article-title>
          <source>Pattern Recognit</source>
          <year>2024</year>
          <volume>151</volume>
          <pub-id pub-id-type="doi">10.1016/j.patcog.2024.110424</pub-id>
          <pub-id pub-id-type="pmid">38559674</pub-id>
          <pub-id pub-id-type="pmcid">PMC10976951</pub-id>
          <ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/pubmed/38559674">https://www.ncbi.nlm.nih.gov/pubmed/38559674</ext-link>
        </element-citation>
      </ref>
      <ref id="ref-74346807e08e">
        <element-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Kaissis</surname>
              <given-names>Georgios A.</given-names>
            </name>
            <name>
              <surname>Makowski</surname>
              <given-names>Marcus R.</given-names>
            </name>
            <name>
              <surname>Rückert</surname>
              <given-names>Daniel</given-names>
            </name>
            <name>
              <surname>Braren</surname>
              <given-names>Rickmer F.</given-names>
            </name>
          </person-group>
          <article-title>Secure, privacy-preserving and federated machine learning in medical imaging</article-title>
          <source>Nature Machine Intelligence</source>
          <year>2020</year>
          <volume>2</volume>
          <issue>6</issue>
          <fpage>305</fpage>
          <lpage>311</lpage>
          <pub-id pub-id-type="doi">10.1038/s42256-020-0186-1</pub-id>
        </element-citation>
      </ref>
    </ref-list>
  </back>
  <floats-group>
    <fig id="fig-1" specific-use="aside-float: width=full-width; anchor=blk-1e5b180323a3" position="float">
      <label>Figure 1</label>
      <caption>
        <p>Workflow of data preparation and preprocessing.</p>
      </caption>
      <graphic xlink:href="1.jpg"/>
    </fig>
    <fig id="fig-2" specific-use="aside-float: width=full-width; anchor=blk-d0aee204444b" position="float">
      <label>Figure 2</label>
      <caption>
        <p>Overview of the proposed federated vision transformer framework.</p>
      </caption>
      <graphic xlink:href="2.jpg"/>
    </fig>
    <fig id="fig-3" specific-use="aside-float: width=full-width; anchor=blk-d0aee204444b" position="float">
      <label>Figure 3</label>
      <caption>
        <p>Evolution of global accuracy, F1-score, and AUC across federated communication rounds, illustrating convergence under non-IID multi-institutional data.</p>
      </caption>
      <graphic xlink:href="3.jpg"/>
    </fig>
    <fig id="fig-4" specific-use="aside-float: width=full-width; anchor=blk-e05d238453c2" position="float">
      <label>Figure 4</label>
      <caption>
        <p>Confusion matrix of the final global federated model evaluated on the aggregated test set.</p>
      </caption>
      <graphic xlink:href="4.jpg"/>
    </fig>
    <fig id="fig-5" specific-use="aside-float: width=full-width; anchor=blk-fa3d44a191fd" position="float">
      <label>Figure 5</label>
      <caption>
        <p>Client-wise performance comparison of the federated hybrid CNN–Vision Transformer model across heterogeneous breast imaging datasets.</p>
      </caption>
      <graphic xlink:href="5.jpg"/>
    </fig>
    <fig id="fig-6" specific-use="aside-float: width=full-width; anchor=blk-5a213c5df214" position="float">
      <label>Figure 6</label>
      <caption>
        <p>Gradient-based explainability maps for representative test samples: (a) true positive, (b) false positive, and (c) false negative predictions.</p>
      </caption>
      <graphic xlink:href="6.jpg"/>
    </fig>
    <table-wrap id="tbl-1" specific-use="aside-float: layout=full-width; anchor=blk-8f28901e13fe" position="float">
      <label>Table 1</label>
      <caption>
        <p>Dataset Reproducibility Details and Label Harmonization</p>
      </caption>
      <table>
        <thead>
          <tr id="row-51726b8a9c87">
            <th id="cell-87dd6df7bd5d">
              <bold>Dataset</bold>
            </th>
            <th id="cell-eed1f0a3d4dd">
              <bold>Imaging Modality</bold>
            </th>
            <th id="cell-f3f8fb02d940">
              <bold>Unit of Data</bold>
            </th>
            <th id="cell-15ecf4fec6cc">
              <bold>Total Samples Used</bold>
            </th>
            <th id="cell-f6e0d8c99cc9">
              <bold>Benign Samples</bold>
            </th>
            <th id="cell-9c198912ede0">
              <bold>Malignant Samples</bold>
            </th>
            <th id="cell-00652546bdb0">
              <bold>Patient/Case Count</bold>
            </th>
            <th id="cell-16c2d84a72a5">
              <bold>Original Labels</bold>
            </th>
            <th id="cell-77c9c5663fac">
              <bold>Binary Label Mapping</bold>
            </th>
            <th id="cell-c1be388d3b37">
              <bold>Split Unit</bold>
            </th>
          </tr>
        </thead>
        <tbody>
          <tr id="row-b79c988d9196">
            <td id="cell-d7734281a72f">BreakHis</td>
            <td id="cell-d18eaa110c5e">Histopathology</td>
            <td id="cell-459892b36084">Image patches (×40–×400)</td>
            <td id="cell-ca082c227450">2,482</td>
            <td id="cell-df8d5db88dc8">1,244</td>
            <td id="cell-455aba1e7f01">1,238</td>
            <td id="cell-05d5a645ab77">82 patients</td>
            <td id="cell-79262178ff2f">8 tumor subtypes</td>
            <td id="cell-5efc17f509eb">Benign subtypes: Benign; Malignant subtypes: Malignant</td>
            <td id="cell-aeeae2acd143">Patient-level</td>
          </tr>
          <tr id="row-f20bd76fc34d">
            <td id="cell-fb9eda63c40f">INbreast</td>
            <td id="cell-55160acf3047">Mammography</td>
            <td id="cell-0c4274e1d25c">Full mammogram images</td>
            <td id="cell-177e15d53219">410</td>
            <td id="cell-dcd49b27ab24">205</td>
            <td id="cell-3609fc332272">205</td>
            <td id="cell-a4bc5815f68e">115 patients</td>
            <td id="cell-76bedd5c453e">BI-RADS / pathology</td>
            <td id="cell-9caf3b6dcc0e">Normal &amp; benign: Benign; Malignant: Malignant</td>
            <td id="cell-3ec77d491eb3">Patient-level</td>
          </tr>
          <tr id="row-f297d8267c53">
            <td id="cell-15a8ef7d7ec6">CBIS-DDSM</td>
            <td id="cell-ece0f52de817">Mammography</td>
            <td id="cell-ea1eb4e33783">ROI images</td>
            <td id="cell-07faba46ee91">1,696</td>
            <td id="cell-c1fff30d50ae">847</td>
            <td id="cell-2e7f6e509056">849</td>
            <td id="cell-5d7877823825">~753 cases</td>
            <td id="cell-8d023e227eee">Mass/Calcification with pathology</td>
            <td id="cell-6d1702e0ad90">Benign: Benign; Malignant: Malignant</td>
            <td id="cell-31dfdb929707">Case-level</td>
          </tr>
          <tr id="row-925f8f868fcc">
            <td id="cell-deb77388c4d9">BUSI</td>
            <td id="cell-3ec3b7263cfd">Ultrasound</td>
            <td id="cell-77d8098f6ee4">Image</td>
            <td id="cell-efe7c6f19954">780</td>
            <td id="cell-ec99d4f15c7f">487</td>
            <td id="cell-a32d18db12a9">293</td>
            <td id="cell-fcffcefc0674">Not specified</td>
            <td id="cell-54517c95030e">Normal / Benign / Malignant</td>
            <td id="cell-ce16178f4199">Benign &amp; Normal: Benign; Malignant: Malignant</td>
            <td id="cell-2b6fad27f4e0">Image-level</td>
          </tr>
        </tbody>
      </table>
      <table-wrap-foot>
        <p>ROI, region of interest; BI-RADS, Breast Imaging Reporting and Data System; CBIS-DDSM, Curated Breast Imaging Subset of the Digital Database for Screening Mammography.</p>
      </table-wrap-foot>
    </table-wrap>
    <table-wrap id="tbl-2" specific-use="aside-float: layout=full-width; anchor=blk-8f28901e13fe" position="float">
      <label>Table 2</label>
      <caption>
        <p>Quantification of Non-IID Characteristics Across Federated Clients</p>
      </caption>
      <table>
        <thead>
          <tr id="row-0b162242f915">
            <th id="cell-5777e0424e5a">
              <bold>Dataset</bold>
            </th>
            <th id="cell-20c237249c6f">
              <bold>Total Samples</bold>
            </th>
            <th id="cell-673566bfec05">
              <bold>Benign (%)</bold>
            </th>
            <th id="cell-ae40b5a4bec6">
              <bold>Malignant (%)</bold>
            </th>
            <th id="cell-f854cb3a81bf">
              <bold>Majority/Minority Ratio</bold>
            </th>
          </tr>
        </thead>
        <tbody>
          <tr id="row-4a9abca8af98">
            <td id="cell-656311005977">BreakHis</td>
            <td id="cell-1e891c8ae48d">2,482</td>
            <td id="cell-7195d3077290">50.12%</td>
            <td id="cell-82f5212f7448">49.88%</td>
            <td id="cell-d4fb005150b2">1.005</td>
          </tr>
          <tr id="row-ce044c7a8b44">
            <td id="cell-0f0a66243eea">INbreast</td>
            <td id="cell-1e8e0fbda3dd">410</td>
            <td id="cell-e8a3a6c40519">50.00%</td>
            <td id="cell-47540542e91e">50.00%</td>
            <td id="cell-fd9464a7047f">1.000</td>
          </tr>
          <tr id="row-58da8e3979ec">
            <td id="cell-f805ee25eaca">CBIS-DDSM</td>
            <td id="cell-01e829314bf5">1,696</td>
            <td id="cell-ccf292871ab0">49.94%</td>
            <td id="cell-f3034a3ed2bf">50.06%</td>
            <td id="cell-2bf265c3922b">1.002</td>
          </tr>
          <tr id="row-94730bcc3b33">
            <td id="cell-50b3afece3ea">BUSI</td>
            <td id="cell-ca1aea7273c9">780</td>
            <td id="cell-9c40720b6f37">62.44%</td>
            <td id="cell-7ec9ad7ccec5">37.56%</td>
            <td id="cell-c63ff3aa3143">1.662</td>
          </tr>
        </tbody>
      </table>
      <table-wrap-foot>
        <p>BUSI, Breast Ultrasound Images dataset; CBIS-DDSM, Curated Breast Imaging Subset of the Digital Database for Screening Mammography.</p>
      </table-wrap-foot>
    </table-wrap>
    <table-wrap id="tbl-3" specific-use="aside-float: layout=full-width; anchor=blk-fa3d44a191fd" position="float">
      <label>Table 3</label>
      <caption>
        <p>Per-client classification performance of the final federated model evaluated on individual held-out test sets under simulated non-IID conditions</p>
      </caption>
      <table>
        <thead>
          <tr id="row-536631df22b3">
            <th id="cell-29e918f6fe98">
              <bold>Client (Dataset)</bold>
            </th>
            <th id="cell-1e37c288d0e9">
              <bold>Test Samples</bold>
            </th>
            <th id="cell-41a257691803">
              <bold>Accuracy</bold>
            </th>
            <th id="cell-5c730d7daefd">
              <bold>AUC</bold>
            </th>
            <th id="cell-cbca98f82713">
              <bold>F1-Score</bold>
            </th>
            <th id="cell-b258119efbe0">
              <bold>Sensitivity</bold>
            </th>
            <th id="cell-74d6690baa16">
              <bold>Specificity</bold>
            </th>
          </tr>
        </thead>
        <tbody>
          <tr id="row-b3e59b4a0082">
            <td id="cell-054ace6fca01">BreakHis</td>
            <td id="cell-8e46c0f13efc">1187</td>
            <td id="cell-d7ba7f311939">0.8972</td>
            <td id="cell-712a901b6dd0">0.9870</td>
            <td id="cell-6ea638d46546">0.8910</td>
            <td id="cell-556e2a7d9bba">0.9975</td>
            <td id="cell-b888b2f261d4">0.6774</td>
          </tr>
          <tr id="row-8c0af5801ddc">
            <td id="cell-56260a65f803">INbreast</td>
            <td id="cell-a73dc1800ee7">410</td>
            <td id="cell-92dac1769faa">0.9633</td>
            <td id="cell-052785d28a0a">0.9973</td>
            <td id="cell-40db82fd16b1">0.9628</td>
            <td id="cell-bb3398cb650a">0.9949</td>
            <td id="cell-a77c5b978aa2">0.8950</td>
          </tr>
          <tr id="row-c2c6b815b3f1">
            <td id="cell-25923cfcdd25">CBIS-DDSM</td>
            <td id="cell-db0363bb14b5">463</td>
            <td id="cell-05b776501ab4">0.5162</td>
            <td id="cell-addd5340e110">0.5847</td>
            <td id="cell-9c72ebbfff3c">0.5043</td>
            <td id="cell-988dd805e6e3">0.7344</td>
            <td id="cell-5ede100802e1">0.3616</td>
          </tr>
          <tr id="row-a48fb062a1b6">
            <td id="cell-7907730c2f31">BUSI</td>
            <td id="cell-91129202eaab">98</td>
            <td id="cell-57816fafe62e">0.4490</td>
            <td id="cell-ea19295c27cf">0.5465</td>
            <td id="cell-8b103eb9b639">0.4446</td>
            <td id="cell-69b405bd2e04">0.7419</td>
            <td id="cell-d0af71ee07f2">0.3134</td>
          </tr>
        </tbody>
      </table>
      <table-wrap-foot>
        <p>AUC, area under the receiver operating characteristic curve; F1-score, harmonic mean of precision and recall.</p>
      </table-wrap-foot>
    </table-wrap>
  </floats-group>
</article>
