<?xml version="1.0" encoding="UTF-8" standalone="no"?><!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.2 20190208//EN" "http://jats.nlm.nih.gov/publishing/1.2/JATS-journalpublishing1.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="other" dtd-version="1.2" xml:lang="en">
    <front>
        <journal-meta>
            <journal-id journal-id-type="pmc">Open Res Europe</journal-id>
            <journal-title-group>
                <journal-title>Open Research Europe</journal-title>
            </journal-title-group>
            <issn pub-type="epub">2732-5121</issn>
            <publisher>
                <publisher-name>F1000 Research Limited</publisher-name>
                <publisher-loc>London, UK</publisher-loc>
            </publisher>
        </journal-meta>
        <article-meta>
            <article-id pub-id-type="doi">10.12688/openreseurope.21016.2</article-id>
            <article-categories>
                <subj-group subj-group-type="heading">
                    <subject>Case Study</subject>
                </subj-group>
                <subj-group>
                    <subject>Articles</subject>
                </subj-group>
            </article-categories>
            <title-group>
                <article-title>How the First Medical Imaging Cancer Atlas EUCAIM Was Populated: The Experience of a Reference Hospital.</article-title>
                <fn-group content-type="pub-status">
                    <fn>
                        <p>[version 2; peer review: 2 approved, 1 approved with reservations]</p>
                    </fn>
                </fn-group>
            </title-group>
            <contrib-group>
                <contrib contrib-type="author" corresp="yes">
                    <name>
                        <surname>Penad&#xE9;s Blasco</surname>
                        <given-names>Ana</given-names>
                    </name>
                    <role content-type="http://credit.niso.org/">Conceptualization</role>
                    <role content-type="http://credit.niso.org/">Data Curation</role>
                    <role content-type="http://credit.niso.org/">Investigation</role>
                    <role content-type="http://credit.niso.org/">Methodology</role>
                    <role content-type="http://credit.niso.org/">Project Administration</role>
                    <role content-type="http://credit.niso.org/">Writing &#x2013; Original Draft Preparation</role>
                    <uri content-type="orcid">https://orcid.org/0000-0002-2979-2123</uri>
                    <xref ref-type="corresp" rid="c1">a</xref>
                    <xref ref-type="aff" rid="a1">1</xref>
                </contrib>
                <contrib contrib-type="author" corresp="no">
                    <name>
                        <surname>Cerd&#xE1; Alberich</surname>
                        <given-names>Leonor</given-names>
                    </name>
                    <role content-type="http://credit.niso.org/">Conceptualization</role>
                    <role content-type="http://credit.niso.org/">Investigation</role>
                    <role content-type="http://credit.niso.org/">Methodology</role>
                    <role content-type="http://credit.niso.org/">Validation</role>
                    <role content-type="http://credit.niso.org/">Writing &#x2013; Review &amp; Editing</role>
                    <xref ref-type="aff" rid="a1">1</xref>
                </contrib>
                <contrib contrib-type="author" corresp="no">
                    <name>
                        <surname>de Marco Garc&#xED;a</surname>
                        <given-names>Ana</given-names>
                    </name>
                    <role content-type="http://credit.niso.org/">Conceptualization</role>
                    <role content-type="http://credit.niso.org/">Investigation</role>
                    <role content-type="http://credit.niso.org/">Methodology</role>
                    <role content-type="http://credit.niso.org/">Validation</role>
                    <role content-type="http://credit.niso.org/">Writing &#x2013; Review &amp; Editing</role>
                    <uri content-type="orcid">https://orcid.org/0009-0004-6906-3779</uri>
                    <xref ref-type="aff" rid="a1">1</xref>
                </contrib>
                <contrib contrib-type="author" corresp="no">
                    <name>
                        <surname>Soler Pons</surname>
                        <given-names>Carina</given-names>
                    </name>
                    <role content-type="http://credit.niso.org/">Data Curation</role>
                    <role content-type="http://credit.niso.org/">Methodology</role>
                    <role content-type="http://credit.niso.org/">Writing &#x2013; Review &amp; Editing</role>
                    <xref ref-type="aff" rid="a1">1</xref>
                </contrib>
                <contrib contrib-type="author" corresp="no">
                    <name>
                        <surname>Mar&#xED;n Radoszynski</surname>
                        <given-names>Irene</given-names>
                    </name>
                    <role content-type="http://credit.niso.org/">Data Curation</role>
                    <role content-type="http://credit.niso.org/">Methodology</role>
                    <role content-type="http://credit.niso.org/">Writing &#x2013; Review &amp; Editing</role>
                    <uri content-type="orcid">https://orcid.org/0009-0001-1841-7379</uri>
                    <xref ref-type="aff" rid="a1">1</xref>
                </contrib>
                <contrib contrib-type="author" corresp="no">
                    <name>
                        <surname>Mart&#xED;nez</surname>
                        <given-names>Ricard</given-names>
                    </name>
                    <role content-type="http://credit.niso.org/">Conceptualization</role>
                    <role content-type="http://credit.niso.org/">Investigation</role>
                    <role content-type="http://credit.niso.org/">Methodology</role>
                    <role content-type="http://credit.niso.org/">Validation</role>
                    <role content-type="http://credit.niso.org/">Writing &#x2013; Review &amp; Editing</role>
                    <xref ref-type="aff" rid="a2">2</xref>
                </contrib>
                <contrib contrib-type="author" corresp="no">
                    <name>
                        <surname>Segrelles-Quilis</surname>
                        <given-names>Dami&#xE1;n</given-names>
                    </name>
                    <role content-type="http://credit.niso.org/">Data Curation</role>
                    <role content-type="http://credit.niso.org/">Methodology</role>
                    <role content-type="http://credit.niso.org/">Writing &#x2013; Review &amp; Editing</role>
                    <xref ref-type="aff" rid="a3">3</xref>
                </contrib>
                <contrib contrib-type="author" corresp="no">
                    <name>
                        <surname>Blanquer</surname>
                        <given-names>Ignacio</given-names>
                    </name>
                    <role content-type="http://credit.niso.org/">Data Curation</role>
                    <role content-type="http://credit.niso.org/">Methodology</role>
                    <role content-type="http://credit.niso.org/">Writing &#x2013; Review &amp; Editing</role>
                    <xref ref-type="aff" rid="a3">3</xref>
                </contrib>
                <contrib contrib-type="author" corresp="no">
                    <name>
                        <surname>Mart&#xED;-Bonmat&#xED;</surname>
                        <given-names>Luis</given-names>
                    </name>
                    <role content-type="http://credit.niso.org/">Conceptualization</role>
                    <role content-type="http://credit.niso.org/">Methodology</role>
                    <role content-type="http://credit.niso.org/">Validation</role>
                    <role content-type="http://credit.niso.org/">Writing &#x2013; Review &amp; Editing</role>
                    <xref ref-type="aff" rid="a1">1</xref>
                    <xref ref-type="aff" rid="a4">4</xref>
                </contrib>
                <aff id="a1">
                    <label>1</label>Biomedical Imaging Research Group (GIBI230), La Fe Health Research Institute, Valencia, Spain, 46026, Spain</aff>
                <aff id="a2">
                    <label>2</label>IRTIC- Instituto Universitario de Investigaci&#xF3;n de Rob&#xF3;tica y Tecnolog&#xED;as de la Informaci&#xF3;n y Comunicaci&#xF3;n, Valencia, Spain, 46980, Spain</aff>
                <aff id="a3">
                    <label>3</label>Instituto de Instrumentaci&#xF3;n para la Imagen Molecular, Universitat Polit&#xE8;cnica de Val&#xE8;ncia, Valencia, Spain, 46011, Spain</aff>
                <aff id="a4">
                    <label>4</label>Medical Imaging Department, La Fe University and Polytechnic Hospital, Valencia, Spain, 46026, Spain</aff>
            </contrib-group>
            <author-notes>
                <corresp id="c1">
                    <label>a</label>
                    <email xlink:href="mailto:ana_penades@iislafe.es">ana_penades@iislafe.es</email>
                </corresp>
                <fn fn-type="conflict">
                    <p>No competing interests were disclosed.</p>
                </fn>
            </author-notes>
            <pub-date pub-type="epub">
                <day>24</day>
                <month>11</month><year>2025</year>
            </pub-date>
            <pub-date pub-type="collection"><year>2025</year>
            </pub-date><volume>5</volume>
            <elocation-id>310</elocation-id>
            <history>
                <date date-type="accepted">
                    <day>21</day>
                    <month>11</month><year>2025</year>
                </date>
            </history>
            <permissions>
                <copyright-statement>Copyright: &#xA9; 2025 Penad&#xE9;s Blasco A et al.</copyright-statement>
                <copyright-year>2025</copyright-year>
                <license xlink:href="https://creativecommons.org/licenses/by/4.0/">
                    <license-p>This is an open access article distributed under the terms of the Creative Commons Attribution Licence, which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.</license-p>
                </license>
            </permissions>
            <self-uri content-type="pdf" xlink:href="https://open-research-europe.ec.europa.eu/articles/5-310/pdf"/>
            <abstract>
                <p>The fragmentation and decentralization of medical data, including radiological imaging, continue to challenge large-scale observational research across Europe. Artificial Intelligence (AI) applied to big datasets is transforming diagnosis and treatments towards precision medicine across many diseases, yet the lack of findable, accessible, and interoperable datasets still limits model development, validation, and final clinical translation. The European Federation for Cancer Images (EUCAIM) project was launched in 2023 to address these challenges by establishing a secure centralized and federated infrastructure for the secondary use of large-scale oncological imaging and related clinical data.</p>
                <p>By consolidating fragmented datasets, EUCAIM lays the groundwork for harmonized data governance and trusted cross-border sharing. Implementing a robust documentation framework is essential to ensure regulatory compliance, safeguard data integrity, and support secure data flows across institutional and national boundaries, fully aligned with European regulations and ethical standards.</p>
                <p>EUCAIM builds on the AI for Health Imaging (AI4HI) initiative (Predictive In-silico Multiscale Analytics to support cancer personalized diagnosis and prognosis, empowered by imaging biomarkers - PRIMAGE, Accelerating the lab to market transition of AI tools for cancer management - CHAIMELEON, Novel pan-European imaging platform for artificial intelligence advances in oncology - EuCanImage, An AI Platform integrating imaging data and models, supporting precision care through prostate cancer&#x2019;s continuum - ProCancer-I, A multimodal AI-based toolbox and an interoperable health imaging repository for the empowerment of imaging analysis related to the diagnosis, prediction and follow-up of cancer - INCISIVE and integrates over 94 partners and more than 180 stakeholders spanning medical imaging, high performance computing, data standardization, innovation, and legal compliance. This large collaborative ecosystem reinforces EUCAIM&#x2019;s role as a reference for General Data Protection Regulation (GDPR) and European Health Data Space Regulation (EHDSR) adherence.</p>
                <p>This publication presents the real-world experience of integrating imaging and clinical data from a reference university hospital into the EUCAIM infrastructure. It outlines the procedural, ethical, and legal challenges encountered, and details the strategies implemented to ensure compliance with data protection regulations, including privacy, security, and ethical standards. These insights offer a practical framework for future large-scale oncological imaging datasets harmonization and AI development, contributing to scalable, reproducible, and legally compliant research that strengthens Europe&#x2019;s capacity for trustworthy AI-driven oncology solutions.</p>
            </abstract>
            <abstract abstract-type="plain-language-summary">
                <title>Plain Language Summary</title>
                <p>The EUCAIM (European Federation for Cancer Images) project was launched in 2023 to improve the way cancer imaging and related clinical data are shared and reused across Europe. Medical data, such as radiology images, are often fragmented and stored in different places, making it hard for researchers to access large, high-quality datasets. This limits the development, testing and validation of artificial intelligence (AI) tools that could help improve diagnosis and treatment.</p>
                <p>EUCAIM aims to solve these difficulties by creating a secure and centralized infrastructure that brings together cancer imaging data from across Europe. It supports the safe and ethical reuse of these data for research, following strict rules on privacy, security, and informed consent. The project includes over 94 partners and 180 stakeholders from different fields (such as medical imaging, computing, data standards, and law) ensuring strong collaboration and extensive expertise.</p>
                <p>EUCAIM builds on previous European projects (like PRIMAGE, CHAIMELEON, and ProCancer-I) and aligns with major EU regulations like the General Data Protection Regulation (GDPR) and the European Health Data Space Regulation (EHDSR). A key part of the project is ensuring that the infrastructure meets all legal and ethical requirements for handling sensitive health data.</p>
                <p>This article shares the real-life experience of a reference hospital contributing imaging and clinical data to EUCAIM and describes the technical and legal steps taken to ensure compliance and protect patients' information. These insights offer a practical example for other institutions considering participation in similar initiatives.</p>
                <p>Overall, EUCAIM is helping to build a trusted European framework for using medical imaging data in AI research, aiming to accelerate innovation in cancer care while protecting patients&#x2019; rights.</p>
            </abstract>
            <kwd-group kwd-group-type="author">
                <kwd>federated infrastructures</kwd>
                <kwd>sustainability</kwd>
                <kwd>medical imaging</kwd>
                <kwd>Artificial Intelligence</kwd>
                <kwd>cancer research</kwd>
                <kwd>data governance</kwd>
                <kwd>innovation</kwd>
            </kwd-group>
            <funding-group>
                <award-group id="fund-1">
                    <funding-source>Conseller&#xED;a de Innovaci&#xF3;n, Industria, Comercio y Turismo (Valencian Community)</funding-source>
                </award-group>
                <award-group id="fund-2">
                    <funding-source>European Commission Digital Europe Programme</funding-source>
                    <award-id>101100633</award-id>
                </award-group>
                <funding-statement>This project has received funding from the Digital Europe Programme (DIGITAL) under grant agreement No 101100633 (European Federation for Cancer Images [EUCAIM]).&#xA0;</funding-statement>
                <funding-statement>
                    <italic>The funders had no role in study design, data collection and analysis, decision to publish, or preparation of the manuscript.</italic>
                </funding-statement>
            </funding-group>
        </article-meta>
        <notes>
            <sec sec-type="version-changes">
                <label>Revised</label>
                <title>Amendments from Version 1</title>
                <p>Dear Readers, We have incorporated into the manuscript a detailed description of the computational performance metrics and storage requirements of the EUCAIM UPV Reference Node. In addition, we have included links to publicly accessible resources, such as the anonymization script and metadata repository. A new figure has also been added to summarise the relevant legal documentation steps within the EUCAIM infrastructure. Finally, we have carefully reviewed all acronyms throughout the manuscript to ensure consistency. We hope that these revisions have improved the clarity and overall understanding of the manuscript. Sincerely, Ana Penad&#xE9;s - Blasco</p>
            </sec>
        </notes>
    </front>
    <body>
        <sec sec-type="intro">
            <title>Introduction</title>
            <p>Access to secondary use of medical imaging data for AI-driven research are hindered by fragmentation, lack of interoperability, and complex regulations across Europe regarding privacy and security issues. Although many research initiatives have created imaging repositories, these are often project-specific, temporally limited, and constrained in their potential for reuse. This is especially challenging in oncology
                <sup>
                    <xref ref-type="bibr" rid="ref-1">1</xref>&#x2013;
                    <xref ref-type="bibr" rid="ref-3">3</xref>
                </sup>, where AI tools for precise diagnosis, prognosis and treatment estimations require large, standardized and annotated datasets to reach full clinical applicability
                <sup>
                    <xref ref-type="bibr" rid="ref-4">4</xref>,
                    <xref ref-type="bibr" rid="ref-5">5</xref>
                </sup>.</p>
            <p>The 
                <ext-link ext-link-type="uri" xlink:href="https://cancerimage.eu/">European Federation for Cancer Images (EUCAIM) project</ext-link> addresses these challenges by developing a large-scale, secure, and federated digital infrastructure for medical imaging and related clinical data. EUCAIM accelerates AI research in personalized medicine, offering tools that support clinical decision-making in complex settings
                <sup>
                    <xref ref-type="bibr" rid="ref-6">6</xref>
                </sup>. EUCAIM builds on the 
                <ext-link ext-link-type="uri" xlink:href="https://ai4hi.net/">AI for Health Imaging (AI4HI)</ext-link> network, which includes five major Horizon Europe projects (PRIMAGE, CHAIMELEON, EuCanImage, ProCancer-I, and INCISIVE). With over 90 partners and more than 180 stakeholders, EUCAIM is creating a GDPR-compliant, sustainable ecosystem for AI-based cancer research
                <sup>
                    <xref ref-type="bibr" rid="ref-7">7</xref>,
                    <xref ref-type="bibr" rid="ref-8">8</xref>
                </sup>.</p>
            <p>While EUCAIM offers a unique opportunity to harmonize and share annotated oncological imaging data, its success depends on overcoming technical barriers (standardization, transformation, anonymization, curation), strict compliance with the General Data Protection Regulation (
                <ext-link ext-link-type="uri" xlink:href="https://eur-lex.europa.eu/legal-content/EN/TXT/PDF/?uri=CELEX:32016R0679">GDPR</ext-link>), the European Health Data Space Regulation (
                <ext-link ext-link-type="uri" xlink:href="https://eur-lex.europa.eu/legal-content/EN/TXT/PDF/?uri=OJ:L_202500327">EHDSR</ext-link>), and the 
                <ext-link ext-link-type="uri" xlink:href="https://eur-lex.europa.eu/legal-content/EN/TXT/PDF/?uri=OJ:L_202401689">Artificial Intelligence Act</ext-link> (AI Act), and with a clear governance frameworks
                <sup>
                    <xref ref-type="bibr" rid="ref-9">9</xref>
                </sup>. The EUCAIM platform aims to host over 60 million images from more than 100,000 patients, transforming medical data use in personalized actionable care
                <sup>
                    <xref ref-type="bibr" rid="ref-10">10</xref>
                </sup>.</p>
            <p>EUCAIM&#x2019;s hybrid model allows Data Providers to transfer data to a reference node or set up their own federated node. This paper analyses the use case of real-world data transfer from a referral hospital, detailing workflows and practical solutions to secure legal compliant integration into EU-wide repositories.</p>
        </sec>
        <sec sec-type="materials | methods">
            <title>Material and methods</title>
            <p>Data integration in EUCAIM follows either a 
                <italic toggle="yes">Data Sharing Agreement</italic> (DSA), where data stays within the institution via a federated node; or a 
                <italic toggle="yes">Data Transfer Agreement</italic> (DTA), where data is physically moved to a reference node which comprises 10 Gigabyte R283-ZF0-AAL1 nodes, each equipped with two AMD EPYC 9474F 48-core processors (a total of 960 cores), 7.68 TB of RAM, 15 NVIDIA A30 and 10 NVIDIA L40S GPU accelerators with 48 GB of RAM each, as well as an additional storage server with 16 TB of NVMe SSD disks connected to the nodes via dual 25 GbE links. In the experiments performed in previous works
                <sup>
                    <xref ref-type="bibr" rid="ref-11">11</xref>
                </sup>, this capacity is sufficient to deal with a workload of over 30 concurrent users.</p>
            <p>For the retrospective imaging data included in this study, no informed consent was required, an exemption of informed consent was submitted to and approved by the Ethics Committee. All these nodes are core to EUCAIM&#x2019;s infrastructure.</p>
            <sec>
                <title>Data extraction and exposure pipeline</title>
                <p>To ensure standardized, anonymized, and compliant data integration into the EUCAIM infrastructure, a structured Extraction, Transformation, and Loading (ETL) pipeline was defined, incorporating best practices in data governance, security, and regulatory adherence. Imaging data were extracted from the hospital Picture Archiving and Communication System (PACS) through specific nodes in Digital Imaging and Communication in Medicine (DICOM) format according to each project&#x2019;s inclusion criteria. Corresponding clinical data were retrieved from the Electronic Health Record (EHR) system to maintain proper alignment. Pseudonymization and compliance checks were carried out on a restricted-access virtual machine by dedicated personnel, ensuring a clear separation from researchers and full GDPR compliance. This process follows guidelines from the DICOM Standards Committee and the AI4HI initiative to guarantee interoperability with EUCAIM.</p>
                <p>Key pseudonymization steps included replacing the PatientID with the project code (for both imaging and clinical data) and applying the 
                    <ext-link ext-link-type="uri" xlink:href="https://www.blake2.net/">Blake2b</ext-link> hashing function to replace the AccessionNumber and StudyID metadata. Folder structures were adjusted to remove any Personally Identifiable Information (PII), and all public/private DICOM metadata potentially containing personally identifiable information were removed, following EUCAIM&#x2019;s anonymization profile. Secondary captures and screenshots, which may embed sensitive information, were excluded when identified by terms like &#x201C;DERIVED,&#x201D; &#x201C;SECONDARY,&#x201D; or &#x201C;SCREEN SAVE&#x201D; in the ImageType metadata.</p>
                <p>Pseudonymized data was then transferred to a separate file system, the Pseudonymized Medical Imaging Repository (RIMP). The DICOM File Integrity Checker developed by our group was used to detect corrupted or missing files. Data were finally ingested into EUCAIM datasets through the 
                    <ext-link ext-link-type="uri" xlink:href="https://quibim.com/qp-insights/">QP-Insights API</ext-link>, ensuring stronger anonymization, and secure transfer or sharing. This two-layer anonymization approach mitigates re-identification risks, aligning with procedures recommended by the 
                    <ext-link ext-link-type="uri" xlink:href="https://www.aepd.es/">Spanish Data Protection Agency</ext-link> and 
                    <ext-link ext-link-type="uri" xlink:href="https://www.pdpc.gov.sg/help-and-resources/2018/01/basic-anonymisation">best practices from the Singaporean authority</ext-link> (
                    <xref ref-type="fig" rid="f1">Figure 1</xref>). For data standardization, all DICOM files and metadata were validated for compliance with EUCAIM&#x2019;s structure and 
                    <ext-link ext-link-type="uri" xlink:href="https://zenodo.org/records/15558108">interoperability standards</ext-link>.</p>
                <fig fig-type="figure" id="f1" orientation="portrait" position="float">
                    <label>Figure 1. </label>
                    <caption>
                        <title>Steps for anonymization.</title>
                        <p>Source: Adapted from AEPD. Guide to basic anonymization. Prepared by the National Data Protection Authority of Singapore (PDPC - Personal Data Protection Commission Singapore.</p>
                    </caption>
                    <graphic orientation="portrait" position="float" xlink:href="https://openreseurope-files.f1000.com/manuscripts/23735/f003d0c5-cb35-4ca6-b5e7-5c5df9a6e157_figure1.gif"/>
                </fig>
                <p>The 
                    <ext-link ext-link-type="uri" xlink:href="https://eucaim-node.i3m.upv.es/dataset-service/datasets?invalidated=false">EUCAIM Federated Node</ext-link> provides an open-source, fully integrated Data Lake, Registry, and Secure Virtual Research Environment, backed by dedicated computing resources. Its API enables the ingestion while logging all transfers for traceability and auditability. Federated nodes are fully integrated within the EUCAIM platform and meet the EHDSR requirements for Secure Processing Environments (SPEs). SPEs ensure that data processing stays local, under the health data holder&#x2019;s continuous control, significantly reducing transfer risks. Specialized mediation software enforces strict anonymization rules, restricts user actions, and logs all activities to prevent misuse and ensure full accountability.</p>
                <p>By leveraging SPEs and federated nodes, EUCAIM provides a robust, compliant, and auditable framework for the secure secondary use of health data. This rigorous ETL pipeline ensures scalability, data quality, and legal compliance, positioning EUCAIM&#x2019;s reference and federated nodes as the backbone for integrating data from research projects, clinical trials
                    <sup>
                        <xref ref-type="bibr" rid="ref-12">12</xref>
                    </sup>, and future initiatives, in line with European regulations such as the 
                    <ext-link ext-link-type="uri" xlink:href="https://eur-lex.europa.eu/legal-content/EN/TXT/PDF/?uri=OJ:L_202302854&amp;qid=1752574267434">Data Act</ext-link> and 
                    <ext-link ext-link-type="uri" xlink:href="https://eur-lex.europa.eu/legal-content/EN/TXT/PDF/?uri=CELEX:32022R0868">Data Governance Act</ext-link>. This strategy guarantees EUCAIM&#x2019;s long-term sustainability and impact on AI research in medicine
                    <sup>
                        <xref ref-type="bibr" rid="ref-13">13</xref>
                    </sup>.</p>
            </sec>
            <sec>
                <title>Documentation framework for data integration to EUCAIM</title>
                <p>Ensuring compliant data integration within EUCAIM requires strict adherence to clear documentation protocols. At its current stage, decisions on the final integration of a dataset and/or federated node are verified by an Access Committee (AC). During this phase, it is mandatory to provide documentary evidence that guarantees accountability in EUCAIM&#x2019;s operation and demonstrates the application reliability.</p>
                <p>For data transfers to EUCAIM, the following documents are considered to safeguard data integrity and regulatory compliance previous to the signature of the DTA:</p>
                <list list-type="bullet">
                    <list-item>
                        <p>Data Protection Officer (DPO) Report or Self-Declaration: formal declaration by the transferring institution affirming that the data processing activities comply with GDPR requirements, including data minimization, data protection by design, and risk mitigation measures.</p>
                    </list-item>
                    <list-item>
                        <p>Codes of conduct or certification schemes adherence: a non-mandatory certification of GDPR declaring the institution provides evidence of adherence to established codes of conduct or certifications relevant to GDPR compliance.</p>
                    </list-item>
                    <list-item>
                        <p>Data Protection Impact Assessment (DPIA). documented evidence of the completed DPIA must be submitted in cases where, in accordance with Article 25 of the GDPR and the relevant national authority's whitelist, it is required.</p>
                    </list-item>
                    <list-item>
                        <p>Ethical Approval Documentation from the relevant Ethics Committee: certifying that the data transfer aligns with ethical research practices and data protection regulations. The requirement for ethics approval, its specific content, or any granted exemption is determined by national law. In cases where ethics approval is not required, it is recommended to be explicitly documented in the statement provided by the DPO.</p>
                    </list-item>
                    <list-item>
                        <p>Legal Representation Declaration: signed by the institution&#x2019;s legal representative, confirming that the data transfer adheres to legal obligations and that the representative is duly authorized to sign agreements on behalf of the institution. This documentary requirement is crucial, as requests might be handled by researchers or individuals who lack the legal capacity to make binding declarations of intent.</p>
                    </list-item>
                </list>
                <p>For Data Sharing Agreements (DSAs) via federated nodes, the same requirements apply, with an additional Security Report or certification (e.g., ISO 27001) document verifying that data sharing meets EUCAIM&#x2019;s security standards for encryption, user authentication, and access control.</p>
                <p>All documentation must be verifiable and may undergo audits. Federated nodes must show secure operations and technical interoperability with the EUCAIM platform. Likewise, datasets claimed to be anonymized will be checked: any re-identification risk means the data will be returned for further processing until fully compliant. This framework guarantees that all partners contribute data responsibly, supporting secure, transparent, and efficient data sharing within the EHDSR.</p>
                <p>The relevant legal documentation can be summarized in 
                    <xref ref-type="fig" rid="f2">Figure 2</xref>
                </p>
                <fig fig-type="figure" id="f2" orientation="portrait" position="float">
                    <label>Figure 2. </label>
                    <caption>
                        <title>Summary of the relevant legal documentation in EUCAIM.</title>
                    </caption>
                    <graphic orientation="portrait" position="float" xlink:href="https://openreseurope-files.f1000.com/manuscripts/23735/f003d0c5-cb35-4ca6-b5e7-5c5df9a6e157_figure2.gif"/>
                </fig>
            </sec>
            <sec>
                <title>Use case of data transfer from completed research projects</title>
                <p>The PRIMAGE project (GA: 826494) demonstrated how historical datasets could be successfully transferred and validated within the EUCAIM reference node. The main steps included:</p>
                <list list-type="bullet">
                    <list-item>
                        <p>Legal and Ethical Compliance: all approvals and documentation were secured.</p>
                    </list-item>
                    <list-item>
                        <p>Data Processing &amp; Transfer: imaging data were extracted, anonymized, and reformatted into EUCAIM-compliant DICOM structures. Clinical data was similarly processed, aligned with the EUCAIM Construction, Design and Management (CDM) and hyper-ontology based on mCODE specifications.</p>
                    </list-item>
                </list>
            </sec>
            <sec>
                <title>Use case of data transfer from oncology clinical trials</title>
                <p>The previous approach was also applied to ongoing clinical trials, expanding EUCAIM&#x2019;s reference node cases. Additional challenges emerged, particularly regarding data ownership and sponsor agreements. Key strategies included:</p>
                <list list-type="bullet">
                    <list-item>
                        <p>Data Ownership Clarification: a legal framework defined that imaging data belong to the patient&#x2019;s EHR, not the trial sponsor, minimizing transfer conflicts and ensuring EU compliance.</p>
                    </list-item>
                    <list-item>
                        <p>Informed Consent: exemptions were requested as data were fully anonymized and restricted to research use.</p>
                    </list-item>
                    <list-item>
                        <p>Periodic Scheduling: semi-annual transfers allow continuous integration in line with regulatory approvals.</p>
                    </list-item>
                </list>
            </sec>
            <sec>
                <title>Use case of data sharing from ongoing research projects</title>
                <p>The CHAIMELEON project (GA: 952172) illustrated a continuous dataset sharing process without disrupting ongoing research, serving as a model for federated nodes. For these projects, a more dynamic workflow was required due to evolving data collection and regulatory requirements:</p>
                <list list-type="bullet">
                    <list-item>
                        <p>Ethics Committee Approval: an addendum updated the original approval to permit secondary data use in EUCAIM.</p>
                    </list-item>
                    <list-item>
                        <p>Data Harmonization &amp; Monitoring: a real-time tracking system ensured seamless, compliant integration.</p>
                    </list-item>
                </list>
                <p>CHAIMELEON reached a high maturity level by preparing a robust DSA, anonymizing data, implementing a secure node, and completing a DPIA, setting a strong precedent for future compliant transfers.</p>
            </sec>
            <sec>
                <title>Ethical and legal considerations</title>
                <p>Each use case prioritizes full compliance with data protection regulations, especially GDPR. All data transferring and sharing activities were approved by the relevant ethics committees, ensuring confidentiality, integrity, and security. This rigorous approach has enabled the secure transfer and sharing of large volume of anonymized data, contributing to advanced AI research. Within EUCAIM, data use is strictly limited to users with approved research projects that meet institutional ethics standards and follow the Access Committee&#x2019;s protocols.</p>
                <p>The recent approval of the EHDSR marks a significant milestone for EUCAIM and similar initiatives, providing a harmonized framework for secure access, sharing, and secondary use of large-scale medical datasets. This framework sets clear rules for research and innovation, supported by strict requirements for security, privacy, and informed consent, fostering an interoperable and sustainable data ecosystem. EUCAIM&#x2019;s activities have been progressively aligned with this evolving landscape, positioning as a pilot platform under 
                    <ext-link ext-link-type="uri" xlink:href="https://acceptance.data.health.europa.eu/">HealthData@EU</ext-link>. Its formal integration into the EHDS ecosystem will strengthen data accessibility and protection, reinforcing the project&#x2019;s long-term sustainability and scientific value, as data volume and diversity grow.</p>
                <p>EUCAIM addresses sustainability through the establishment of a European Digital Infrastructure (EDIC) to enable multi-country investments in large-scale projects. The proposed model combines national and node contributions with European funding to ensure long-term operation of the Central Hub and integration of new partners.</p>
            </sec>
        </sec>
        <sec sec-type="results">
            <title>Results</title>
            <p>The implementation of a dedicated, harmonized pipeline for data transfer and sharing enabled the successful ingestion of cancer imaging data into the EUCAIM infrastructure. This process established a replicable and scalable workflow suitable for future large-scale federated data integration.</p>
            <p>In total, 12,484 medical imaging studies at our hospital and research institute were identified, reviewed, and prepared for integration into EUCAIM, covering both research datasets and real-world clinical trial data. Of these, 10,892 studies (87.24%) from 6,105 patients have been fully processed and validated for upload through either a reference or federated node approach. Our institution has contributed to several initiatives, including the completed PRIMAGE project with 878 studies, the ongoing CHAIMELEON project with 5,346 studies, and a series of Oncology Clinical Trials comprising 4,668 studies. The datasets include multiple modalities ((Computed Tomography - CT, Magnetic Resonance - MR, mammography, Positron Emission Tomography-Computed Tomography - PET-CT)) and represent a broad spectrum of cancer cases and patient populations (
                <xref ref-type="table" rid="T1">Table 1</xref>).</p>
            <table-wrap id="T1" orientation="portrait" position="anchor">
                <label>Table 1. </label>
                <caption>
                    <title>Overview of prepared Data Sources for Integration into the EUCAIM Reference Node and Federated Node.</title>
                </caption>
                <table content-type="article-table" frame="hsides">
                    <thead>
                        <tr>
                            <th align="center" colspan="1" rowspan="1" valign="middle">Acronym</th>
                            <th align="center" colspan="1" rowspan="1" valign="middle">Project status</th>
                            <th align="center" colspan="1" rowspan="1" valign="middle">Number of identified
                                <break/>imaging studies</th>
                            <th align="center" colspan="1" rowspan="1" valign="middle">Number of prepared
                                <break/>imaging studies</th>
                            <th align="center" colspan="1" rowspan="1" valign="middle">Node</th>
                        </tr>
                    </thead>
                    <tbody>
                        <tr>
                            <td align="center" colspan="1" rowspan="1" valign="middle">PRIMAGE</td>
                            <td align="center" colspan="1" rowspan="1" valign="middle">Completed</td>
                            <td align="center" colspan="1" rowspan="1" valign="middle">878</td>
                            <td align="center" colspan="1" rowspan="1" valign="middle">878</td>
                            <td align="center" colspan="1" rowspan="1" valign="middle">Reference</td>
                        </tr>
                        <tr>
                            <td align="center" colspan="1" rowspan="1" valign="middle">CHAIMELEON</td>
                            <td align="center" colspan="1" rowspan="1" valign="middle">Ongoing</td>
                            <td align="center" colspan="1" rowspan="1" valign="middle">5,346</td>
                            <td align="center" colspan="1" rowspan="1" valign="middle">5,346</td>
                            <td align="center" colspan="1" rowspan="1" valign="middle">Federated</td>
                        </tr>
                        <tr>
                            <td align="center" colspan="1" rowspan="1" valign="middle">Clinical Trials</td>
                            <td align="center" colspan="1" rowspan="1" valign="middle">Completed and
                                <break/>ongoing</td>
                            <td align="center" colspan="1" rowspan="1" valign="middle">6,260</td>
                            <td align="center" colspan="1" rowspan="1" valign="middle">4,668</td>
                            <td align="center" colspan="1" rowspan="1" valign="middle">Reference</td>
                        </tr>
                    </tbody>
                </table>
            </table-wrap>
            <p>The ETL pipeline achieved a processing efficiency of 98.6%, with minimal data loss. Automated DICOM anonymization, combined with manual checks, ensured GDPR compliance while maintaining clinical relevance. PRIMAGE delivered one of EUCAIM&#x2019;s largest structured paediatric oncology datasets, setting a benchmark for rare disease integration. CHAIMELEON validated a stepwise model for harmonized data sharing, and the inclusion of ongoing clinical trial data demonstrated the feasibility of incorporating prospective datasets into the federated infrastructure.</p>
            <p>Several challenges emerged that required refinements to optimize efficiency, scalability, and compliance. A key technical challenge was achieving interoperability across diverse imaging formats and legacy PACS metadata, which often required extensive transformation and custom mapping. Variability in clinical data mappings also highlighted the need for more automated and standardized processes.</p>
            <p>Administrative and legal aspects added complexity. Ethics approvals often needed significant revisions depending on the project type, and requests for informed consent exemptions required careful justification and additional documentation. These barriers underscored the lesson learned: the EHDS will require organisations to renew their ethical and legal governance models to ensure predictable, efficient data sharing, including data holders, research infrastructures, and users.</p>
            <p>Scalability was initially hindered by reliance on manual anonymization checks, slowing processing and introducing inconsistencies. Introducing automated validation tools and batch workflows significantly improved efficiency, though manual review remains necessary for nuanced cases like secondary capture images.</p>
            <p>These insights guided refinements to our protocols. Automated metadata validation minimized errors, while standardized ethics templates helped accelerate approval timelines. Experience gained across projects emphasized the importance of early alignment with EUCAIM requirements, ensuring technical, compliance, and governance aspects are clear from the outset.</p>
            <p>Despite initial hurdles, this process strengthened our capacity to securely transfer oncological imaging data with the highest standards of security, privacy, and interoperability, providing a solid foundation for future data-sharing efforts within EUCAIM and similar large-scale initiatives.</p>
        </sec>
        <sec sec-type="discussion">
            <title>Discussion</title>
            <p>EUCAIM is established as a landmark initiative in oncological imaging, offering Europe&#x2019;s first large-scale hybrid infrastructure for the secondary use of imaging, clinical, and molecular data. By integrating datasets from completed research, ongoing projects, and clinical trials, EUCAIM ensures alignment with the EHDSR and GDPR, thereby supporting the full potential of artificial intelligence in advancing predictive and personalized cancer care. Nevertheless, significant challenges remain, particularly data fragmentation, legacy system integration, and cross-border regulatory complexity.</p>
            <p>A key barrier to building a unified repository has been the variability in data formats, metadata, and institutional governance practices. EUCAIM addresses this challenge through structured documentation frameworks and standardized ETL pipelines, which enhance data governance and promote interoperability across systems. Even so, harmonizing legacy PACS data and oncological metadata will continue to demand refinements and shared best practices.</p>
            <p>From the outset, ethical and legal considerations have been central. Multi-step anonymization, robust audit logs, and clear governance models ensure compliance and traceability. However, achieving consistent ethics approvals and seamless cross-border data sharing still requires effort. The EHDS framework will be instrumental in easing these processes, but only if health data holders, research infrastructures, and users renew and align their governance models.</p>
            <p>The lessons learned from large-scale data integration reinforce the value of early legal and technical alignment. PRIMAGE, CHAIMELEON and ongoing clinical trials each illustrate how proactive regulatory planning and clear data management frameworks enable scalable, compliant data sharing.</p>
            <p>Long-term sustainability will rest on strict adherence to Findable, Accessible, Interoperable and Reusable (FAIR) principles, supported by reference and federated nodes, FAIR Data APIs, and persistent identifiers that guarantee data findability and reusability. Automating compliance checks and deploying secure cloud-based environments will be essential to handle ever-growing data volumes, while standardized metadata schemas will further support seamless reuse.</p>
            <p>Through the continuously refinement of its governance strategies and proactive alignment with the evolving regulatory landscape under the EHDS, EUCAIM is positioned to become a cornerstone for AI-driven oncology research. Its robust infrastructure not only accelerates the clinical translation of AI tools but also strengthens a culture of trustworthy, collaborative data sharing across Europe, reinforcing EUCAIM&#x2019;s pivotal role in shaping the future of cancer imaging and precision medicine. Central to this effort is the essential contribution of data providers, whose engagement ensures the availability, diversity, and quality of datasets reinforcing EUCAIM in shaping the future of precision medicine.</p>
        </sec>
        <sec>
            <title>Ethical approval statement</title>
            <p>This manuscript has been approved by the Research Ethics Committee on Medicinal Products of La Fe University and Polytechnic Hospital under registration number 2022-437-1. For the retrospective imaging data included in this study, no informed consent was required, an exemption of informed consent was submitted to and approved by the Ethics Committee.</p>
        </sec>
    </body>
    <back>
        <sec sec-type="data-availability">
            <title>Data availability</title>
            <p>No data are associated to this manuscript.</p>
        </sec>
        <ack>
            <title>Acknowledgments section</title>
            <p>Patricia Serrano Candelas, Silvia Flor Arnal, Lucas Espuig Peir&#xF3;, Antonio Ordu&#xF1;a Gal&#xE1;n, Cayetano Hernandez Mar&#xED;n and Javier Medina &#xC1;lvarez.</p>
        </ack>
        <ref-list>
            <ref id="ref-1">
                <label>1</label>
                <mixed-citation publication-type="journal">
                    <person-group person-group-type="author">

                        <name name-style="western">
                            <surname>Genovese</surname>
                            <given-names>S</given-names>
                        </name>

                        <name name-style="western">
                            <surname>Bengoa</surname>
                            <given-names>R</given-names>
                        </name>

                        <name name-style="western">
                            <surname>Bowis</surname>
                            <given-names>J</given-names>
                        </name>

                        <etal/>
</person-group>:
                    <article-title>The European Health Data Space: a step towards digital and integrated care systems.</article-title>
                    <source>

                        <italic toggle="yes">J Integr Care.</italic>
</source><year>2022</year>;<volume>30</volume>(<issue>4</issue>):<fpage>363</fpage>&#x2013;<lpage>372</lpage>.
                    <pub-id pub-id-type="doi">10.1108/JICA-11-2021-0059</pub-id></mixed-citation>
            </ref>
            <ref id="ref-2">
                <label>2</label>
                <mixed-citation publication-type="journal">
                    <person-group person-group-type="author">

                        <name name-style="western">
                            <surname>Lehne</surname>
                            <given-names>M</given-names>
                        </name>

                        <name name-style="western">
                            <surname>Sass</surname>
                            <given-names>J</given-names>
                        </name>

                        <name name-style="western">
                            <surname>Essenwanger</surname>
                            <given-names>A</given-names>
                        </name>

                        <etal/>
</person-group>:
                    <article-title>Why digital medicine depends on Interoperability.</article-title>
                    <source>

                        <italic toggle="yes">NPJ Digit Med.</italic>
</source><year>2019</year>;<volume>2</volume>(<issue>1</issue>): 79.
                    <pub-id pub-id-type="pmid">31453374</pub-id>
                    <pub-id pub-id-type="doi">10.1038/s41746-019-0158-1</pub-id>
                    <pub-id pub-id-type="pmcid">6702215</pub-id></mixed-citation>
            </ref>
            <ref id="ref-3">
                <label>3</label>
                <mixed-citation publication-type="journal">
                    <person-group person-group-type="author">

                        <name name-style="western">
                            <surname>Pierce</surname>
                            <given-names>HH</given-names>
                        </name>

                        <name name-style="western">
                            <surname>Dev</surname>
                            <given-names>A</given-names>
                        </name>

                        <name name-style="western">
                            <surname>Statham</surname>
                            <given-names>E</given-names>
                        </name>

                        <etal/>
</person-group>:
                    <article-title>Credit data generators for data reuse.</article-title>
                    <source>

                        <italic toggle="yes">Nature.</italic>
</source><year>2019</year>;<volume>570</volume>(<issue>7759</issue>):<fpage>30</fpage>&#x2013;<lpage>32</lpage>.
                    <pub-id pub-id-type="pmid">31164773</pub-id>
                    <pub-id pub-id-type="doi">10.1038/d41586-019-01715-4</pub-id>
                </mixed-citation>
            </ref>
            <ref id="ref-4">
                <label>4</label>
                <mixed-citation publication-type="journal">
                    <person-group person-group-type="author">

                        <name name-style="western">
                            <surname>Norgeot</surname>
                            <given-names>B</given-names>
                        </name>

                        <name name-style="western">
                            <surname>Glicksberg</surname>
                            <given-names>BS</given-names>
                        </name>

                        <name name-style="western">
                            <surname>Butte</surname>
                            <given-names>AJ</given-names>
                        </name>
</person-group>:
                    <article-title>A call for deep-learning healthcare.</article-title>
                    <source>

                        <italic toggle="yes">Nat Med.</italic>
</source><year>2019</year>;<volume>25</volume>(<issue>1</issue>):<fpage>14</fpage>&#x2013;<lpage>15</lpage>.
                    <pub-id pub-id-type="pmid">30617337</pub-id>
                    <pub-id pub-id-type="doi">10.1038/s41591-018-0320-3</pub-id>
                </mixed-citation>
            </ref>
            <ref id="ref-5">
                <label>5</label>
                <mixed-citation publication-type="journal">
                    <person-group person-group-type="author">

                        <name name-style="western">
                            <surname>Koh</surname>
                            <given-names>DM</given-names>
                        </name>

                        <name name-style="western">
                            <surname>Papanikolaou</surname>
                            <given-names>N</given-names>
                        </name>

                        <name name-style="western">
                            <surname>Bick</surname>
                            <given-names>U</given-names>
                        </name>

                        <etal/>
</person-group>:
                    <article-title>Artificial intelligence and machine learning in cancer imaging.</article-title>
                    <source>

                        <italic toggle="yes">Commun Med (Lond).</italic>
</source><year>2022</year>;<volume>2</volume>(<issue>1</issue>): 133.
                    <pub-id pub-id-type="pmid">36310650</pub-id>
                    <pub-id pub-id-type="doi">10.1038/s43856-022-00199-0</pub-id>
                    <pub-id pub-id-type="pmcid">9613681</pub-id>
                </mixed-citation>
            </ref>
            <ref id="ref-6">
                <label>6</label>
                <mixed-citation publication-type="journal">
                    <person-group person-group-type="author">

                        <name name-style="western">
                            <surname>Radclyffe</surname>
                            <given-names>C</given-names>
                        </name>

                        <name name-style="western">
                            <surname>Ribeiro</surname>
                            <given-names>M</given-names>
                        </name>

                        <name name-style="western">
                            <surname>Wortham</surname>
                            <given-names>RH</given-names>
                        </name>
</person-group>:
                    <article-title>The assessment list for trustworthy Artificial Intelligence: a review and recommendations.</article-title>
                    <source>

                        <italic toggle="yes">Front Artif Intell.</italic>
</source><year>2023</year>;<volume>6</volume>:
                    <elocation-id>1020592</elocation-id>.
                    <pub-id pub-id-type="pmid">36967834</pub-id>
                    <pub-id pub-id-type="doi">10.3389/frai.2023.1020592</pub-id>
                    <pub-id pub-id-type="pmcid">10034015</pub-id>
                </mixed-citation>
            </ref>
            <ref id="ref-7">
                <label>7</label>
                <mixed-citation publication-type="journal">
                    <person-group person-group-type="author">

                        <name name-style="western">
                            <surname>Gagliardi</surname>
                            <given-names>D</given-names>
                        </name>

                        <etal/>
</person-group>:
                    <article-title>AI4HI: AI and health imaging research priorities.</article-title>
                    <source>

                        <italic toggle="yes">European Journal of Radiology Open.</italic>
</source><year>2023</year>.</mixed-citation>
            </ref>
            <ref id="ref-8">
                <label>8</label>
                <mixed-citation publication-type="journal">
                    <person-group person-group-type="author">

                        <name name-style="western">
                            <surname>Kondylakis</surname>
                            <given-names>H</given-names>
                        </name>

                        <name name-style="western">
                            <surname>Ciarrocchi</surname>
                            <given-names>E</given-names>
                        </name>

                        <name name-style="western">
                            <surname>Cerda-Alberich</surname>
                            <given-names>L</given-names>
                        </name>

                        <etal/>
</person-group>:
                    <article-title>Position of the AI for Health Imaging (AI4HI) network on metadata models for imaging biobanks.</article-title>
                    <source>

                        <italic toggle="yes">Eur Radiol Exp.</italic>
</source><year>2022</year>;<volume>6</volume>(<issue>1</issue>): 29.
                    <pub-id pub-id-type="pmid">35773546</pub-id>
                    <pub-id pub-id-type="doi">10.1186/s41747-022-00281-1</pub-id>
                    <pub-id pub-id-type="pmcid">9247122</pub-id>
                </mixed-citation>
            </ref>
            <ref id="ref-9">
                <label>9</label>
                <mixed-citation publication-type="book">
                    <person-group person-group-type="author">

                        <name name-style="western">
                            <surname>Voigt</surname>
                            <given-names>P</given-names>
                        </name>

                        <name name-style="western">
                            <surname>Von der Bussche</surname>
                            <given-names>A</given-names>
                        </name>
</person-group>:
                    <article-title>The EU General Data Protection Regulation (GDPR)</article-title>. A Practical Guide, 1st Ed., Cham: Springer International Publishing,<year>2017</year>;<volume>10</volume>(<issue>3152676</issue>):<fpage>10</fpage>&#x2013;<lpage>5555</lpage>.</mixed-citation>
            </ref>
            <ref id="ref-10">
                <label>10</label>
                <mixed-citation publication-type="journal">
                    <person-group person-group-type="author">

                        <name name-style="western">
                            <surname>Wilkinson</surname>
                            <given-names>MD</given-names>
                        </name>

                        <name name-style="western">
                            <surname>Dumontier</surname>
                            <given-names>M</given-names>
                        </name>

                        <name name-style="western">
                            <surname>Aalbersberg</surname>
                            <given-names>IJ</given-names>
                        </name>

                        <etal/>
</person-group>:
                    <article-title>The FAIR guiding principles for scientific data management and stewardship.</article-title>
                    <source>

                        <italic toggle="yes">Sci Data.</italic>
</source><year>2016</year>;<volume>3</volume>(<issue>1</issue>): 160018.
                    <pub-id pub-id-type="pmid">26978244</pub-id>
                    <pub-id pub-id-type="doi">10.1038/sdata.2016.18</pub-id>
                    <pub-id pub-id-type="pmcid">4792175</pub-id>
                </mixed-citation>
            </ref>
            <ref id="ref-11">
                <label>11</label>
                <mixed-citation publication-type="journal">
                    <person-group person-group-type="author">

                        <name name-style="western">
                            <surname>Segrelles Quilis</surname>
                            <given-names>JD</given-names>
                        </name>

                        <name name-style="western">
                            <surname>Lozano</surname>
                            <given-names>P</given-names>
                        </name>

                        <name name-style="western">
                            <surname>Blanco-Sanchez</surname>
                            <given-names>A</given-names>
                        </name>

                        <etal/>
</person-group>:
                    <article-title>Experiences on using the CHAIMELEON secure processing environment in an open competition addressing five AI challenges.</article-title>
                    <pub-id pub-id-type="doi">10.2139/ssrn.5574632</pub-id>
                </mixed-citation>
            </ref>
            <ref id="ref-12">
                <label>12</label>
                <mixed-citation publication-type="journal">
                    <person-group person-group-type="author">

                        <name name-style="western">
                            <surname>Gyrard</surname>
                            <given-names>A</given-names>
                        </name>

                        <name name-style="western">
                            <surname>Abedian</surname>
                            <given-names>S</given-names>
                        </name>

                        <name name-style="western">
                            <surname>Gribbon</surname>
                            <given-names>P</given-names>
                        </name>

                        <etal/>
</person-group>:
                    <article-title>Lessons learned from European health data projects with cancer use cases: implementation of health standards and internet of things semantic interoperability.</article-title>
                    <source>

                        <italic toggle="yes">J Med Internet Res.</italic>
</source><year>2025</year>;<volume>27</volume>:
                    <elocation-id>e66273</elocation-id>.
                    <pub-id pub-id-type="pmid">40126534</pub-id>
                    <pub-id pub-id-type="doi">10.2196/66273</pub-id>
                    <pub-id pub-id-type="pmcid">11976176</pub-id>
                </mixed-citation>
            </ref>
            <ref id="ref-13">
                <label>13</label>
                <mixed-citation publication-type="journal">
                    <person-group person-group-type="author">

                        <name name-style="western">
                            <surname>Mart&#xED;-Bonmat&#xED;</surname>
                            <given-names>L</given-names>
                        </name>

                        <name name-style="western">
                            <surname>Blanquer</surname>
                            <given-names>I</given-names>
                        </name>

                        <name name-style="western">
                            <surname>Tsiknakis</surname>
                            <given-names>M</given-names>
                        </name>

                        <etal/>
</person-group>:
                    <article-title>Empowering cancer research in Europe: the EUCAIM cancer imaging infrastructure.</article-title>
                    <source>

                        <italic toggle="yes">Insights Imaging.</italic>
</source><year>2025</year>;<volume>16</volume>(<issue>1</issue>): 47.
                    <pub-id pub-id-type="pmid">39992532</pub-id>
                    <pub-id pub-id-type="doi">10.1186/s13244-025-01913-x</pub-id>
                    <pub-id pub-id-type="pmcid">11850660</pub-id>
                </mixed-citation>
            </ref>
        </ref-list>
    </back>
    <sub-article article-type="reviewer-report" id="report64762">
        <front-stub>
            <article-id pub-id-type="doi">10.21956/openreseurope.23735.r64762</article-id>
            <title-group>
                <article-title>Reviewer response for version 2</article-title>
            </title-group>
            <contrib-group>
                <contrib contrib-type="author">
                    <name>
                        <surname>Neha</surname>
                        <given-names>Neha</given-names>
                    </name>
                    <xref ref-type="aff" rid="r64762a1">1</xref>
                    <role>Referee</role>
                    <uri content-type="orcid">https://orcid.org/0009-0004-3702-2382</uri>
                </contrib>
                <contrib contrib-type="author">
                    <name>
                        <surname>Shukla</surname>
                        <given-names>Deepak Kumar</given-names>
                    </name>
                    <xref ref-type="aff" rid="r64762a2">2</xref>
                    <role>Co-referee</role>
                </contrib>
                <aff id="r64762a1">
                    <label>1</label>Kent State University, Kent, USA</aff>
                <aff id="r64762a2">
                    <label>2</label>Rutgers Business School, Rutgers University Newark Business School (Ringgold ID: 33869), Newark, Jersey, USA</aff>
            </contrib-group>
            <author-notes>
                <fn fn-type="conflict">
                    <p>
                        <bold>Competing interests: </bold>No competing interests were disclosed.</p>
                </fn>
            </author-notes>
            <pub-date pub-type="epub">
                <day>2</day>
                <month>12</month><year>2025</year>
            </pub-date>
            <permissions>
                <copyright-statement>Copyright: &#xA9; 2025 Neha N and Shukla DK</copyright-statement>
                <copyright-year>2025</copyright-year>
                <license xlink:href="https://creativecommons.org/licenses/by/4.0/">
                    <license-p>This is an open access peer review report distributed under the terms of the Creative Commons Attribution Licence, which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.</license-p>
                </license>
            </permissions>
            <related-article ext-link-type="doi" id="relatedArticleReport64762" related-article-type="peer-reviewed-article" xlink:href="10.12688/openreseurope.21016.2"/>
            <custom-meta-group>
                <custom-meta>
                    <meta-name>recommendation</meta-name>
                    <meta-value>approve</meta-value>
                </custom-meta>
            </custom-meta-group>
        </front-stub>
        <body>
            <p>Author has made all the suggested updates.</p>
            <p>Is the case presented with sufficient detail to be useful for teaching or other practitioners?</p>
            <p>Partly</p>
            <p>Is the work clearly and accurately presented and does it cite the current literature?</p>
            <p>Yes</p>
            <p>If applicable, is the statistical analysis and its interpretation appropriate?</p>
            <p>Not applicable</p>
            <p>Are all the source data underlying the results available to ensure full reproducibility?</p>
            <p>Partly</p>
            <p>Are the conclusions drawn adequately supported by the results?</p>
            <p>Yes</p>
            <p>Is the background of the case&#x2019;s history and progression described in sufficient detail?</p>
            <p>Yes</p>
            <p>Reviewer Expertise:</p>
            <p>Artificial IntelligenceComputer VisionImage ProcessingDeep learningMedical Imaging</p>
            <p>We confirm that we have read this submission and believe that we have an appropriate level of expertise to confirm that it is of an acceptable scientific standard.</p>
        </body>
    </sub-article>
    <sub-article article-type="reviewer-report" id="report62149">
        <front-stub>
            <article-id pub-id-type="doi">10.21956/openreseurope.22734.r62149</article-id>
            <title-group>
                <article-title>Reviewer response for version 1</article-title>
            </title-group>
            <contrib-group>
                <contrib contrib-type="author">
                    <name>
                        <surname>Montero</surname>
                        <given-names>Ana M. Barrag&#xE1;n</given-names>
                    </name>
                    <xref ref-type="aff" rid="r62149a1">1</xref>
                    <role>Referee</role>
                    <uri content-type="orcid">https://orcid.org/0000-0002-9485-3076</uri>
                </contrib>
                <aff id="r62149a1">
                    <label>1</label>Molecular Imaging, Radiation and Oncology (MIRO) Laboratory, UCLouvain, UCLouvain, Belgium</aff>
            </contrib-group>
            <author-notes>
                <fn fn-type="conflict">
                    <p>
                        <bold>Competing interests: </bold>No competing interests were disclosed.</p>
                </fn>
            </author-notes>
            <pub-date pub-type="epub">
                <day>22</day>
                <month>11</month><year>2025</year>
            </pub-date>
            <permissions>
                <copyright-statement>Copyright: &#xA9; 2025 Montero AMB</copyright-statement>
                <copyright-year>2025</copyright-year>
                <license xlink:href="https://creativecommons.org/licenses/by/4.0/">
                    <license-p>This is an open access peer review report distributed under the terms of the Creative Commons Attribution Licence, which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.</license-p>
                </license>
            </permissions>
            <related-article ext-link-type="doi" id="relatedArticleReport62149" related-article-type="peer-reviewed-article" xlink:href="10.12688/openreseurope.21016.1"/>
            <custom-meta-group>
                <custom-meta>
                    <meta-name>recommendation</meta-name>
                    <meta-value>approve-with-reservations</meta-value>
                </custom-meta>
            </custom-meta-group>
        </front-stub>
        <body>
            <p>
                <bold>Review of the manuscript Open Research Europe 2025, 5:310 &#x201C;How the first medical imaging cancer atlas EUCAIM was populated: the experience of a reference hospital&#x201D;</bold>
            </p>
            <p> 
                <bold>The manuscript describes the EUCAIM ecosystem and the methodology to submit images, illustrated with the experience of a hospital in the consortium. The article is well written and clear, and it is suitable for publication in the journal. However, I have some comments that I believe can help to increase the quality of the manuscript and before final publication.&#xA0;&#xA0;</bold>
            </p>
            <p> 
                <bold>MAJOR COMMENTS&#xA0;</bold> 
                <list list-type="order">
                    <list-item>
                        <p>
                            <bold>Improve the contextualisation and illustrate the state of the art. In the first paragraph, they describe the current ecosystem for data sharing and the limitations, highlighting the need for an initiative like EUCAIM. While the paragraph is well written, I am missing a few concrete examples and references that support the arguments given in the paragraph. For instance, the authors mention that &#x201C;many research initiatives have created imaging repositories, but they are often project-specific, temporally limited, &#x2026;&#x201D; but no example is given. Please provide at least two or three examples. Also what is the potential of EUCAIM with respect to other existing initiatives for large-scale data collection like TCIA (
                                <ext-link ext-link-type="uri" xlink:href="https://www.cancerimagingarchive.net/">https://www.cancerimagingarchive.net/</ext-link>), DESIRE (
                                <ext-link ext-link-type="uri" xlink:href="https://www.straaleterapi.dk/en/desire/about-desire/">https://www.straaleterapi.dk/en/desire/about-desire/</ext-link>), GrandChallenges (
                                <ext-link ext-link-type="uri" xlink:href="https://grand-challenge.org/challenges/">https://grand-challenge.org/challenges/</ext-link>), etc I think a paragraph discussing this and giving specific examples will help the reader to better see the potential and value of EUCAIM</bold>
                        </p>
                    </list-item>
                </list> &#xA0; 
                <list list-type="order">
                    <list-item>
                        <p>
                            <bold>Describe more in detail the system for hosting&#xA0; and using the data. The authors distinguish between two options (materials and method, paragraph 1): 1) DSA, where the data stays in the system, or DTA, where the data goes to the federated node. Can you provide more technical information about this part? Or either reference to already published protocols/papers describing this part? Some questions here: Why these two options and not only DTA directly? For hospitals choosing 1), is it up to them to buy a server to host the data? How is the connection then to other hospitals willing to use the data through the DSA? Does EUCAIM provide a technical team to make all these installations in the local computers or the hospital should do it? How can other hospitals use the DSA data to train models if the data did not leave the hospital? For DTA data, can I download the data into my servers to perform research (e.g. trained AI models)?</bold>
                        </p>
                        <p>
                            <bold> I understand that the paper here focuses on the population of the atlas, but some clarification (a paragraph will be enough) here is important, so that the community understands how the data can be used later. Otherwise, it seems a very nice platform to store data but the article does not give any clue about the potential to use the data later.&#xA0;</bold>
                        </p>
                    </list-item>
                </list> &#xA0; 
                <list list-type="order">
                    <list-item>
                        <p>
                            <bold>Disclose or add reference for already available material. This is published in Open Research journal, so I guess that the goal is to make as open as possible the results presented here. There are many parts where you mention that the authors (or the EUCAIM team) has developed a lot of material, but no reference is given. For instance, in material and methods &#x201C;For data standardization, all DICOM files and metadata were validated for compliance with EUCAIM structure&#x201D;, &#x201C;The DICOM File Integrity Checker developed by our group&#x201D; &#x2192; is this material published elsewhere or made open-source through a github/github or similar repository? The amount of work that has been developed here has an incredible value for the community and should be made public whenever possible.&#xA0;</bold>
                        </p>
                    </list-item>
                </list> &#xA0; 
                <list list-type="order">
                    <list-item>
                        <p>
                            <bold>Pseudonymisation versus Anonymisation. I have a question regarding this, since both are mentioned during the text, it is not clear to me if the data uploaded is pseudonymised, and there is a log file somewhere (in the hospital submitting the data) that keeps a link (e.g. PatientID) between the data submitted and the real patient identity, or the data is fully anonymised. If the data is totally anonymised and there is no log-file, how do we prevent errors in the future if the hospital submit the same patient (e.g. submission in 2025 for project X, new patient ID projectX_01 - submission in 2026 of same patient for project Y, new patient ID project Y_01). Also, besides removing DICOM tags, can you describe if any action on the images has been done or not? There are some people that advocate to remove facial information from head-and-neck scans, but this can affect the potential use of the data (e.g. in head-and-neck cancer patients).&#xA0;</bold>
                        </p>
                    </list-item>
                </list> &#xA0; 
                <list list-type="order">
                    <list-item>
                        <p>
                            <bold>Data quality and documentation. You do mention DICOM compliance, anonymisation, etc, but what about data quality and documentation?&#xA0;</bold> 
                            <list list-type="order">
                                <list-item>
                                    <p>
                                        <bold>About data quality, is there any metadata or any quality check to measure the quality of the submitted data? I am thinking more on the annotations of the data, for instance, if the annotations have been done by experts and following consensus international guidelines, etc A concrete example would be the adherence to contouring guidelines in radiotherapy images. Data annotations not being compliant to these guidelines and used to train models might carry future problems. Also, we could think about data quality of the images, if a hospital submit images with very low resolution, artefacts, etc. A way to go would be to enable a community rating for each dataset &#x2026;</bold>
                                    </p>
                                </list-item>
                                <list-item>
                                    <p>
                                        <bold>About data documentation. I am putting myself on the user side, and the first question that comes to my mind is: how do I find and decide which data I use for my project? Having data documentation is very important, like a Data Sheet (similar to what they have done in 
                                            <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/1803.09010">https://arxiv.org/abs/1803.09010</ext-link>, 
                                            <ext-link ext-link-type="uri" xlink:href="https://datanutrition.org/">https://datanutrition.org/</ext-link>, &#x2026;). Is there any similar documentation standard in EUCAIM? </bold>
                                    </p>
                                </list-item>
                            </list> </p>
                    </list-item>
                    <list-item>
                        <p>
                            <bold>In general some paragraphs feel too much &#x201C;GPT-style&#x201D;, sometimes repeating the same but with different words. Please review and try to make the text less redundant and more concise whenever possible.&#xA0;</bold>
                        </p>
                    </list-item>
                </list>
            </p>
            <p>Is the case presented with sufficient detail to be useful for teaching or other practitioners?</p>
            <p>Partly</p>
            <p>Is the work clearly and accurately presented and does it cite the current literature?</p>
            <p>Partly</p>
            <p>If applicable, is the statistical analysis and its interpretation appropriate?</p>
            <p>Not applicable</p>
            <p>Are all the source data underlying the results available to ensure full reproducibility?</p>
            <p>Not applicable</p>
            <p>Are the conclusions drawn adequately supported by the results?</p>
            <p>Yes</p>
            <p>Is the background of the case&#x2019;s history and progression described in sufficient detail?</p>
            <p>Yes</p>
            <p>Reviewer Expertise:</p>
            <p>Artificial Intelligence for medical imaging, radiotherapy treatment planning</p>
            <p>I confirm that I have read this submission and believe that I have an appropriate level of expertise to confirm that it is of an acceptable scientific standard, however I have significant reservations, as outlined above.</p>
        </body>
        <sub-article article-type="response" id="comment4908-62149">
            <front-stub>
                <contrib-group>
                    <contrib contrib-type="author">
                        <name>
                            <surname>Penades Blasco</surname>
                            <given-names>Ana</given-names>
                        </name>
                        <aff>Biomedical Imaging Research Group, La Fe Health Research Institute, Valencia, Spain, Spain</aff>
                    </contrib>
                </contrib-group>
                <author-notes>
                    <fn fn-type="conflict">
                        <p>
                            <bold>Competing interests: </bold>No competing interests were disclosed.</p>
                    </fn>
                </author-notes>
                <pub-date pub-type="epub">
                    <day>3</day>
                    <month>12</month><year>2025</year>
                </pub-date>
            </front-stub>
            <body>
                <p>
                    <bold>
                        <underline>Response to the REVIEWER #3 COMMENTS</underline>
                    </bold>
                </p>
                <p> Thank you very much for your comments and feedback. We have worked on improving the manuscript with your suggestions and comments.</p>
                <p> 
                    <bold>Reviewer #3:</bold>
                </p>
                <p> 
                    <bold>Reviewer 3.1</bold>: Improve the contextualisation and illustrate the state of the art. In the first paragraph, they describe the current ecosystem for data sharing and the limitations, highlighting the need for an initiative like EUCAIM. While the paragraph is well written, I am missing a few concrete examples and references that support the arguments given in the paragraph. For instance, the authors mention that &#x201C;many research initiatives have created imaging repositories, but they are often project-specific, temporally limited, &#x2026;&#x201D; but no example is given. Please provide at least two or three examples. Also what is the potential of EUCAIM with respect to other existing initiatives for large-scale data collection like TCIA (
                    <ext-link ext-link-type="uri" xlink:href="https://www.cancerimagingarchive.net/">https://www.cancerimagingarchive.net/</ext-link>), DESIRE (
                    <ext-link ext-link-type="uri" xlink:href="https://www.straaleterapi.dk/en/desire/about-desire/">https://www.straaleterapi.dk/en/desire/about-desire/</ext-link>), GrandChallenges (
                    <ext-link ext-link-type="uri" xlink:href="https://grand-challenge.org/challenges/">https://grand-challenge.org/challenges/</ext-link>), etc I think a paragraph discussing this and giving specific examples will help the reader to better see the potential and value of EUCAIM</p>
                <p> 
                    <bold>Authors:</bold> Following the comments of the reviewer we now incorporated several sentences at the Introduction: &#x201C;Multiple initiatives have previously generated valuable imaging repositories, examples include some Horizon Europe projects providing temporal available datasets such as Predictive In-silico Multiscale Analytics to support cancer personalized diagnosis and prognosis, empowered by imaging biomarkers - PRIMAGE, Accelerating the lab to market transition of AI tools for cancer management - CHAIMELEON, Novel pan-European imaging platform for artificial intelligence advances in oncology - EuCanImage, An AI Platform integrating imaging data and models, supporting precision care through prostate cancer&#x2019;s continuum - ProCancer-I, and A multimodal AI-based toolbox and an interoperable health imaging repository for the empowerment of imaging analysis related to the diagnosis, prediction and follow-up of cancer - INCISIVE. And also some existing repositories such as The Cancer Imaging Archive (TCIA), which provides curated datasets but largely centred on disease-specific collections; GrandChallenges, where datasets are created to support competitions and are rarely updated or expanded once each challenge is completed; and domain-focused initiatives such as DESIRE in radiotherapy, which target specific clinical use cases and remain confined to their original scope. These projects have demonstrated the scientific value of shared imaging data, but they lack the harmonised governance, long-term sustainability mechanisms, and structured clinical context that are required for large-scale, reproducible AI research. EUCAIM complements these existing resources since it provides a hybrid federated&#x2013;centralised infrastructure with common standards for anonymisation, metadata, clinical linkage and data quality, enabling continuous expansion of the atlas and supporting cross-border and GDPR compliant analysis at scale.&#x201D;</p>
                <p> </p>
                <p> 
                    <bold>Reviewer 3.2</bold>: Describe more in detail the system for hosting and using the data. The authors distinguish between two options (materials and method, paragraph 1): 1) DSA, where the data stays in the system, or DTA, where the data goes to the federated node. Can you provide more technical information about this part? Or reference to already published protocols/papers describing this part? Some questions here: Why these two options and not only DTA directly? For hospitals choosing 1), is it up to them to buy a server to host the data? How is the connection then to other hospitals willing to use the data through the DSA? Does EUCAIM provide a technical team to make all these installations in the local computers, or the hospital should do it? How can other hospitals use the DSA data to train models if the data did not leave the hospital? For DTA data, can I download the data into my servers to perform research (e.g. trained AI models)?</p>
                <p> I understand that the paper here focuses on the population of the atlas, but some clarification (a paragraph will be enough) here is important, so that the community understands how the data can be used later. Otherwise, it seems a very nice platform to store data, but the article does not give any clue about the potential to use the data later
                    <bold>.</bold>
                </p>
                <p> 
                    <bold>Authors:</bold> To address this comment, we have included the following paragraph on page 6: &#x201C;The choice between a DSA and a DTA reflects the heterogeneous technical capabilities and data governance preferences of participating hospitals. Some institutions require that data processing remains fully under their control; for these cases, EUCAIM enables the deployment of a federated node within the hospital infrastructure, hosted on local servers and integrated with the EUCAIM platform through secure, encrypted channels. This environment operates as a Secure Processing Environment under the EHDS framework, allowing authorised users to run analytics and train models locally without any image data leaving the institution. EUCAIM provides technical specifications, containerised services, and remote support to assist hospitals in the installation and configuration of these nodes, although the hardware is typically procured and maintained by the institution according to its internal policies. Other hospitals opt for a DTA, transferring data to the EUCAIM reference node where harmonisation, storage and computation are centrally managed. While software models trained within this environment may be exported subject to the Access Committee approval, original imaging data are never downloadable to external servers. This dual approach offers flexibility for centres with different resources and legal constraints while ensuring that all data, whether local or centralised, can be incorporated into cross-site analyses through a unified and privacy-preserving architecture.&#x201D;</p>
                <p> </p>
                <p> 
                    <bold>Reviewer 3.3: </bold>Disclose or add reference for already available material. This is published in Open Research journal, so I guess that the goal is to make as open as possible the results presented here. There are many parts where you mention that the authors (or the EUCAIM team) have developed a lot of material, but no reference is given. For instance, in material and methods &#x201C;For data standardization, all DICOM files and metadata were validated for compliance with EUCAIM structure&#x201D;, &#x201C;The DICOM File Integrity Checker developed by our group&#x201D; &#x2192; is this material published elsewhere or made open-source through a github/github or similar repository? The amount of work that has been developed here has an incredible value for the community and should be made public whenever possible.&#xA0;</p>
                <p> 
                    <bold>Authors: </bold>Thank you for pointing this out. Additional references have been incorporated on page 6 to clarify the availability of the tools and materials mentioned. 
                    <list list-type="bullet">
                        <list-item>
                            <p> 
                                <list list-type="bullet">
                                    <list-item>
                                        <p>DICOM File Integrity Checker: The DICOM File Integrity Checker developed by the group is not a public tool; however, it is accessible to EUCAIM users for data preprocessing within the project. The manuscript now includes a reference to its entry in the EUCAIM bio.tools catalogue:</p>
                                    </list-item>
                                </list> </p>
                        </list-item>
                    </list> &#xA0;
                    <ext-link ext-link-type="uri" xlink:href="https://bio.tools/dicom_file_integrity_checker_by_gibi230">https://bio.tools/dicom_file_integrity_checker_by_gibi230</ext-link> 
                    <list list-type="bullet">
                        <list-item>
                            <p> 
                                <list list-type="bullet">
                                    <list-item>
                                        <p>EUCAIM Handbook: A reference to the EUCAIM Handbook has been added. This resource provides a detailed description of the data standardization workflow and the tools available within the project: 
                                            <ext-link ext-link-type="uri" xlink:href="https://eucaim.gitbook.io/handbook">https://eucaim.gitbook.io/handbook</ext-link>
                                        </p>
                                    </list-item>
                                </list> </p>
                        </list-item>
                    </list> 
                    <bold>Reviewer 3.4: </bold>Pseudonymisation versus Anonymisation. I have a question regarding this, since both are mentioned during the text, it is not clear to me if the data uploaded is pseudonymised, and there is a log file somewhere (in the hospital submitting the data) that keeps a link (e.g. PatientID) between the data submitted and the real patient identity, or the data is fully anonymised. If the data is totally anonymised and there is no log-file, how do we prevent errors in the future if the hospital submit the same patient (e.g. submission in 2025 for project X, new patient ID projectX_01 - submission in 2026 of same patient for project Y, new patient ID project Y_01). Also, besides removing DICOM tags, can you describe if any action on the images has been done or not? There are some people that advocate to remove facial information from head-and-neck scans, but this can affect the potential use of the data (e.g. in head-and-neck cancer patients).
                    <bold>&#xA0;</bold> 
                    <bold>Authors: </bold>To address this comment, we have included the following explanation and sentences on The Data extraction and exposure pipeline (page 6): &#x201C;The data preparation workflow includes a two-stage privacy-preserving process: 
                    <list list-type="order">
                        <list-item>
                            <p>Local pseudonymisation:</p>
                        </list-item>
                    </list> Pseudonymisation is performed on a restricted-access Virtual Machine by authorized personnel of the Experimental Radiology and Imaging Biomarkers Platform (PREBI) using an in-house tool developed by the Biomedical Imaging Research Group (GIBI230). The process includes: replacement of PatientID, PatientName and AccessionNumber with specific pseudonym and hashes using Blake2b, renaming of folder structures to remove personally identifiable information (PII), removal of public and private DICOM metadata that may contain PII, exclusion of screenshots (ImageType = SCREEN SAVE) and manual review of secondary captures (DERIVED or SECONDARY). This step ensures consistent pseudonymised identifiers for repeated submissions of the same patient at the hospital level. The mapping between original and pseudonymised identifiers is maintained only temporarily by the hospital IT service and is not accessible externally. 
                    <list list-type="order">
                        <list-item>
                            <p>EUCAIM anonymization:</p>
                        </list-item>
                    </list> Once pseudonymised data are transferred to the Pseudonymised Medical Imaging Repository, the 
                    <ext-link ext-link-type="uri" xlink:href="https://bio.tools/lethe_dicom_anonymizer">EUCAIM anonymization pipeline</ext-link> is applied. This pipeline enforces the 
                    <ext-link ext-link-type="uri" xlink:href="https://github.com/cbml-forth/lethe_anon_pipeline/blob/main/ctp/anon.script">EUCAIM DICOM Anonymization Profile</ext-link> and generates a unique hash per patient, linked to the project and site, with no retained traceability. Sensitive DICOM headers are removed, and OCR-based pixel-level text detection can be applied to eliminate &#x201C;burned-in&#x201D; personal information. 
                    <ext-link ext-link-type="uri" xlink:href="https://bio.tools/dicom_image_similarity-duplicate_checker">File integrity checks and pixel-level duplicate detection</ext-link> can be performed to prevent repeated inclusion of the same images across different projects.&#x201D; We have also reorganized this section adding the following sentence: &#x201C;This two-layer anonymization approach mitigates re-identification risks, aligning with procedures recommended by the 
                    <ext-link ext-link-type="uri" xlink:href="https://www.aepd.es/">Spanish Data Protection Agency</ext-link> and 
                    <ext-link ext-link-type="uri" xlink:href="https://www.pdpc.gov.sg/help-and-resources/2018/01/basic-anonymisation">best practices from the Singaporean authority</ext-link> ( Figure 1). &#xA0;</p>
                <p> </p>
                <p> 
                    <ext-link ext-link-type="uri" xlink:href="https://openreseurope-files.f1000.com/linked/263811.image_1.gif">
                        <italic>https://openreseurope-files.f1000.com/linked/263811.image_1.gif</italic>
                    </ext-link>
                </p>
                <p> </p>
                <p> &#xA0;
                    <bold>Figure 1. Steps for anonymization.</bold> Source: Adapted from AEPD. Guide to basic anonymization. Prepared by the National Data Protection Authority of Singapore (PDPC - Personal Data Protection Commission Singapore. Additionally, after transferring it to Pseudonymised Medical Imaging Repository and anonymising the datasets, the 
                    <ext-link ext-link-type="uri" xlink:href="https://bio.tools/dicom_file_integrity_checker_by_gibi230">DICOM File Integrity Checker</ext-link> developed by our group was used to detect corrupted or missing files. Data were finally ingested into EUCAIM by transferring them to the Reference Node through the 
                    <ext-link ext-link-type="uri" xlink:href="https://quibim.com/qp-insights/">QP-Insights API</ext-link>, or by sharing them through a federated node.. For data standardization, all DICOM files and 
                    <ext-link ext-link-type="uri" xlink:href="https://zenodo.org/records/15435112">metadata</ext-link> were validated for compliance with EUCAIM&#x2019;s structure and 
                    <ext-link ext-link-type="uri" xlink:href="https://zenodo.org/records/15558108">interoperability standards</ext-link>. The data preparation process and the tools involved are defined in the 
                    <ext-link ext-link-type="uri" xlink:href="https://eucaim.gitbook.io/handbook">EUCAIM Handbook</ext-link>. Additionally, the metadata of all datasets were registered in the 
                    <ext-link ext-link-type="uri" xlink:href="https://catalogue.eucaim.cancerimage.eu/Eucaim/eucaim-ui/#/catalogue">EUCAIM Public Catalogue</ext-link>, which follows the Health DCAT-AP standard and ensures compliance with FAIR principles at the dataset level. An additional layer of dataset discoverability is provided to EUCAIM Data Users through the 
                    <ext-link ext-link-type="uri" xlink:href="https://explorer.eucaim.cancerimage.eu/">Federated Query tool</ext-link>, which enables them to perform queries based on specific criteria and retrieve the number of cases (as aggregated numerical results) that meet those criteria&#x201D;</p>
                <p> </p>
                <p> 
                    <bold>Reviewer 3.5: </bold>Data quality and documentation. You do mention DICOM compliance, anonymisation, etc, but what about data quality and documentation?&#xA0; About data quality, is there any metadata or any quality check to measure the quality of the submitted data? I am thinking more on the annotations of the data, for instance, if the annotations have been done by experts and following consensus international guidelines, etc A concrete example would be the adherence to contouring guidelines in radiotherapy images. Data annotations not being compliant to these guidelines and used to train models might carry future problems. Also, we could think about data quality of the images, if a hospital submit images with very low resolution, artefacts, etc. A way to go would be to enable a community rating for each dataset &#x2026; About data documentation. I am putting myself on the user side, and the first question that comes to my mind is: how do I find and decide which data I use for my project? Having data documentation is very important, like a Data Sheet (similar to what they have done in&#xA0;
                    <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/1803.09010">https://arxiv.org/abs/1803.09010</ext-link>,&#xA0;
                    <ext-link ext-link-type="uri" xlink:href="https://datanutrition.org/">https://datanutrition.org/</ext-link>, &#x2026;). Is there any similar documentation standard in EUCAIM?</p>
                <p> 
                    <bold>Authors: </bold>We have reviewed your relevant comment about data quality and we would like to add that efforts are ongoing to define and implement European-level guidelines for data quality, such as those being developed within the 
                    <ext-link ext-link-type="uri" xlink:href="https://quantumproject.eu/">QUANTUM project</ext-link>. At the time of dataset preprocessing for this manuscript, these guidelines were not yet applied. In addition, EUCAIM Data Users have access to other tools and workflows to assess data quality. We have included the following comment on the manuscript on page 8: &#x201C;This rigorous ETL pipeline ensures scalability, data quality through the access to tools and workflows within EUCAIM&#x201D; In relation with data augmentation, we would like to confirm that EUCAIM provides dataset metadata in the Public Catalogue, which follows the Health DCAT-AP standard and FAIR principles. Researchers can explore datasets and evaluate suitability using the Federated Query tool, which returns aggregated counts for specific criteria, supporting informed dataset selection. We have included the following comment on the manuscript on page 6:&#x201D; The data preparation process and the tools involved are defined in the 
                    <ext-link ext-link-type="uri" xlink:href="https://eucaim.gitbook.io/handbook">EUCAIM Handbook</ext-link>. Additionally, the metadata of all datasets were registered in the 
                    <ext-link ext-link-type="uri" xlink:href="https://catalogue.eucaim.cancerimage.eu/Eucaim/eucaim-ui/#/catalogue">EUCAIM Public Catalogue</ext-link>, which follows the Health DCAT-AP standard and ensures compliance with FAIR principles at the dataset level. An additional layer of dataset discoverability is provided to EUCAIM Data Users through the 
                    <ext-link ext-link-type="uri" xlink:href="https://explorer.eucaim.cancerimage.eu/">Federated Query tool</ext-link>, which enables them to perform queries based on specific criteria and retrieve the number of cases (as aggregated numerical results) that meet those criteria&#x201D;</p>
            </body>
        </sub-article>
    </sub-article>
    <sub-article article-type="reviewer-report" id="report63504">
        <front-stub>
            <article-id pub-id-type="doi">10.21956/openreseurope.22734.r63504</article-id>
            <title-group>
                <article-title>Reviewer response for version 1</article-title>
            </title-group>
            <contrib-group>
                <contrib contrib-type="author">
                    <name>
                        <surname>Strzelecki</surname>
                        <given-names>Michal</given-names>
                    </name>
                    <xref ref-type="aff" rid="r63504a1">1</xref>
                    <role>Referee</role>
                </contrib>
                <aff id="r63504a1">
                    <label>1</label>Lodz University of Technology, Lodz, Poland</aff>
            </contrib-group>
            <author-notes>
                <fn fn-type="conflict">
                    <p>
                        <bold>Competing interests: </bold>No competing interests were disclosed.</p>
                </fn>
            </author-notes>
            <pub-date pub-type="epub">
                <day>11</day>
                <month>11</month><year>2025</year>
            </pub-date>
            <permissions>
                <copyright-statement>Copyright: &#xA9; 2025 Strzelecki M</copyright-statement>
                <copyright-year>2025</copyright-year>
                <license xlink:href="https://creativecommons.org/licenses/by/4.0/">
                    <license-p>This is an open access peer review report distributed under the terms of the Creative Commons Attribution Licence, which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.</license-p>
                </license>
            </permissions>
            <related-article ext-link-type="doi" id="relatedArticleReport63504" related-article-type="peer-reviewed-article" xlink:href="10.12688/openreseurope.21016.1"/>
            <custom-meta-group>
                <custom-meta>
                    <meta-name>recommendation</meta-name>
                    <meta-value>approve</meta-value>
                </custom-meta>
            </custom-meta-group>
        </front-stub>
        <body>
            <p>The paper addresses a highly significant issue concerning access to medical imaging data. Such data are essential for the development and implementation of AI algorithms that support diagnostic imaging. Through the EUCAIM project, an effective method for accessing these data has been devised, ensuring compliance with extremely stringent legal and ethical requirements while overcoming technical challenges&#x2014;primarily related to incompatibilities among hospital PACS systems.</p>
            <p> The case study described demonstrates that access to such data is indeed feasible, although at this stage it encompasses only approximately 12,000 images. However, the authors&#x2019; conclusions are not optimistic. They anticipate difficulties in scaling the developed data-sharing process for broader clinical implementation, due to diverse data management models within medical institutions and legislative discrepancies governing access to medical data.</p>
            <p> As a result, narrowing the gap in the development of such algorithms between Europe and North America or certain Asian countries may become increasingly challenging. Nevertheless, the reviewed work is of considerable importance for the advancement of diagnostic imaging, and its indexing is strongly recommend.</p>
            <p>Is the case presented with sufficient detail to be useful for teaching or other practitioners?</p>
            <p>Yes</p>
            <p>Is the work clearly and accurately presented and does it cite the current literature?</p>
            <p>Yes</p>
            <p>If applicable, is the statistical analysis and its interpretation appropriate?</p>
            <p>Not applicable</p>
            <p>Are all the source data underlying the results available to ensure full reproducibility?</p>
            <p>Yes</p>
            <p>Are the conclusions drawn adequately supported by the results?</p>
            <p>Yes</p>
            <p>Is the background of the case&#x2019;s history and progression described in sufficient detail?</p>
            <p>Yes</p>
            <p>Reviewer Expertise:</p>
            <p>medical imaging, AI, machine learning</p>
            <p>I confirm that I have read this submission and believe that I have an appropriate level of expertise to confirm that it is of an acceptable scientific standard.</p>
        </body>
    </sub-article>
    <sub-article article-type="reviewer-report" id="report63501">
        <front-stub>
            <article-id pub-id-type="doi">10.21956/openreseurope.22734.r63501</article-id>
            <title-group>
                <article-title>Reviewer response for version 1</article-title>
            </title-group>
            <contrib-group>
                <contrib contrib-type="author">
                    <name>
                        <surname>Neha</surname>
                        <given-names>Neha</given-names>
                    </name>
                    <xref ref-type="aff" rid="r63501a2">2</xref>
                    <role>Referee</role>
                    <uri content-type="orcid">https://orcid.org/0009-0004-3702-2382</uri>
                </contrib>
                <contrib contrib-type="author">
                    <name>
                        <surname>Shukla</surname>
                        <given-names>Deepak Kumar</given-names>
                    </name>
                    <xref ref-type="aff" rid="r63501a1">1</xref>
                    <role>Co-referee</role>
                </contrib>
                <aff id="r63501a1">
                    <label>1</label>Rutgers Business School, Rutgers University Newark Business School (Ringgold ID: 33869), Newark, Jersey, USA</aff>
                <aff id="r63501a2">
                    <label>2</label>Kent State University, Kent, USA</aff>
            </contrib-group>
            <author-notes>
                <fn fn-type="conflict">
                    <p>
                        <bold>Competing interests: </bold>No competing interests were disclosed.</p>
                </fn>
            </author-notes>
            <pub-date pub-type="epub">
                <day>11</day>
                <month>11</month><year>2025</year>
            </pub-date>
            <permissions>
                <copyright-statement>Copyright: &#xA9; 2025 Shukla DK and Neha N</copyright-statement>
                <copyright-year>2025</copyright-year>
                <license xlink:href="https://creativecommons.org/licenses/by/4.0/">
                    <license-p>This is an open access peer review report distributed under the terms of the Creative Commons Attribution Licence, which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.</license-p>
                </license>
            </permissions>
            <related-article ext-link-type="doi" id="relatedArticleReport63501" related-article-type="peer-reviewed-article" xlink:href="10.12688/openreseurope.21016.1"/>
            <custom-meta-group>
                <custom-meta>
                    <meta-name>recommendation</meta-name>
                    <meta-value>approve-with-reservations</meta-value>
                </custom-meta>
            </custom-meta-group>
        </front-stub>
        <body>
            <p>This article documents the implementation of the&#xA0;European Federation for Cancer Images (EUCAIM)&#xA0;initiative, focusing on the integration of imaging and clinical data from a reference university hospital into the EUCAIM infrastructure.&#xA0;</p>
            <p> </p>
            <p> Good points are:</p>
            <p> 1. High data-processing throughput (12,484 studies, 98.6% efficiency).</p>
            <p> 2. Clarity and structure suitable for replication within similar EU projects.</p>
            <p> 3.&#xA0;Integrates lessons from multiple large-scale initiatives.</p>
            <p> </p>
            <p> However&#xA0;Recommendations for Revision:</p>
            <p> There is minimal discussion on computational performance metrics or storage requirements. Also;</p>
            <p> 1. provide pseudonymization scripts, metadata dictionaries, or anonymized examples through Zenodo or EUCAIM&#x2019;s public repositories.</p>
            <p> 2. include workflow diagrams or tabular summaries of ethical and legal documentation steps.</p>
            <p> 3. address sustainability models and future interoperability challenges beyond current pilot hospitals.</p>
            <p> 3. ensure consistent acronym expansion.</p>
            <p>Is the case presented with sufficient detail to be useful for teaching or other practitioners?</p>
            <p>Partly</p>
            <p>Is the work clearly and accurately presented and does it cite the current literature?</p>
            <p>Yes</p>
            <p>If applicable, is the statistical analysis and its interpretation appropriate?</p>
            <p>Not applicable</p>
            <p>Are all the source data underlying the results available to ensure full reproducibility?</p>
            <p>Partly</p>
            <p>Are the conclusions drawn adequately supported by the results?</p>
            <p>Yes</p>
            <p>Is the background of the case&#x2019;s history and progression described in sufficient detail?</p>
            <p>Yes</p>
            <p>Reviewer Expertise:</p>
            <p>Artificial IntelligenceComputer VisionImage ProcessingDeep learningMedical Imaging</p>
            <p>We confirm that we have read this submission and believe that we have an appropriate level of expertise to confirm that it is of an acceptable scientific standard, however we have significant reservations, as outlined above.</p>
        </body>
        <back>
            <ref-list>
                <title>References</title>
                <ref id="rep-ref-63501-1">
                    <label>1</label>
                    <mixed-citation publication-type="journal">
                        <person-group person-group-type="author"/>:
                        <article-title>Retrieval-Augmented Generation (RAG) in Healthcare: A Comprehensive Review</article-title>.
                        <source>
                            <italic>AI</italic>
                        </source>.<year>2025</year>;<volume>6</volume>(<issue>9</issue>) :
                        <elocation-id>10.3390/ai6090226</elocation-id>
                        <pub-id pub-id-type="doi">10.3390/ai6090226</pub-id>
                    </mixed-citation>
                </ref>
            </ref-list>
        </back>
    </sub-article>
</article>