{
  "framework": "EVALLM",
  "title": "ESMO framework for the evaluation of clinician-facing large language models in oncology",
  "version": "1.0.0",
  "scope": {
    "covers": "Clinician-facing AI systems used for clinical decision support in oncology (ELCAP type 2).",
    "excludes": [
      "Patient-facing systems (ELCAP type 1)",
      "Background institutional systems (ELCAP type 3)",
      "Autonomous agents that initiate clinical actions without a direct human prompt (EVAGENT)"
    ],
    "note": "EVALLM complements, and does not replace, existing law and regulation, and is not a substitute for regulatory approval."
  },
  "consensus": {
    "scale": "1 to 9",
    "agreementBand": "7 to 9",
    "acceptanceThreshold": 75,
    "disagreementCeiling": 15,
    "strongConsensusThreshold": 96,
    "rounds": [
      {
        "round": 1,
        "month": "April 2026",
        "statementsRated": 87,
        "panellists": 29
      },
      {
        "round": 2,
        "month": "June 2026",
        "statementsRated": 8,
        "panellists": 27
      }
    ]
  },
  "vocabularies": {
    "actor": [
      "Evaluator",
      "Developer",
      "Institution",
      "Vendor",
      "Clinician"
    ],
    "stage": [
      "Pre-deployment",
      "Deployment decision",
      "Post-deployment"
    ],
    "appliesTo": [
      "All systems",
      "Treatment recommendation",
      "Biomarker interpretation",
      "Trial matching",
      "Multimodal systems",
      "Imaging inputs",
      "Screening or triage use",
      "Patient-facing outputs",
      "Complex or MTB cases",
      "Cross-language use",
      "Open-source vs cloud choice"
    ],
    "mapsTo": [
      "TRIPOD+AI",
      "TRIPOD-LLM",
      "DECIDE-AI",
      "CONSORT-AI",
      "SPIRIT-AI",
      "ELCAP",
      "EBAI",
      "EVAGENT",
      "EVALLM-specific"
    ],
    "category": [
      "Strong consensus",
      "Consensus"
    ]
  },
  "domains": [
    {
      "number": 1,
      "title": "Clinical accuracy",
      "summary": "Whether the output of the system is correct. In oncology this is difficult, because the correct answer is itself often uncertain and depends on factors including patient preference, and because the clinical reasoning path matters as much as the final output. Six of fifteen statements reached strong consensus, the highest proportion of any domain.",
      "statementCount": 15
    },
    {
      "number": 2,
      "title": "Safety and harm",
      "summary": "Safety is distinct from accuracy: a system can be accurate on average and still produce individual outputs that cause harm. The preparatory review found hallucination rates reported inconsistently, potentially harmful recommendations rarely documented, and no standardised harm classification scale in use.",
      "statementCount": 11
    },
    {
      "number": 3,
      "title": "Bias and equity",
      "summary": "This domain achieved the lowest agreement in the framework, showing where additional evidence is most needed. Only one of nine statements reached strong consensus and four sit at 79.3%. Agreement was high on measuring subgroup performance and lower on the obligations that arise beyond the evaluation dataset.",
      "statementCount": 9
    },
    {
      "number": 4,
      "title": "Transparency, explainability and reproducibility",
      "summary": "What must be disclosed for a published result to be interpretable at all: the exact system, version and date of access, the prompting strategy, the evidence behind each output, and whether repeated identical queries agree. The same model name can denote different systems at different dates.",
      "statementCount": 8
    },
    {
      "number": 5,
      "title": "Robustness, reliability and post-deployment monitoring",
      "summary": "Performance drifts as medical knowledge evolves while training data stays static, so validation is not a single event. All four strong-consensus statements in this domain concern what happens after a system enters clinical use rather than before it; agreement was lower on the evidence required before deployment.",
      "statementCount": 11
    },
    {
      "number": 6,
      "title": "Usability, workflow integration and clinical utility",
      "summary": "A system can be accurate, safe and stable and still provide no clinical utility. The single strong-consensus statement here is the one defining utility as the measurable effect of AI use on actual clinical decisions, rather than accuracy in isolation.",
      "statementCount": 11
    },
    {
      "number": 7,
      "title": "Governance, legal and ethical considerations",
      "summary": "Institutional rules on what patient data may leave the institution, on responsibility for the decision, on regulatory classification, and on disclosure to patients. The panel also addressed unofficial or shadow use directly, through guidance and education rather than prohibition alone.",
      "statementCount": 8
    },
    {
      "number": 8,
      "title": "Validation study design",
      "summary": "How the evaluation itself should be built: proportionate to clinical risk, pre-registered, powered a priori, screened for data contamination, and conducted in oncology rather than borrowed from other specialties. Statements are written at the level of clinical tasks so that they survive changes in the underlying technology.",
      "statementCount": 11
    }
  ],
  "statements": [
    {
      "id": "1.1",
      "domain": 1,
      "group": "Human reference standard",
      "statement": "Every evaluation of a clinician-facing AI system in oncology should compare AI performance with an appropriate human reference standard (e.g. an oncologist or multidisciplinary tumor board) performing the same task under similar conditions.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "TRIPOD+AI",
        "DECIDE-AI"
      ],
      "agreement": 93.1,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "1.4",
      "domain": 1,
      "group": "Human reference standard",
      "statement": "In clinical situations where no established guideline exists (e.g. complex molecular tumor board decisions, rare cancers, or emerging biomarker-defined subgroups), AI outputs should be evaluated by a panel of subspecialty experts against the best available evidence rather than against a fixed guideline reference.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "Treatment recommendation",
      "mapsTo": [
        "EVALLM-specific"
      ],
      "agreement": 89.7,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "1.6",
      "domain": 1,
      "group": "Human reference standard",
      "statement": "For clinical decision support use cases, AI outputs should be evaluated by multiple independent clinical experts (typically a multidisciplinary tumor board or equivalent panel) using a predefined scoring scheme appropriate to the clinical task. The number of raters and required level of independence should be determined based on task complexity, and inter-rater reliability should be reported using appropriate metrics (intraclass correlation coefficient or Fleiss kappa).",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "TRIPOD-LLM"
      ],
      "agreement": 85.2,
      "round": 2,
      "category": "Consensus"
    },
    {
      "id": "1.11",
      "domain": 1,
      "group": "Human reference standard",
      "statement": "For biomarker interpretation tasks (e.g. actionability of a genomic alteration, interpretation of a complex molecular or multi-omic profile), AI accuracy should be benchmarked against molecular tumor board decisions or equivalent expert review.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "Biomarker interpretation",
      "mapsTo": [
        "EBAI"
      ],
      "agreement": 86.2,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "1.2",
      "domain": 1,
      "group": "Guidelines as benchmark",
      "statement": "Guideline concordance (the proportion of AI-generated recommendations that align with current evidence-based guidelines such as ESMO guidelines) should be reported as one measure of clinical accuracy but should not be the sole benchmark.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "Treatment recommendation",
      "mapsTo": [
        "EVALLM-specific"
      ],
      "agreement": 93.1,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "1.3",
      "domain": 1,
      "group": "Guidelines as benchmark",
      "statement": "As clinical guidelines may be incomplete, outdated, or absent for certain scenarios (e.g. rare molecular profiles, novel combinations, rapidly evolving treatment landscapes), the evaluation framework should also allow for AI recommendations to differ from current guidelines if they are supported by credible published evidence or expert clinical reasoning.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "Treatment recommendation",
      "mapsTo": [
        "EVALLM-specific"
      ],
      "agreement": 86.2,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "1.5",
      "domain": 1,
      "group": "Guidelines as benchmark",
      "statement": "When an AI recommendation differs from existing guidelines, the evaluation should distinguish between (a) a genuinely incorrect or unsupported recommendation and (b) a recommendation that is concordant with emerging evidence, recent trial results, or expert consensus not yet captured in current guidelines.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "Treatment recommendation",
      "mapsTo": [
        "EVALLM-specific"
      ],
      "agreement": 96.6,
      "round": 1,
      "category": "Strong consensus"
    },
    {
      "id": "1.9",
      "domain": 1,
      "group": "Error and completeness",
      "statement": "When AI systems are used for treatment recommendations, the evaluation should assess not only the top recommendation but also the completeness and appropriate ranking of all reasonable alternatives, including their supporting evidence.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "Treatment recommendation",
      "mapsTo": [
        "EVALLM-specific"
      ],
      "agreement": 89.7,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "1.10",
      "domain": 1,
      "group": "Error and completeness",
      "statement": "Accuracy evaluation should distinguish between factual errors (wrong information), omission errors (missing critical information), and reasoning errors (correct facts but flawed clinical logic).",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "EVALLM-specific"
      ],
      "agreement": 100,
      "round": 1,
      "category": "Strong consensus"
    },
    {
      "id": "1.15",
      "domain": 1,
      "group": "Error and completeness",
      "statement": "Evaluation of clinical accuracy should include assessment of output comprehensiveness, including whether clinically relevant options and supporting information are adequately captured.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "EVALLM-specific"
      ],
      "agreement": 93.1,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "1.8",
      "domain": 1,
      "group": "Task and modality specificity",
      "statement": "Clinical accuracy should be reported separately for each intended task (e.g. diagnosis, treatment recommendation, toxicity management, biomarker interpretation, clinical trial matching, molecular tumor board support), as performance may vary substantially across tasks.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "TRIPOD-LLM"
      ],
      "agreement": 100,
      "round": 1,
      "category": "Strong consensus"
    },
    {
      "id": "1.12",
      "domain": 1,
      "group": "Task and modality specificity",
      "statement": "For AI systems that process multimodal inputs (e.g. combining pathology images, radiology, genomic data, and clinical text), accuracy should be evaluated both for the integrated output and, where feasible, for each input modality individually to identify modality-specific weaknesses.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "Multimodal systems",
      "mapsTo": [
        "EVALLM-specific"
      ],
      "agreement": 89.7,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "1.7",
      "domain": 1,
      "group": "Datasets and statistical reporting",
      "statement": "Evaluation datasets should be representative of the AI system's intended target population. For systems with a broad clinical scope, this should include a mix of common and rare cancer types, stages, and treatment lines. Accuracy metrics should be reported both overall and stratified by clinically meaningful subgroups (e.g. disease stage, treatment line, previous treatments, rare vs common presentation).",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "TRIPOD+AI"
      ],
      "agreement": 100,
      "round": 2,
      "category": "Strong consensus"
    },
    {
      "id": "1.13",
      "domain": 1,
      "group": "Datasets and statistical reporting",
      "statement": "AI accuracy metrics should be reported with confidence intervals and, where applicable, inter-rater agreement (e.g. Cohen's kappa) among the human evaluators who serve as the reference standard, the proportion of cases with expert disagreement and the adjudication method.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "TRIPOD+AI"
      ],
      "agreement": 96.6,
      "round": 1,
      "category": "Strong consensus"
    },
    {
      "id": "1.14",
      "domain": 1,
      "group": "Datasets and statistical reporting",
      "statement": "AI system performance should be evaluated on both standardized clinical vignettes and real-world, unstructured clinical data (e.g. actual electronic health records with missing or inconsistent information) to assess robustness under realistic conditions. The scope of validation (e.g. per institution, per EHR system, or per clinical setting) should be specified and justified.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "DECIDE-AI"
      ],
      "agreement": 96.6,
      "round": 1,
      "category": "Strong consensus"
    },
    {
      "id": "2.1",
      "domain": 2,
      "group": "Error and harm measurement",
      "statement": "Evaluation studies should report the hallucination rate, defined as outputs that contain fabricated clinical information not supported by the input data or established medical knowledge (e.g. citing a non-existent trial, inventing a drug dose, fabricating a molecular finding, or misinterpreting an image).",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "TRIPOD-LLM"
      ],
      "agreement": 100,
      "round": 1,
      "category": "Strong consensus"
    },
    {
      "id": "2.2",
      "domain": 2,
      "group": "Error and harm measurement",
      "statement": "A standardized harm classification scale for clinical AI outputs should be adopted (e.g. no harm / minor harm / moderate harm / severe or potentially lethal harm), with definitions specific to oncology decision-making. Where applicable, established clinical scales (e.g. CTCAE) should inform the grading.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "EVALLM-specific"
      ],
      "agreement": 86.2,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "2.3",
      "domain": 2,
      "group": "Error and harm measurement",
      "statement": "Safety evaluations should assess whether the AI system generates contraindicated or unsafe recommendations, for example suggesting a drug to which the patient has a documented allergy, recommending a treatment incompatible with organ function, or overlooking a critical imaging finding.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "EVALLM-specific"
      ],
      "agreement": 96.6,
      "round": 1,
      "category": "Strong consensus"
    },
    {
      "id": "2.6",
      "domain": 2,
      "group": "Error and harm measurement",
      "statement": "Evaluation studies should include analysis of outputs that could have caused patient harm if acted on without clinician review (a “near-miss” analysis).",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "EVALLM-specific"
      ],
      "agreement": 93.1,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "2.4",
      "domain": 2,
      "group": "Built-in safety behaviour",
      "statement": "AI systems used for clinical decision support should communicate uncertainty in a clear, interpretable, and clinically meaningful way.",
      "actors": [
        "Developer"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "DECIDE-AI"
      ],
      "agreement": 93.1,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "2.7",
      "domain": 2,
      "group": "Built-in safety behaviour",
      "statement": "AI systems should incorporate built-in safety checks, including flagging of potential contraindications, prompting the clinician when critical data is missing from the input, and linking recommendations to their supporting evidence sources.",
      "actors": [
        "Developer"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "EVALLM-specific"
      ],
      "agreement": 100,
      "round": 1,
      "category": "Strong consensus"
    },
    {
      "id": "2.12",
      "domain": 2,
      "group": "Built-in safety behaviour",
      "statement": "When an AI system is used to screen or triage clinical information (e.g. laboratory results, imaging reports), the system should define a minimum confidence level below which the output is automatically flagged for mandatory human review.",
      "actors": [
        "Developer"
      ],
      "stage": "Deployment decision",
      "appliesTo": "Screening or triage use",
      "mapsTo": [
        "EVALLM-specific"
      ],
      "agreement": 89.7,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "2.8",
      "domain": 2,
      "group": "Testing under adverse conditions",
      "statement": "AI safety should be tested under conditions of incomplete or contradictory input data (e.g. missing lab values, conflicting radiology and pathology reports, poor-quality imaging), because these situations are common in real-world practice and may trigger unreliable outputs.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "EVALLM-specific"
      ],
      "agreement": 93.1,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "2.9",
      "domain": 2,
      "group": "Testing under adverse conditions",
      "statement": "Safety evaluations should assess the robustness of AI systems to adversarial or manipulative inputs, including prompt injection where relevant (e.g. prompt injection attacks).",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "EVALLM-specific"
      ],
      "agreement": 93.1,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "2.5",
      "domain": 2,
      "group": "Scope and institutional readiness",
      "statement": "AI systems evaluated under this framework must function as decision support tools with human clinicians retaining final responsibility for all treatment decisions. Autonomous therapeutic decision-making by AI systems is outside the scope of this framework.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "ELCAP",
        "EVAGENT"
      ],
      "agreement": 92.6,
      "round": 2,
      "category": "Consensus"
    },
    {
      "id": "2.11",
      "domain": 2,
      "group": "Scope and institutional readiness",
      "statement": "Institutions deploying clinical AI systems should ensure that adequate clinical expertise and fallback procedures are available in case the AI system becomes unavailable or produces unreliable outputs due to technical failure, system errors, or missed updates.",
      "actors": [
        "Institution"
      ],
      "stage": "Deployment decision",
      "appliesTo": "All systems",
      "mapsTo": [
        "DECIDE-AI"
      ],
      "agreement": 89.7,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "3.1",
      "domain": 3,
      "group": "Subgroup performance reporting",
      "statement": "AI evaluation studies should report performance stratified by clinically and demographically relevant patient variables, including at minimum age and sex. Where data are available, additional stratification should include race or ethnicity, socioeconomic factors, and geographic region. Missing demographic data should be reported transparently.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "TRIPOD+AI"
      ],
      "agreement": 92.6,
      "round": 2,
      "category": "Consensus"
    },
    {
      "id": "3.2",
      "domain": 3,
      "group": "Subgroup performance reporting",
      "statement": "AI performance should be evaluated separately for common cancers and for rare or underrepresented cancer types, to identify performance gaps that may disadvantage patients with less common diseases.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "EVALLM-specific"
      ],
      "agreement": 79.3,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "3.9",
      "domain": 3,
      "group": "Subgroup performance reporting",
      "statement": "The evaluation framework should assess whether AI outputs reflect biases present in training data, for example underrepresentation of certain ethnic groups or geographic regions in clinical trial populations, or systematic patterns in imaging datasets from specific institutions.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "TRIPOD+AI"
      ],
      "agreement": 89.7,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "3.4",
      "domain": 3,
      "group": "Setting and health-system context",
      "statement": "AI-generated treatment recommendations should be evaluated for local applicability: whether the recommended treatments are actually available, approved, and reimbursed in the healthcare system where the tool is being used.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Deployment decision",
      "appliesTo": "Treatment recommendation",
      "mapsTo": [
        "EVALLM-specific"
      ],
      "agreement": 79.3,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "3.5",
      "domain": 3,
      "group": "Setting and health-system context",
      "statement": "Evaluation datasets should include clinical scenarios from diverse healthcare settings (academic centers, community hospitals, low-resource settings) to test whether the AI system performs equitably across different levels of clinical infrastructure and data availability.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "TRIPOD+AI"
      ],
      "agreement": 79.3,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "3.8",
      "domain": 3,
      "group": "Language",
      "statement": "AI systems intended for use in, or producing outputs across, a language different from their primary training or development language should be validated in the target language(s), with particular attention to high-stakes content such as treatment selection and toxicity grading.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "Cross-language use",
      "mapsTo": [
        "EVALLM-specific"
      ],
      "agreement": 81.5,
      "round": 2,
      "category": "Consensus"
    },
    {
      "id": "3.6",
      "domain": 3,
      "group": "Task-specific equity",
      "statement": "When AI systems are used for clinical trial matching, evaluations should assess whether eligible patients are identified equitably across demographic and disease groups.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "Trial matching",
      "mapsTo": [
        "EVALLM-specific"
      ],
      "agreement": 82.8,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "3.7",
      "domain": 3,
      "group": "Task-specific equity",
      "statement": "AI outputs should be evaluated for the ability to integrate patient preferences, values, and goals of care (e.g. tolerance for toxicity, preference for quality of life over maximal efficacy) when these are provided as part of the clinical input.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "Treatment recommendation",
      "mapsTo": [
        "EVALLM-specific"
      ],
      "agreement": 79.3,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "3.10",
      "domain": 3,
      "group": "Acquisition and technical bias",
      "statement": "For multimodal AI systems processing imaging data (e.g. pathology slides, radiology), bias evaluation should include assessment across different imaging equipment, staining protocols, and scanning conditions.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "Imaging inputs",
      "mapsTo": [
        "EBAI"
      ],
      "agreement": 96.6,
      "round": 1,
      "category": "Strong consensus"
    },
    {
      "id": "4.1",
      "domain": 4,
      "group": "Disclosure of what was evaluated",
      "statement": "Every published AI evaluation study should disclose the specific system(s) evaluated (including model name, version, and date of access), the prompting or input strategy employed, and any additional methods applied (e.g. fine-tuning on medical data, connecting the system to external knowledge sources, or integration of multiple data modalities).",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "TRIPOD-LLM"
      ],
      "agreement": 100,
      "round": 1,
      "category": "Strong consensus"
    },
    {
      "id": "4.9",
      "domain": 4,
      "group": "Disclosure of what was evaluated",
      "statement": "For multimodal AI systems, the transparency requirements should extend to each input modality: which data types the system processes, how they are combined, and whether the system can indicate which input modality most influenced a given recommendation.",
      "actors": [
        "Developer"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "Multimodal systems",
      "mapsTo": [
        "EVALLM-specific"
      ],
      "agreement": 89.7,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "4.2",
      "domain": 4,
      "group": "Evidence traceability and quality",
      "statement": "AI outputs used for clinical decision support should include traceable references to the evidence sources on which the recommendation is based (e.g. specific guidelines, clinical trials, or publications).",
      "actors": [
        "Developer"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "EVALLM-specific"
      ],
      "agreement": 89.7,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "4.3",
      "domain": 4,
      "group": "Evidence traceability and quality",
      "statement": "Evaluations should assess not only whether evidence is cited, but also whether the cited evidence is relevant, accurate, and appropriate in strength for the recommendation made. A recommendation supported by a phase III randomized trial should be weighted differently from one supported only by a case report, preclinical data, or a source that the system fabricated.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "EVALLM-specific"
      ],
      "agreement": 96.6,
      "round": 1,
      "category": "Strong consensus"
    },
    {
      "id": "4.4",
      "domain": 4,
      "group": "Reproducibility and replication",
      "statement": "Evaluation studies should report the reproducibility of AI outputs: whether repeated queries with identical clinical inputs produce consistent answers, or whether outputs vary meaningfully across runs.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "TRIPOD-LLM"
      ],
      "agreement": 96.6,
      "round": 1,
      "category": "Strong consensus"
    },
    {
      "id": "4.7",
      "domain": 4,
      "group": "Reproducibility and replication",
      "statement": "Evaluation studies should make key evaluation materials available to support independent replication, including datasets, vignettes, prompts, and scoring rubrics, where legally and ethically feasible.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "TRIPOD+AI"
      ],
      "agreement": 89.7,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "4.5",
      "domain": 4,
      "group": "Deployment model choice",
      "statement": "When both open-source (locally installed) and proprietary (cloud-based) systems are being considered, the evaluation should explicitly report trade-offs in transparency, data privacy, ability to customize the system, and clinical performance.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Deployment decision",
      "appliesTo": "Open-source vs cloud choice",
      "mapsTo": [
        "EVALLM-specific"
      ],
      "agreement": 89.7,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "4.8",
      "domain": 4,
      "group": "Deployment model choice",
      "statement": "The choice between deploying an open-source (locally hosted) system vs. a cloud-based commercial system should be documented as an institutional governance decision, with explicit consideration of data privacy, regulatory compliance, and the level of institutional control over the system.",
      "actors": [
        "Institution"
      ],
      "stage": "Deployment decision",
      "appliesTo": "Open-source vs cloud choice",
      "mapsTo": [
        "EVALLM-specific"
      ],
      "agreement": 82.8,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "5.1",
      "domain": 5,
      "group": "Validation pathway",
      "statement": "AI system performance should be validated in at least one external institution (external validation) before the tool is recommended for routine clinical use. The scope of external validation should account for relevant differences in clinical infrastructure, including the electronic health record system in use.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "TRIPOD+AI",
        "EBAI"
      ],
      "agreement": 79.3,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "5.2",
      "domain": 5,
      "group": "Validation pathway",
      "statement": "AI evaluation should follow a three-stage validation pathway: (1) retrospective evaluation on historical clinical cases, (2) controlled prospective evaluation in a clinical setting, (3) monitored real-world deployment with ongoing outcome tracking.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "DECIDE-AI"
      ],
      "agreement": 82.8,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "5.3",
      "domain": 5,
      "group": "Drift and re-evaluation",
      "statement": "After deployment, systematic monitoring should be in place to detect performance degradation over time, which may occur as medical knowledge evolves, new drugs are approved, or clinical guidelines are updated while the system's training data remains static.",
      "actors": [
        "Institution"
      ],
      "stage": "Post-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "EBAI"
      ],
      "agreement": 100,
      "round": 1,
      "category": "Strong consensus"
    },
    {
      "id": "5.4",
      "domain": 5,
      "group": "Drift and re-evaluation",
      "statement": "A defined schedule for re-evaluation of deployed AI systems should be established, for example triggered by major guideline updates, new drug approvals, changes in standard-of-care, or at regular intervals (e.g. annually).",
      "actors": [
        "Institution"
      ],
      "stage": "Post-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "EVALLM-specific"
      ],
      "agreement": 96.6,
      "round": 1,
      "category": "Strong consensus"
    },
    {
      "id": "5.7",
      "domain": 5,
      "group": "Drift and re-evaluation",
      "statement": "Post-deployment monitoring should, where feasible, link AI-supported clinical decisions to actual patient outcomes (e.g. treatment response rates, progression-free survival, adverse event rates) over time, to assess the real-world impact of the tool.",
      "actors": [
        "Institution"
      ],
      "stage": "Post-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "DECIDE-AI"
      ],
      "agreement": 89.7,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "5.5",
      "domain": 5,
      "group": "Version control",
      "statement": "Clear rules should be established for when and how a deployed AI system is updated to a new version, and what level of re-validation is required before the new version replaces the previous one in clinical use. Institutions should have dedicated expertise available to manage this process.",
      "actors": [
        "Institution"
      ],
      "stage": "Post-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "EVALLM-specific"
      ],
      "agreement": 96.6,
      "round": 1,
      "category": "Strong consensus"
    },
    {
      "id": "5.6",
      "domain": 5,
      "group": "Version control",
      "statement": "Validation of a specific AI model version should not automatically extend to successor versions or substantially updated releases. Each major version update should undergo independent evaluation proportionate to the scope of the changes.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Post-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "EVALLM-specific"
      ],
      "agreement": 93.1,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "5.11",
      "domain": 5,
      "group": "Version control",
      "statement": "AI system outputs should demonstrate acceptable stability across different technical configurations (e.g. hardware, software versions, API updates) without clinically meaningful changes in recommendations.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "EVALLM-specific"
      ],
      "agreement": 86.2,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "5.8",
      "domain": 5,
      "group": "Shared responsibility and reporting",
      "statement": "The responsibility for continuous quality monitoring and safety surveillance of clinical AI systems should be shared between the vendor or developer and the deploying institution, with clearly defined roles for each party.",
      "actors": [
        "Vendor",
        "Institution"
      ],
      "stage": "Post-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "EVALLM-specific"
      ],
      "agreement": 82.8,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "5.9",
      "domain": 5,
      "group": "Shared responsibility and reporting",
      "statement": "Vendors of clinical AI products should be required to support structured feedback collection from clinical users and to participate in post-deployment safety monitoring, analogous to existing post-market surveillance requirements for medical devices.",
      "actors": [
        "Vendor"
      ],
      "stage": "Post-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "EBAI"
      ],
      "agreement": 96.6,
      "round": 1,
      "category": "Strong consensus"
    },
    {
      "id": "5.10",
      "domain": 5,
      "group": "Shared responsibility and reporting",
      "statement": "A structured reporting channel should be established for clinicians to report errors, unexpected outputs, and safety concerns to both institutional quality assurance and the AI system developer, enabling continuous post-deployment monitoring and early detection of systematic safety signals.",
      "actors": [
        "Institution",
        "Vendor"
      ],
      "stage": "Post-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "EVALLM-specific"
      ],
      "agreement": 92.6,
      "round": 2,
      "category": "Consensus"
    },
    {
      "id": "6.1",
      "domain": 6,
      "group": "Efficiency",
      "statement": "AI evaluation should include time-efficiency metrics: the time required for the clinician to obtain, review, and (if necessary) correct the AI output should be measured and compared to the time required for the same task without AI assistance.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "DECIDE-AI"
      ],
      "agreement": 86.2,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "6.2",
      "domain": 6,
      "group": "Efficiency",
      "statement": "Clinical utility evaluation should assess whether AI use improves efficiency for the intended task, in comparison with the standard AI-unassisted workflow for the intended task, defined to include time required, cognitive workload, or resource use. Efficiency gains should not be treated as a substitute for accuracy or clinical impact.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "DECIDE-AI"
      ],
      "agreement": 92.6,
      "round": 2,
      "category": "Consensus"
    },
    {
      "id": "6.6",
      "domain": 6,
      "group": "Effect on clinical decisions",
      "statement": "Clinical utility should be evaluated not only as accuracy in isolation but as the measurable impact of AI use on actual clinical decisions (e.g. the proportion of cases where the tumor board decision changed after reviewing AI input).",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "EVALLM-specific"
      ],
      "agreement": 96.6,
      "round": 1,
      "category": "Strong consensus"
    },
    {
      "id": "6.7",
      "domain": 6,
      "group": "Effect on clinical decisions",
      "statement": "Where feasible, prospective studies should assess the incremental value of AI by comparing decisions made with and without AI support.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "CONSORT-AI"
      ],
      "agreement": 89.7,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "6.10",
      "domain": 6,
      "group": "Effect on clinical decisions",
      "statement": "For tasks where AI systems are used to augment expert reasoning in complex cases (e.g. molecular tumor boards, rare cancers, multi-line treatment decisions), the evaluation should specifically measure whether the AI adds clinically relevant information or perspectives that would not have been considered by the clinical team alone.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "Complex or MTB cases",
      "mapsTo": [
        "EVALLM-specific"
      ],
      "agreement": 82.8,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "6.3",
      "domain": 6,
      "group": "Reported outcomes",
      "statement": "Clinician-reported outcomes (e.g. perceived usability, trust in the system, cognitive load, perceived impact on decision quality) should be measured using validated instruments as a secondary outcome in all prospective AI evaluation studies.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "DECIDE-AI"
      ],
      "agreement": 79.3,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "6.4",
      "domain": 6,
      "group": "Reported outcomes",
      "statement": "Where AI-supported decisions directly affect patient-facing interactions (e.g. treatment discussions, informed consent conversations), patient-reported outcomes (e.g. satisfaction with the consultation, understanding of the treatment plan, perceived quality of communication) should be assessed as an additional outcome measure.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "Patient-facing outputs",
      "mapsTo": [
        "DECIDE-AI"
      ],
      "agreement": 89.7,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "6.11",
      "domain": 6,
      "group": "Reported outcomes",
      "statement": "Patients should be involved in the evaluation of AI systems where the outputs directly affect treatment discussions or care plans.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "Patient-facing outputs",
      "mapsTo": [
        "EVALLM-specific"
      ],
      "agreement": 79.3,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "6.5",
      "domain": 6,
      "group": "Workflow integration",
      "statement": "Evaluation studies should assess the effect of AI integration on clinical workflow, including interoperability with existing systems and any additional documentation, verification, or administrative burden.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Deployment decision",
      "appliesTo": "All systems",
      "mapsTo": [
        "DECIDE-AI"
      ],
      "agreement": 75.9,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "6.8",
      "domain": 6,
      "group": "Human-system risks",
      "statement": "The evaluation should consider the risk of de-skilling: whether prolonged reliance on AI support leads to measurable decline in clinicians' ability to reason independently when the tool is unavailable.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Post-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "EVALLM-specific"
      ],
      "agreement": 82.8,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "6.9",
      "domain": 6,
      "group": "Human-system risks",
      "statement": "Evaluation studies should assess the risk of over-reliance (automatic bias) on AI, including clinician acceptance of incorrect outputs without adequate independent review.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "DECIDE-AI"
      ],
      "agreement": 89.7,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "7.1",
      "domain": 7,
      "group": "Institutional policy and shadow use",
      "statement": "Institutions should establish formal policies that either integrate validated AI tools into clinical workflows under defined governance, or clearly define acceptable boundaries for individual, unofficial use of external AI systems by clinicians.",
      "actors": [
        "Institution"
      ],
      "stage": "Deployment decision",
      "appliesTo": "All systems",
      "mapsTo": [
        "ELCAP"
      ],
      "agreement": 89.7,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "7.2",
      "domain": 7,
      "group": "Institutional policy and shadow use",
      "statement": "Institutional governance should explicitly address unofficial or “shadow” use of AI in clinical practice through guidance, education, and risk mitigation. This should be acknowledged and addressed through institutional guidance and education, rather than through prohibition alone.",
      "actors": [
        "Institution"
      ],
      "stage": "Deployment decision",
      "appliesTo": "All systems",
      "mapsTo": [
        "EVALLM-specific"
      ],
      "agreement": 93.1,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "7.6",
      "domain": 7,
      "group": "Institutional policy and shadow use",
      "statement": "Institutions should define which categories of patient information may be submitted to external commercial systems vs. only to locally deployed institutional systems.",
      "actors": [
        "Institution"
      ],
      "stage": "Deployment decision",
      "appliesTo": "All systems",
      "mapsTo": [
        "ELCAP"
      ],
      "agreement": 100,
      "round": 1,
      "category": "Strong consensus"
    },
    {
      "id": "7.4",
      "domain": 7,
      "group": "Responsibility and regulation",
      "statement": "Responsibility for clinical decisions should remain with the treating clinician, regardless of whether an AI system was used in the decision-making process.",
      "actors": [
        "Clinician"
      ],
      "stage": "Deployment decision",
      "appliesTo": "All systems",
      "mapsTo": [
        "ELCAP"
      ],
      "agreement": 89.7,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "7.5",
      "domain": 7,
      "group": "Responsibility and regulation",
      "statement": "The regulatory classification of clinical AI systems should be explicitly considered in the evaluation and deployment framework.",
      "actors": [
        "Institution"
      ],
      "stage": "Deployment decision",
      "appliesTo": "All systems",
      "mapsTo": [
        "EBAI"
      ],
      "agreement": 86.2,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "7.3",
      "domain": 7,
      "group": "Patient transparency and ethics",
      "statement": "When AI systems are formally integrated into institutional clinical workflows, institutions should disclose to patients that AI-assisted decision support is being used as part of their care.",
      "actors": [
        "Institution"
      ],
      "stage": "Deployment decision",
      "appliesTo": "All systems",
      "mapsTo": [
        "ELCAP"
      ],
      "agreement": 81.5,
      "round": 2,
      "category": "Consensus"
    },
    {
      "id": "7.7",
      "domain": 7,
      "group": "Patient transparency and ethics",
      "statement": "AI evaluation studies should address ethical review requirements, including informed consent for patients whose clinical data is used in AI evaluation or real-world deployment studies.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "SPIRIT-AI"
      ],
      "agreement": 82.8,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "7.8",
      "domain": 7,
      "group": "Forward reference",
      "statement": "The legal accountability for autonomous AI agents (systems that can initiate clinical actions without a direct human prompt) is a distinct and unresolved issue that goes beyond the scope of clinician-facing decision support. This topic should be addressed through dedicated legal and regulatory guidance in the planned ESMO EVAGENT workstream.",
      "actors": [
        "Institution"
      ],
      "stage": "Deployment decision",
      "appliesTo": "All systems",
      "mapsTo": [
        "EVAGENT"
      ],
      "agreement": 89.7,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "8.1",
      "domain": 8,
      "group": "Design proportionality and strength",
      "statement": "The study design used to evaluate an oncology AI system should be proportionate to the clinical risk, intended use, and stage of deployment of the system.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "DECIDE-AI"
      ],
      "agreement": 96.6,
      "round": 1,
      "category": "Strong consensus"
    },
    {
      "id": "8.2",
      "domain": 8,
      "group": "Design proportionality and strength",
      "statement": "Randomized or otherwise well-controlled prospective studies should be considered the strongest design for demonstrating the clinical utility of high-impact AI decision-support systems.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "CONSORT-AI"
      ],
      "agreement": 96.6,
      "round": 1,
      "category": "Strong consensus"
    },
    {
      "id": "8.3",
      "domain": 8,
      "group": "Design proportionality and strength",
      "statement": "Evaluation studies targeting hard clinical outcomes (e.g. progression-free survival, overall survival, adverse event rates) should be encouraged as the highest level of evidence, while acknowledging that such studies require large sample sizes and long follow-up.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "CONSORT-AI"
      ],
      "agreement": 86.2,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "8.6",
      "domain": 8,
      "group": "Statistical planning and integrity",
      "statement": "The risk of data contamination (i.e. the possibility that the AI system was trained on the same clinical cases used for evaluation) should be assessed and reported in all evaluation studies, particularly those using published clinical vignettes or board-style examination questions.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "TRIPOD-LLM"
      ],
      "agreement": 93.1,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "8.7",
      "domain": 8,
      "group": "Statistical planning and integrity",
      "statement": "Evaluation studies should pre-register their primary outcomes, scoring rubrics, and statistical analysis plans in a public registry to reduce the risk of selective outcome reporting.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "SPIRIT-AI"
      ],
      "agreement": 79.3,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "8.8",
      "domain": 8,
      "group": "Statistical planning and integrity",
      "statement": "AI evaluation studies should include a priori sample size justification, with an explicit statement of the primary endpoint, target precision or statistical power, and expected performance levels.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "TRIPOD+AI"
      ],
      "agreement": 93.1,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "8.5",
      "domain": 8,
      "group": "Scalability and comparison",
      "statement": "Scalable evaluation approaches that combine automated screening (e.g. using a validated second AI system as a structured reviewer) with targeted expert review may be acceptable, provided the automated screening component has been validated against human expert ratings.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "EVALLM-specific"
      ],
      "agreement": 89.7,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "8.9",
      "domain": 8,
      "group": "Scalability and comparison",
      "statement": "Head-to-head comparison studies evaluating multiple AI systems on the same clinical task using standardized protocols should be encouraged, as these provide the most informative data for clinical and institutional decision-making.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "EVALLM-specific"
      ],
      "agreement": 82.8,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "8.10",
      "domain": 8,
      "group": "Transferability across specialties",
      "statement": "Evidence from AI evaluation studies conducted outside oncology (e.g. in general medicine, cardiology, or other specialties) should not be considered sufficient for recommending deployment in oncology without dedicated oncology-specific validation, given the unique complexity and risk profile of cancer treatment decisions.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Deployment decision",
      "appliesTo": "All systems",
      "mapsTo": [
        "EVALLM-specific"
      ],
      "agreement": 82.8,
      "round": 1,
      "category": "Consensus"
    },
    {
      "id": "8.11",
      "domain": 8,
      "group": "Transferability across specialties",
      "statement": "Evaluation frameworks and study designs should be formulated at the level of clinical tasks and evaluation principles rather than specific technologies, so that they remain applicable as the underlying AI systems evolve.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "All systems",
      "mapsTo": [
        "EVALLM-specific"
      ],
      "agreement": 96.6,
      "round": 1,
      "category": "Strong consensus"
    },
    {
      "id": "8.4",
      "domain": 8,
      "group": "Task-specific design",
      "statement": "For AI-based clinical trial matching tools, prospective evaluation should assess effects on trial identification, time to enrolment, enrolment rate, and representativeness of enrolled patients.",
      "actors": [
        "Evaluator"
      ],
      "stage": "Pre-deployment",
      "appliesTo": "Trial matching",
      "mapsTo": [
        "EVALLM-specific"
      ],
      "agreement": 96.6,
      "round": 1,
      "category": "Strong consensus"
    }
  ],
  "adoptionChecklist": [
    {
      "section": "A",
      "title": "Evidence to look for before adoption",
      "source": "The published evaluation or the product documentation. This section should be answered by reading the paper or the documentation about the model.",
      "conditional": false,
      "conditionPrompt": null,
      "questions": [
        {
          "statementId": "1.8",
          "question": "Does the published evaluation report accuracy separately for each task the system will be used for?",
          "statement": "Clinical accuracy should be reported separately for each intended task (e.g. diagnosis, treatment recommendation, toxicity management, biomarker interpretation, clinical trial matching, molecular tumor board support), as performance may vary substantially across tasks.",
          "agreement": 100,
          "answers": [
            "yes",
            "no",
            "not-assessable"
          ]
        },
        {
          "statementId": "1.14",
          "question": "Does it report performance on real clinical records, or only on standardised clinical vignettes?",
          "statement": "AI system performance should be evaluated on both standardized clinical vignettes and real-world, unstructured clinical data (e.g. actual electronic health records with missing or inconsistent information) to assess robustness under realistic conditions. The scope of validation (e.g. per institution, per EHR system, or per clinical setting) should be specified and justified.",
          "agreement": 96.6,
          "answers": [
            "yes",
            "no",
            "not-assessable"
          ]
        },
        {
          "statementId": "2.1",
          "question": "Does it report a hallucination rate?",
          "statement": "Evaluation studies should report the hallucination rate, defined as outputs that contain fabricated clinical information not supported by the input data or established medical knowledge (e.g. citing a non-existent trial, inventing a drug dose, fabricating a molecular finding, or misinterpreting an image).",
          "agreement": 100,
          "answers": [
            "yes",
            "no",
            "not-assessable"
          ]
        },
        {
          "statementId": "2.2",
          "question": "Does it grade potentially harmful outputs on a defined harm classification scale, rather than describing them narratively?",
          "statement": "A standardized harm classification scale for clinical AI outputs should be adopted (e.g. no harm / minor harm / moderate harm / severe or potentially lethal harm), with definitions specific to oncology decision-making. Where applicable, established clinical scales (e.g. CTCAE) should inform the grading.",
          "agreement": 86.2,
          "answers": [
            "yes",
            "no",
            "not-assessable"
          ]
        },
        {
          "statementId": "3.1",
          "question": "Does it report performance broken down by patient subgroup, at minimum age and sex, and declare any missing demographic data?",
          "statement": "AI evaluation studies should report performance stratified by clinically and demographically relevant patient variables, including at minimum age and sex. Where data are available, additional stratification should include race or ethnicity, socioeconomic factors, and geographic region. Missing demographic data should be reported transparently.",
          "agreement": 92.6,
          "answers": [
            "yes",
            "no",
            "not-assessable"
          ]
        },
        {
          "statementId": "4.1",
          "question": "Does it identify exactly which system was tested — model name, version, date of access, and how it was prompted?",
          "statement": "Every published AI evaluation study should disclose the specific system(s) evaluated (including model name, version, and date of access), the prompting or input strategy employed, and any additional methods applied (e.g. fine-tuning on medical data, connecting the system to external knowledge sources, or integration of multiple data modalities).",
          "agreement": 100,
          "answers": [
            "yes",
            "no",
            "not-assessable"
          ]
        },
        {
          "statementId": "4.4",
          "question": "Does it report whether repeated identical queries produce consistent answers?",
          "statement": "Evaluation studies should report the reproducibility of AI outputs: whether repeated queries with identical clinical inputs produce consistent answers, or whether outputs vary meaningfully across runs.",
          "agreement": 96.6,
          "answers": [
            "yes",
            "no",
            "not-assessable"
          ]
        },
        {
          "statementId": "8.6",
          "question": "Does it assess the risk that the system had already seen the evaluation cases during training?",
          "statement": "The risk of data contamination (i.e. the possibility that the AI system was trained on the same clinical cases used for evaluation) should be assessed and reported in all evaluation studies, particularly those using published clinical vignettes or board-style examination questions.",
          "agreement": 93.1,
          "answers": [
            "yes",
            "no",
            "not-assessable"
          ]
        },
        {
          "statementId": "5.1",
          "question": "Does it include validation at an institution other than the one that developed the system?",
          "statement": "AI system performance should be validated in at least one external institution (external validation) before the tool is recommended for routine clinical use. The scope of external validation should account for relevant differences in clinical infrastructure, including the electronic health record system in use.",
          "agreement": 79.3,
          "answers": [
            "yes",
            "no",
            "not-assessable"
          ]
        },
        {
          "statementId": "5.2",
          "question": "Does the evidence go beyond retrospective cases into a controlled prospective setting?",
          "statement": "AI evaluation should follow a three-stage validation pathway: (1) retrospective evaluation on historical clinical cases, (2) controlled prospective evaluation in a clinical setting, (3) monitored real-world deployment with ongoing outcome tracking.",
          "agreement": 82.8,
          "answers": [
            "yes",
            "no",
            "not-assessable"
          ]
        },
        {
          "statementId": "8.10",
          "question": "Is the supporting evidence drawn from oncology, rather than from other specialties alone?",
          "statement": "Evidence from AI evaluation studies conducted outside oncology (e.g. in general medicine, cardiology, or other specialties) should not be considered sufficient for recommending deployment in oncology without dedicated oncology-specific validation, given the unique complexity and risk profile of cancer treatment decisions.",
          "agreement": 82.8,
          "answers": [
            "yes",
            "no",
            "not-assessable"
          ]
        }
      ]
    },
    {
      "section": "B",
      "title": "Institutional requirements",
      "source": "Institutional policies. Each question should be read in regard to the specific system you are assessing",
      "conditional": false,
      "conditionPrompt": null,
      "questions": [
        {
          "statementId": "7.6",
          "question": "Does this system transmit patient data outside the institution, and are the categories of data it requires permitted under our institutional rules?",
          "statement": "Institutions should define which categories of patient information may be submitted to external commercial systems vs. only to locally deployed institutional systems.",
          "agreement": 100,
          "answers": [
            "yes",
            "no",
            "not-assessable"
          ]
        },
        {
          "statementId": "7.1",
          "question": "Is this system covered by a documented institutional governance decision, rather than being used informally?",
          "statement": "Institutions should establish formal policies that either integrate validated AI tools into clinical workflows under defined governance, or clearly define acceptable boundaries for individual, unofficial use of external AI systems by clinicians.",
          "agreement": 89.7,
          "answers": [
            "yes",
            "no",
            "not-assessable"
          ]
        },
        {
          "statementId": "4.8",
          "question": "Is it documented whether this system is locally hosted or cloud-based, and is that choice recorded as a governance decision with its reasoning?",
          "statement": "The choice between deploying an open-source (locally hosted) system vs. a cloud-based commercial system should be documented as an institutional governance decision, with explicit consideration of data privacy, regulatory compliance, and the level of institutional control over the system.",
          "agreement": 82.8,
          "answers": [
            "yes",
            "no",
            "not-assessable"
          ]
        },
        {
          "statementId": "7.5",
          "question": "Is the regulatory status of this system in our jurisdiction documented?",
          "statement": "The regulatory classification of clinical AI systems should be explicitly considered in the evaluation and deployment framework.",
          "agreement": 86.2,
          "answers": [
            "yes",
            "no",
            "not-assessable"
          ]
        },
        {
          "statementId": "7.4",
          "question": "Does the documentation for this system state that it functions as decision support, with final responsibility for the clinical decision remaining with the treating clinician?",
          "statement": "Responsibility for clinical decisions should remain with the treating clinician, regardless of whether an AI system was used in the decision-making process.",
          "agreement": 89.7,
          "answers": [
            "yes",
            "no",
            "not-assessable"
          ]
        },
        {
          "statementId": "7.3",
          "question": "Is there a documented arrangement for informing patients when this system contributes to decisions about their care?",
          "statement": "When AI systems are formally integrated into institutional clinical workflows, institutions should disclose to patients that AI-assisted decision support is being used as part of their care.",
          "agreement": 81.5,
          "answers": [
            "yes",
            "no",
            "not-assessable"
          ]
        },
        {
          "statementId": "2.11",
          "question": "Is there a documented fallback procedure for when this system is unavailable or produces unreliable output?",
          "statement": "Institutions deploying clinical AI systems should ensure that adequate clinical expertise and fallback procedures are available in case the AI system becomes unavailable or produces unreliable outputs due to technical failure, system errors, or missed updates.",
          "agreement": 89.7,
          "answers": [
            "yes",
            "no",
            "not-assessable"
          ]
        }
      ]
    },
    {
      "section": "C",
      "title": "Evidence to obtain from the vendor",
      "source": "The supplier's documentation, or contract. It refers to the information that the vendor gives about the model.",
      "conditional": false,
      "conditionPrompt": null,
      "questions": [
        {
          "statementId": "5.9",
          "question": "Does the vendor provide evidence of a post-deployment safety monitoring arrangement, including structured feedback collection from clinical users?",
          "statement": "Vendors of clinical AI products should be required to support structured feedback collection from clinical users and to participate in post-deployment safety monitoring, analogous to existing post-market surveillance requirements for medical devices.",
          "agreement": 96.6,
          "answers": [
            "yes",
            "no",
            "not-assessable"
          ]
        },
        {
          "statementId": "5.8",
          "question": "Does the vendor provide a written division of responsibility between supplier and institution for continuous quality and safety monitoring?",
          "statement": "The responsibility for continuous quality monitoring and safety surveillance of clinical AI systems should be shared between the vendor or developer and the deploying institution, with clearly defined roles for each party.",
          "agreement": 82.8,
          "answers": [
            "yes",
            "no",
            "not-assessable"
          ]
        },
        {
          "statementId": "5.10",
          "question": "Does the vendor provide a reporting channel through which a clinician can report an error or unexpected output to both the institution and the developer?",
          "statement": "A structured reporting channel should be established for clinicians to report errors, unexpected outputs, and safety concerns to both institutional quality assurance and the AI system developer, enabling continuous post-deployment monitoring and early detection of systematic safety signals.",
          "agreement": 92.6,
          "answers": [
            "yes",
            "no",
            "not-assessable"
          ]
        },
        {
          "statementId": "5.6",
          "question": "Does the vendor provide evidence that the validation shown applies to the version we will actually be running?",
          "statement": "Validation of a specific AI model version should not automatically extend to successor versions or substantially updated releases. Each major version update should undergo independent evaluation proportionate to the scope of the changes.",
          "agreement": 93.1,
          "answers": [
            "yes",
            "no",
            "not-assessable"
          ]
        }
      ]
    },
    {
      "section": "D",
      "title": "Once the system is in use",
      "source": "Local monitoring records and local practice. It refers to systems already deployed. It should be answered from what can be observed in your own department, and repeated at each re-evaluation.",
      "conditional": true,
      "conditionPrompt": "Is the system already in use in your department?",
      "questions": [
        {
          "statementId": "5.3",
          "question": "Is there documented monitoring in place to detect performance falling off as guidelines change and new drugs are approved?",
          "statement": "After deployment, systematic monitoring should be in place to detect performance degradation over time, which may occur as medical knowledge evolves, new drugs are approved, or clinical guidelines are updated while the system's training data remains static.",
          "agreement": 100,
          "answers": [
            "yes",
            "no",
            "not-assessable"
          ]
        },
        {
          "statementId": "5.4",
          "question": "Is there a documented date, or trigger, at which this system will be formally re-evaluated?",
          "statement": "A defined schedule for re-evaluation of deployed AI systems should be established, for example triggered by major guideline updates, new drug approvals, changes in standard-of-care, or at regular intervals (e.g. annually).",
          "agreement": 96.6,
          "answers": [
            "yes",
            "no",
            "not-assessable"
          ]
        },
        {
          "statementId": "5.5",
          "question": "Are there documented rules for what happens when the vendor releases a new version, and a named person managing that?",
          "statement": "Clear rules should be established for when and how a deployed AI system is updated to a new version, and what level of re-validation is required before the new version replaces the previous one in clinical use. Institutions should have dedicated expertise available to manage this process.",
          "agreement": 96.6,
          "answers": [
            "yes",
            "no",
            "not-assessable"
          ]
        },
        {
          "statementId": "5.7",
          "question": "Is there a documented arrangement linking AI-supported decisions to what actually happened to those patients?",
          "statement": "Post-deployment monitoring should, where feasible, link AI-supported clinical decisions to actual patient outcomes (e.g. treatment response rates, progression-free survival, adverse event rates) over time, to assess the real-world impact of the tool.",
          "agreement": 89.7,
          "answers": [
            "yes",
            "no",
            "not-assessable"
          ]
        },
        {
          "statementId": "6.9",
          "question": "Is there evidence that over-reliance (i.e. clinicians accept incorrect outputs without independent review) is being watched for?",
          "statement": "Evaluation studies should assess the risk of over-reliance (automatic bias) on AI, including clinician acceptance of incorrect outputs without adequate independent review.",
          "agreement": 89.7,
          "answers": [
            "yes",
            "no",
            "not-assessable"
          ]
        },
        {
          "statementId": "5.11",
          "question": "Is there evidence that output remains stable across our hardware, software versions and interface updates?",
          "statement": "AI system outputs should demonstrate acceptable stability across different technical configurations (e.g. hardware, software versions, API updates) without clinically meaningful changes in recommendations.",
          "agreement": 86.2,
          "answers": [
            "yes",
            "no",
            "not-assessable"
          ]
        }
      ]
    }
  ]
}
