@article{obreja_characterizing_2025,
title = {Characterizing the Impact of Training Data on Generalizability: Application in Deep Learning to Estimate Lung Nodule Malignancy Risk},
author = {B. Obreja and J. Bosma and K. V. Venkadesh and Z. Saghir and M. Prokop and C. Jacobs},
url = {https://pubs.rsna.org/doi/10.1148/ryai.240636},
doi = {10.1148/ryai.240636},
year = {2025},
date = {2025-11-01},
urldate = {2025-11-01},
journal = {Radiology: Artificial Intelligence},
volume = {7},
number = {6},
pages = {e240636},
abstract = {Purpose To investigate the relationship between training data volume and performance of a deep learning artificial intelligence (AI) algorithm developed to assess the malignancy risk of pulmonary nodules detected on low-dose CT scans in lung cancer screening.Materials and Methods This retrospective study used a dataset of 16 077 annotated nodules (1249 malignant, 14 828 benign) from the National Lung Screening Trial (NLST) to systematically train an AI algorithm for pulmonary nodule malignancy risk prediction across various stratified subsets ranging from 1.25% to the full dataset. External testing was conducted using data from the Danish Lung Cancer Screening Trial (DLCST) to determine the amount of training data at which the performance of the AI was statistically noninferior to the AI trained on the full NLST cohort. A size-matched cancer-enriched subset of DLCST, in which each malignant nodule had been paired in diameter with the closest two benign nodules, was used to investigate the amount of training data at which the performance of the AI algorithm was statistically noninferior to the average performance of 11 clinicians.Results The external testing set included 599 participants (mean age ± SD, 57.65 years ± 4.84 for female participants and 59.03 years ± 4.94 for male participants) with 883 nodules (65 malignant, 818 benign). The AI achieved a mean area under the receiver operating characteristic curve (AUC) of 0.92 (95% CI: 0.88, 0.96) on the DLCST cohort when trained on the full NLST dataset. Training with 80% of the NLST data resulted in noninferior performance (mean AUC, 0.92; 95% CI: 0.89, 0.96; P = .005). On the size-matched DLCST subset (59 malignant, 118 benign), the AI reached noninferior clinician-level performance (mean AUC, 0.82; 95% CI: 0.77, 0.86) with 20% of the training data (P = .02).Conclusion The deep learning AI algorithm demonstrated excellent performance in assessing pulmonary nodule malignancy risk, achieving clinical level performance with a fraction of the training data and reaching peak performance before using the full dataset. Keywords: Convolutional Neural Network (CNN), CT, Lung, Screening, Diagnosis, Supervised Learning, Lung Cancer Screening, Pulmonary Nodule Malignancy Risk, Deep Learning, Pulmonary Nodule Management Supplemental material is available for this article. © RSNA, 2025 See also commentary by Archer in this issue.},
note = {Publisher: Radiological Society of North America},
keywords = {},
pubstate = {published},
tppubtype = {article}
}