@article{TCR120210,
author = {Jun Yan and Lijun Sun and Song Wang},
title = {A weakly supervised deep learning model for OSCC diagnosis and lesion highlighting in histopathology images},
journal = {Translational Cancer Research},
volume = {15},
number = {7},
year = {2026},
keywords = {},
abstract = {Background: Oral squamous cell carcinoma (OSCC) is the predominant histopathological subtype of oral malignancies, and histopathology-based diagnosis remains central to clinical management. With the rapid development of digital pathology and deep learning, automated analysis of histopathology images has shown considerable potential for improving screening efficiency and reducing inter-observer variability. However, in both public datasets and real-world research settings, only image-level labels are commonly available, whereas region-level annotations are often lacking. This limitation makes weakly supervised modeling necessary and also restricts rigorous validation of spatial interpretability.Methods: To address these challenges, we developed a weakly supervised deep learning framework based on patch-level representation learning and multiple-instance aggregation for image-level OSCC diagnosis with exploratory heatmap visualization. Specifically, histopathology images were divided into patches and encoded into feature embeddings. UNI 2, a recently developed pathology foundation model with highly competitive performance in contemporary computational pathology benchmarks, was adopted as the primary feature extractor to obtain informative patch representations. ResNet-50 was additionally implemented as a conventional baseline encoder for comparative evaluation. A self-attention-based multiple-instance aggregation module was then introduced to capture dependencies among instances within each patch set and to generate binary OSCC predictions.Results: In five-fold cross-validation, the UNI-2-based model achieved a mean area under the curve (AUC) of 0.9919, compared with 0.9899 for the ResNet-50 baseline. On the independent external validation cohort, UNI 2 further achieved an AUC of 0.9992, outperforming ResNet-50, which achieved an AUC of 0.9564. As an exploratory interpretability analysis, attention responses were visualized as heatmaps to inspect model focus patterns and support qualitative error review, without assuming verified lesion-level correspondence in the absence of region-level annotations.Conclusions: This framework demonstrates strong image-level diagnostic performance for OSCC and provides exploratory visualization for qualitative model review without requiring additional fine-grained annotations, offering a reusable technical pathway for weakly supervised OSCC histopathology modeling and future multicenter validation.},
issn = {2219-6803}, url = {https://tcr.amegroups.org/article/view/120210}
}