@InProceedings{pmlr-v267-prinster25a,
title = {{WATCH}: Adaptive Monitoring for {AI} Deployments via Weighted-Conformal Martingales},
author = {Prinster, Drew and Han, Xing and Liu, Anqi and Saria, Suchi},
booktitle = {Proceedings of the 42nd International Conference on Machine Learning},
pages = {49830--49859},
year = {2025},
editor = {Singh, Aarti and Fazel, Maryam and Hsu, Daniel and Lacoste-Julien, Simon and Berkenkamp, Felix and Maharaj, Tegan and Wagstaff, Kiri and Zhu, Jerry},
volume = {267},
series = {Proceedings of Machine Learning Research},
month = {13--19 Jul},
publisher = {PMLR},
url = {https://proceedings.mlr.press/v267/prinster25a.html}
}
This paper develops a framework for the continual safety monitoring of AI deployments called “WATCH” (for Weighted Adaptive Testing for Changepoint Hypotheses). This framework is centered on a weighted generalization of conformal test martingales (WCTMs). WATCH addresses three main challenges in post-deployment AI monitoring: (1) Adaptation: WATCH enables monitoring under test-time adaptation to mild (covariate) shifts, to minimize unnecessary alarms. (2) Fast Detection: Empirically, WATCH rapidly detects more extreme or harmful shifts. (3) Root-Cause Analysis: WATCH aids in diagnosing the root-cause of performance degradation (as a covariate shift in $X$, or a concept shift in $Y \mid X$).