


@Proceedings{ProbML2026,
  title =     {Proceedings of The 1st Symposium on Probabilistic Machine Learning},
  booktitle = {Proceedings of The 1st Symposium on Probabilistic Machine Learning},
  editor =    {Siddharth Swaroop and David Rügamer and Agustinus Kristiadi},
  publisher = {PMLR},
  series =    {Proceedings of Machine Learning Research},
  volume =    327
}



@InProceedings{pmlr-v327-sinaga26a,
  title = 	 {Anchor-Based Heteroscedastic Noise for Preferential Bayesian Optimization},
  author =       {Sinaga, Marshal Arijona and Martinelli, Julien and Kaski, Samuel},
  booktitle = 	 {Proceedings of The 1st Symposium on Probabilistic Machine Learning},
  pages = 	 {1--26},
  year = 	 {2026},
  editor = 	 {Swaroop, Siddharth and Rügamer, David and Kristiadi, Agustinus},
  volume = 	 {327},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {05 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/v327/main/assets/sinaga26a/sinaga26a.pdf},
  url = 	 {https://proceedings.mlr.press/v327/sinaga26a.html},
  abstract = 	 { Preferential Bayesian optimization (PBO) learns latent utilities from pairwise comparisons, but most existing methods assume homoscedastic comparison noise. This is inadequate in human-in-the-loop settings, where a user may compare some designs reliably and others only hesitantly. We propose a heteroscedastic noise model for PBO: before optimization, the user provides a small set of reliable examples, called anchors, and a kernel density estimator (KDE) turns these anchors into an input-dependent map of user uncertainty. We incorporate this map into preferential GP surrogates and derive risk-averse acquisition functions that trade off utility and ease of comparison. We further show that a risk-adjusted variant of the popular expected utility of the best option (EUBO) preserves the one-step Bayes-optimality guarantee up to an additive constant, and that under an idealized i.i.d. anchor model the KDE estimator enjoys standard consistency and concentration rates. Experiments on synthetic problems and human-preference datasets show improved risk-adjusted performance and clarify how anchor placement affects the method. }
}



@InProceedings{pmlr-v327-xuan26a,
  title = 	 {Wavelet Conditional Neural Processes},
  author =       {Xuan, Junyu and Wu, Mengjing},
  booktitle = 	 {Proceedings of The 1st Symposium on Probabilistic Machine Learning},
  pages = 	 {27--49},
  year = 	 {2026},
  editor = 	 {Swaroop, Siddharth and Rügamer, David and Kristiadi, Agustinus},
  volume = 	 {327},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {05 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/v327/main/assets/xuan26a/xuan26a.pdf},
  url = 	 {https://proceedings.mlr.press/v327/xuan26a.html},
  abstract = 	 { Conditional neural processes (CNPs) are a new family of stochastic processes defined by deep neural networks, characterized by the necessary properties of marginal consistency and exchangeability. Thanks to their generalization capabilities across tasks, popular applications of CNPs include meta-learning and multi-task learning. The existing CNPs map a context set to a vector or function space where all samples are considered homogeneously, which limits their representational power. In this paper, we introduce a Wavelet Conditional Neural Process (WaveCNP) as a new member of the CNP family, based on wavelet transform theory. We propose mapping the context set into a nested multiresolution function space sequence rather than a singular space, achieved through the efficient and adaptive discrete wavelet transform. We demonstrate that our WaveCNP can outperform existing CNPs in terms of conditional predictive distribution modeling and multiresolution prediction. }
}



@InProceedings{pmlr-v327-saravanan26a,
  title = 	 {Universality of Singular Complexity for Hyvärinen Generalized Bayes: Exact Transfer in Gaussian Factor Analysis},
  author =       {Saravanan, Manoj and Salla, Rohit Kumar},
  booktitle = 	 {Proceedings of The 1st Symposium on Probabilistic Machine Learning},
  pages = 	 {50--85},
  year = 	 {2026},
  editor = 	 {Swaroop, Siddharth and Rügamer, David and Kristiadi, Agustinus},
  volume = 	 {327},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {05 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/v327/main/assets/saravanan26a/saravanan26a.pdf},
  url = 	 {https://proceedings.mlr.press/v327/saravanan26a.html},
  abstract = 	 { Singular statistical models are asymptotically governed by local birational invariants such as the real log canonical threshold (RLCT), rather than by ambient parameter dimension. We study this singular complexity under generalized Bayes updates based on the Hyvärinen loss. We first prove a local comparison theorem: nonnegative analytic excess-loss germs that are locally comparable near a common zero set have the same local RLCT pair. As a corollary, losses that share an analytic minimizer map and have a nondegenerate quadratic germ lie in the same singular universality class. We then show that in analytic zero-mean Gaussian covariance models, both the population excess Gaussian log-loss and the excess Hyvärinen loss are locally equivalent to $\lVert \Sigma(\theta)- \Sigma_0 \rVert_{\mathrm{F}}^2$. Consequently, ordinary Gaussian Bayes and Hyvärinen generalized Bayes have identical local RLCT pairs at every covariance fiber. Applying this transfer principle to Gaussian factor analysis yields exact transfer of known local learning-coefficient results from ordinary Gaussian Bayes. In the one-factor model, the Hyvärinen posterior therefore has coefficients $p$, $(2p-1)/2$, and $3p/4$ on the corresponding covariance strata. Numerical experiments confirm the predicted local quadratic equivalence and the cancellation of the shared $ \lambda \log n$ term in paired free-energy differences. }
}



@InProceedings{pmlr-v327-grunefeld26a,
  title = 	 {An Isotropic Approach to Efficient Uncertainty Quantification with Gradient Norms},
  author =       {Gr{\"u}nefeld, Nils and Frellsen, Jes and Hardmeier, Christian},
  booktitle = 	 {Proceedings of The 1st Symposium on Probabilistic Machine Learning},
  pages = 	 {86--118},
  year = 	 {2026},
  editor = 	 {Swaroop, Siddharth and Rügamer, David and Kristiadi, Agustinus},
  volume = 	 {327},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {05 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/v327/main/assets/grunefeld26a/grunefeld26a.pdf},
  url = 	 {https://proceedings.mlr.press/v327/grunefeld26a.html},
  abstract = 	 { Existing methods for quantifying predictive uncertainty in neural networks are either computationally intractable for large language models or require access to training data that is typically unavailable. We derive a lightweight alternative through two approximations: a first-order Taylor expansion that expresses uncertainty in terms of the gradient of the prediction and the parameter covariance, and an isotropy assumption on the parameter covariance. Together, these yield epistemic uncertainty as the squared gradient norm and aleatoric uncertainty as the Bernoulli variance of the point prediction, from a single forward-backward pass through an unmodified pretrained model. We justify the isotropy assumption by showing that covariance estimates built from non-training data introduce structured distortions that isotropic covariance avoids, and that theoretical results on the spectral properties of large networks support the approximation at scale. Validation against reference Markov Chain Monte Carlo estimates on synthetic problems shows strong correspondence that improves with model size. We then use the estimates to investigate when each uncertainty type carries useful signal for predicting answer correctness in question answering with large language models, revealing a benchmark-dependent divergence: the combined estimate achieves the highest mean AUROC on TruthfulQA, where questions involve genuine conflict between plausible answers, but falls to near chance on TriviaQA’s factual recall, suggesting that parameter-level uncertainty captures a fundamentally different signal than self-assessment methods. }
}



@InProceedings{pmlr-v327-rahman26a,
  title = 	 {Causal Temporal Graphs for Counterfactual Validation of Temporal Link Prediction},
  author =       {Rahman, Aniq Ur and Coon, Justin},
  booktitle = 	 {Proceedings of The 1st Symposium on Probabilistic Machine Learning},
  pages = 	 {119--134},
  year = 	 {2026},
  editor = 	 {Swaroop, Siddharth and Rügamer, David and Kristiadi, Agustinus},
  volume = 	 {327},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {05 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/v327/main/assets/rahman26a/rahman26a.pdf},
  url = 	 {https://proceedings.mlr.press/v327/rahman26a.html},
  abstract = 	 { Temporal link prediction (TLP) models are commonly evaluated based on predictive accuracy, yet such evaluations do not assess whether these models capture the causal mechanisms that govern temporal interactions. In this work, we propose a framework for counterfactual validation of TLP models by generating causal temporal interaction graphs (CTIGs) with known ground-truth causal structure. We first introduce a structural equation model for continuous-time event sequences that supports both excitatory and inhibitory effects, and then extend this mechanism to temporal interaction graphs. To compare causal models, we propose a divergence metric based on cross-model predictive error, and empirically validate the hypothesis that predictors trained on one causal model degrade when evaluated on sufficiently distant models. Finally, we instantiate counterfactual evaluation under (i) controlled causal shifts between generating models and (ii) timestamp shuffling as a stochastic distortion with measurable causal divergence. Our framework provides a foundation for causality-aware benchmarking. }
}



@InProceedings{pmlr-v327-lu26a,
  title = 	 {Neural Stochastic Differential Equations on Compact State Spaces: Theory, Methods, and Application to Suicide Risk Modeling},
  author =       {Lu, Malinda and Liu, Yue-Jane and Nock, Matthew K. and Yacoby, Yaniv},
  booktitle = 	 {Proceedings of The 1st Symposium on Probabilistic Machine Learning},
  pages = 	 {135--188},
  year = 	 {2026},
  editor = 	 {Swaroop, Siddharth and Rügamer, David and Kristiadi, Agustinus},
  volume = 	 {327},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {05 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/v327/main/assets/lu26a/lu26a.pdf},
  url = 	 {https://proceedings.mlr.press/v327/lu26a.html},
  abstract = 	 { Ecological Momentary Assessment (EMA) studies enable the collection of high-frequency self-reports of suicidal thoughts and behaviors (STBs) via smartphones. Latent stochastic differential equations (SDEs) are a promising model class for EMA data, as it is irregularly sampled, noisy, and partially observed. But SDE-based models suffer from two key limitations. (a) These models often violate domain constraints, undermining scientific validity and clinical trust of the model. (b) Training is numerically unstable without ad hoc fixes (e.g. oversimplified dynamics) that are ill-suited for high-stakes applications. Here, we develop a novel class of expressive SDEs whose solutions are provably confined to a prescribed compact polyhedral state space, matching the domains of EMA data. In this work, (1) we show why chain-rule based constructions of SDEs on compact domains fail, theoretically and empirically; (2) we derive constraints on drift and diffusion for general and stationary SDEs so their solutions remain in the desired state space; and (3), we introduce a parameterization that maps arbitrary (neural or expert-given) dynamics into constraint-satisfying SDEs. On several real EMA datasets, including a large suicide-risk study, our parameterization improves forecasts and optimization dynamics over standard latent neural SDE baselines. These contributions pave the way for principled, trustworthy continuous-time models of suicide risk and other clinical time series and extend applications of SDE-based methods (e.g. diffusion models) to domains with hard state constraints. }
}



@InProceedings{pmlr-v327-chang26a,
  title = 	 {Conditionally Identifiable Latent Representation for Multivariate Time Series with Structural Dynamics},
  author =       {Chang, Minkey and Kim, Jae-Young},
  booktitle = 	 {Proceedings of The 1st Symposium on Probabilistic Machine Learning},
  pages = 	 {189--213},
  year = 	 {2026},
  editor = 	 {Swaroop, Siddharth and Rügamer, David and Kristiadi, Agustinus},
  volume = 	 {327},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {05 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/v327/main/assets/chang26a/chang26a.pdf},
  url = 	 {https://proceedings.mlr.press/v327/chang26a.html},
  abstract = 	 { We propose the Identifiable Variational Dynamic Factor Model (iVDFM) for multivariate time series. The model combines variational inference with \emph{partial} identifiability guarantees up to a known ambiguity class. Innovations follow a conditional exponential-family prior over observed auxiliary context and regime embeddings. Linear diagonal dynamics map innovations to factors and preserve that class, so factors are partially identifiable up to permutation and component-wise affine maps. We train by maximizing the ELBO and estimate uncertainty in both latent trajectories and forecasts. Under explicit assumptions, we prove partial identifiability up to the stated class and provide the full proof in the appendix. We evaluate factor recovery on synthetic DGPs, intervention behavior on synthetic SCMs, and forecasting on real benchmarks with CRPS and MSE. }
}



@InProceedings{pmlr-v327-wang26a,
  title = 	 {When Individually Calibrated Models Become Collectively Miscalibrated },
  author =       {Wang, Zhaohui Geoffrey},
  booktitle = 	 {Proceedings of The 1st Symposium on Probabilistic Machine Learning},
  pages = 	 {214--255},
  year = 	 {2026},
  editor = 	 {Swaroop, Siddharth and Rügamer, David and Kristiadi, Agustinus},
  volume = 	 {327},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {05 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/v327/main/assets/wang26a/wang26a.pdf},
  url = 	 {https://proceedings.mlr.press/v327/wang26a.html},
  abstract = 	 { Probabilistic prediction systems often aggregate probability estimates from multiple models into a single decision. A natural assumption is that if each model is individually calibrated, the aggregate prediction will also be well calibrated. We show that this assumption fails in multi-agent settings: individually calibrated predictors can become collectively miscalibrated when their predictions interact strategically. This phenomenon arises in settings where predictions originate from multiple independent agents, including federated healthcare, multi-vendor intrusion detection, and crowdsourced forecasting, where agents optimize their own objectives. Specifically, we prove that under Brier-score-based aggregation with correlated beliefs, each agent’s individually optimal report systematically underestimates the positive-class probability, producing a Price of Anarchy (PoA) of 7.25x (mean aggregate bias -0.375). In contrast, VCG-based aggregation, which rewards each agent’s marginal contribution to aggregate accuracy, achieves the lowest PoA among all mechanisms studied (PoA $\approx$ 1.0x under dominant-strategy equilibrium). On three real-world datasets (NSL-KDD, UNSW-NB15, Credit Card Fraud) with feature-partitioned agents, VCG provides the strongest robustness guarantees among the aggregation methods we evaluate, while maintaining comparable accuracy. In data-sparse regimes ($n \leq 500$), VCG consistently outperforms stacking and majority voting; under adversarial agents, VCG maintains substantially lower false-negative rates than robust aggregation baselines. Adaptive weight updates further reduce false negatives by 20–22% under distribution shift, with $O(\sqrt{T})$ online regret guarantees. These results establish that how probabilistic predictions are aggregated matters as much as how well individual models are calibrated. }
}



@InProceedings{pmlr-v327-young26a,
  title = 	 {Characterizing the Representational Capacity of Neural Processes},
  author =       {Young, Robin},
  booktitle = 	 {Proceedings of The 1st Symposium on Probabilistic Machine Learning},
  pages = 	 {256--289},
  year = 	 {2026},
  editor = 	 {Swaroop, Siddharth and Rügamer, David and Kristiadi, Agustinus},
  volume = 	 {327},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {05 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/v327/main/assets/young26a/young26a.pdf},
  url = 	 {https://proceedings.mlr.press/v327/young26a.html},
  abstract = 	 { What functions can Neural Processes represent? We analyze the representational capacity of popular NP architectures: Conditional Neural Processes (CNPs), Attentive Neural Processes (ANPs), Transformer Neural Processes (TNPs), and their latent variants. We prove these architectures form a strict hierarchy. CNP-representable functions are exactly those depending on finitely many expected features of the context distribution. ANPs strictly generalize CNPs via query-dependent reweighting, enabling kernel smoothers. ConvCNPs and ANPs are incomparable; each contains functions outside the other, separated by stationarity versus translation equivariance. TNPs with $L$ self-attention layers capture $L$-hop context interactions. For latent NPs, we show finite-dimensional latents provide coherent sampling but do not circumvent encoder limitations; matching GP posterior distributions requires latent dimension scaling with context size. These results provide a theoretical foundation for architecture selection based on task structure. }
}



@InProceedings{pmlr-v327-yeo26a,
  title = 	 {Identifiability, Fisher Information, and Amortized Inference for Heterogeneous Diffusion from Discrete-Time Noisy Observations},
  author =       {Yeo, Zhen Yuan},
  booktitle = 	 {Proceedings of The 1st Symposium on Probabilistic Machine Learning},
  pages = 	 {290--313},
  year = 	 {2026},
  editor = 	 {Swaroop, Siddharth and Rügamer, David and Kristiadi, Agustinus},
  volume = 	 {327},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {05 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/v327/main/assets/yeo26a/yeo26a.pdf},
  url = 	 {https://proceedings.mlr.press/v327/yeo26a.html},
  abstract = 	 { Estimating diffusion coefficients from discrete-time noisy particle trajectories is a core problem in single-particle tracking , yet its statistical foundations remain incomplete. We establish exact identifiability structure for single-species and heterogeneous-population models, derive closed-form Fisher information bounds, and show how these results jointly define a feasibility phase diagram over the normalized diffusion scale $ \alpha = 2D\Delta t/\sigma^2$ and effective particle occupancy $ \beta = \rho\sigma^2$. A key finding is that one-step increment distributions alone cannot separate the diffusion coefficient from localization noise, but temporal autocorrelation structure resolves this ambiguity without additional calibration. For heterogeneous populations, identifiability holds under a variance separation condition, and Fisher information for rare components degrades quadratically with mixture weight, setting a fundamental limit on what unlabeled trajectory data can recover. We use these theoretical results to derive an amortized inference architecture: the time-averaging aggregation, factored posterior parameterization , and pretraining objective each follow directly from the theory. }
}



@InProceedings{pmlr-v327-tang26a,
  title = 	 {Heterogeneous Coupled Diffusion for Graph Generation with $α$-Stable Node Feature Noise},
  author =       {Tang, Chengyu and Kuruoglu, Ercan Engin},
  booktitle = 	 {Proceedings of The 1st Symposium on Probabilistic Machine Learning},
  pages = 	 {314--345},
  year = 	 {2026},
  editor = 	 {Swaroop, Siddharth and Rügamer, David and Kristiadi, Agustinus},
  volume = 	 {327},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {05 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/v327/main/assets/tang26a/tang26a.pdf},
  url = 	 {https://proceedings.mlr.press/v327/tang26a.html},
  abstract = 	 { Graph generation requires jointly modeling node semantics and graph structure. Existing coupled graph diffusion models mostly use light-tailed perturbations throughout the graph state, which may be too conservative for node features that encode rare semantic roles or long-tailed attribute patterns. We propose a heterogeneous coupled graph diffusion model that applies discrete-time symmetric $\alpha$-stable noise to node features and Gaussian DDPM noise to adjacency, while keeping reverse denoising joint over the full graph state. Using a Gaussian scale-mixture representation, we derive an exact augmented variational objective for the stable feature branch and a same-marginal coupled deterministic sampler. On a controlled benchmark, stable feature diffusion improves rare-role recall and joint generation quality, while Gaussian adjacency provides the more reliable structural bias. On molecular and generic graph benchmarks, keeping adjacency Gaussian and varying only the feature tail index often improves over both a GDSS baseline retrained under our setup and a Gaussian-limit variant, though the best tail index remains dataset-dependent. These results support stable diffusion on features in coupled graph diffusion, while the case for Gaussian adjacency is primarily established by the controlled probe. }
}



@InProceedings{pmlr-v327-lo26a,
  title = 	 {Uncertainty Propagation Through Green’s Kernels and Gaussian Process Inference Dynamics},
  author =       {Lo, Chi-Jen Roger and Lasenby, Joan},
  booktitle = 	 {Proceedings of The 1st Symposium on Probabilistic Machine Learning},
  pages = 	 {346--366},
  year = 	 {2026},
  editor = 	 {Swaroop, Siddharth and Rügamer, David and Kristiadi, Agustinus},
  volume = 	 {327},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {05 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/v327/main/assets/lo26a/lo26a.pdf},
  url = 	 {https://proceedings.mlr.press/v327/lo26a.html},
  abstract = 	 { We study uncertainty propagation under Gaussian process priors whose covariance is induced by Green or resolvent operators associated with latent generators. Rather than specifying non-stationary dependence through input warping or other coordinate deformations, we construct covariance from discounted propagation under an underlying semigroup. This yields a separation between covariance geometry and transport: the self-adjoint dissipative part of the generator provides the operator from which prior covariance is constructed, while the skew-adjoint part contributes transport within the full propagation without being encoded directly as covariance. We derive observation, posterior, and propagation identities in operator form, and compare this framework with pullback, stationary, and Matérn baselines. Experiments are organised as a four-stage ladder ascending from three-dimensional electrostatics through a frequency-domain Helmholtz proxy and time-domain semigroup propagation to an operator-mismatch ablation. At each stage, the Green prior is realised as a Gaussian Markov random field (GMRF) whose precision is the discretised PDE operator; after assembly, the posterior mean is obtained by a single sparse linear solve. Across the reported experiments, Green’s kernel priors often outperform tuned Matérn-$3/2$ and RBF baselines on boundary-aware and PDE-residual metrics, with larger gains under heterogeneous material coefficients and semigroup propagation. In the late-time propagation setting the normalised RMSE gap exceeds $4{\times}$. The central claim is that, when an informative governing operator is available, covariance induced by that operator yields boundary-respecting and dynamically interpretable uncertainty—especially on operator-sensitive diagnostics—under the reported synthetic conditions. }
}



@InProceedings{pmlr-v327-fox26a,
  title = 	 {{RAMP}: Recognition parametrisation by Amortised Message Passing},
  author =       {Fox, Lior and Biegun, Kai and Heald, James and Hromadka, Samo and Rosinski, Arielle and Sahani, Maneesh},
  booktitle = 	 {Proceedings of The 1st Symposium on Probabilistic Machine Learning},
  pages = 	 {367--397},
  year = 	 {2026},
  editor = 	 {Swaroop, Siddharth and Rügamer, David and Kristiadi, Agustinus},
  volume = 	 {327},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {05 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/v327/main/assets/fox26a/fox26a.pdf},
  url = 	 {https://proceedings.mlr.press/v327/fox26a.html},
  abstract = 	 { A central aim of unsupervised learning is to uncover latent factors that explain dependencies among observations. Probabilistic models typically achieve this by introducing multiple latent variables linked through a graph of conditional relationships, with distributional parameters and their dependence learnt from data. Learning relies either on distributional choices that allow tractable belief propagation, or on approximations that scale poorly with model size and complexity. We build on the recently developed recognition-parametrised modelling paradigm to propose an alternative approach: RAMP, a method that implicitly defines latent structure by learning a flexible, nonlinear, amortised message-passing framework. We show that RAMP enables efficient likelihood-based recovery of latent-variable distributions within expressive nonlinear models acting on complex high-dimensional data. }
}



@InProceedings{pmlr-v327-mahadew26a,
  title = 	 {Intention Inference Under Execution Noise: Separating Aleatoric and Epistemic Uncertainty in Social Dilemmas},
  author =       {Mahadew, Kival and Shock, Jonathan P.},
  booktitle = 	 {Proceedings of The 1st Symposium on Probabilistic Machine Learning},
  pages = 	 {398--424},
  year = 	 {2026},
  editor = 	 {Swaroop, Siddharth and Rügamer, David and Kristiadi, Agustinus},
  volume = 	 {327},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {05 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/v327/main/assets/mahadew26a/mahadew26a.pdf},
  url = 	 {https://proceedings.mlr.press/v327/mahadew26a.html},
  abstract = 	 {In noisy social dilemmas, intended actions are stochastically corrupted before execution, so an observed defection may reflect hostile intent or action error. Standard Markov Decision Process (MDP) formulations treat executed actions as states, structurally precluding this distinction and causing systematic over retaliation. We introduce a Partially Observable MDP (POMDP) formulation encoding opponent intentions as latent states and executed actions as noisy observations, solved within the active inference (AIF) framework with a cost function that decomposes into epistemic and pragmatic components that jointly address inferring current intent and learning how intent evolves. In the Iterated Prisoner’s Dilemma with symmetric noise, we derive a critical noise threshold governing cooperation collapse, connecting it to a fixed-point condition on learned priors. Experiments reveal that the value of intention inference is context-dependent: the POMDP provides consistent advantages against conditionally cooperative opponents, but mutual intention inference under sufficient noise produces correlated belief-driven collapse. The advantage is specific to games where intent attribution is decision-relevant.}
}



@InProceedings{pmlr-v327-amiri26a,
  title = 	 {Latent Semantic Regularization: Enhancing Semantic Integrity in Tabular Data Synthesis},
  author =       {Amiri, Saba and Nijhuis, Carlijn and Nalisnick, Eric and Belloum, Adam and Klous, Sander and Gommans, Leon},
  booktitle = 	 {Proceedings of The 1st Symposium on Probabilistic Machine Learning},
  pages = 	 {425--443},
  year = 	 {2026},
  editor = 	 {Swaroop, Siddharth and Rügamer, David and Kristiadi, Agustinus},
  volume = 	 {327},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {05 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/v327/main/assets/amiri26a/amiri26a.pdf},
  url = 	 {https://proceedings.mlr.press/v327/amiri26a.html},
  abstract = 	 { Generative models trained on finite data could potentially assign non-negligible probability mass to regions that are statistically plausible but semantically invalid as a structural consequence of distribution estimation from samples. We propose Latent Semantic Regularization (LSR), which learns implicit semantic constraints directly from data and uses them to regularize a generative model, requiring no explicit constraint definitions, feasibility oracles, or negative examples. We instantiate LSR with a Conditional VAE using a tail- adaptive normalizing flow prior as the validator and a WGAN-GP as the synthesizer. We also introduce Exploration Factor metric measuring exploration/exploitation, and a controlled semantic integrity protocol based on injected zero-probability constraints with known ground truth. Across five benchmark datasets and against GAN- based, diffusion-based, and LLM-based baselines, LSR reduces semantic violation rates consistently while remaining competitive on statistical fidelity and downstream utility. }
}



