


@Proceedings{UAI2018,
  title =     {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  booktitle = {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  editor =    {Amir Globerson and Ricardo Silva},
  publisher = {PMLR},
  series =    {Proceedings of Machine Learning Research},
  volume =    R16
}



@InProceedings{pmlr-vR16-jin18a,
  title = 	 {Testing for Conditional Mean Independence with Covariates through Martingale Difference Divergence},
  author =       {Jin, Ze and Yan, Xiaohan and Matteson, David S.},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {1--11},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/jin18a/jin18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/jin18a.html},
  abstract = 	 {A crucial problem in statistics is to decide whether additional variables are needed in a regression model. We propose a new multivariate test to investigate the conditional mean independence of Y given X conditioning on some known effect Z, i.e., E(Y|X, Z) = E(Y|Z). Assuming that E(Y|Z) and Z are linearly related, we reformulate an equivalent notion of conditional mean independence through transformation, which is approximated in practice. We apply the martingale difference divergence (Shao and Zhang, 2014) to measure conditional mean dependence, and show that the estimation error from approximation is negligible, as it has no impact on the asymptotic distribution of the test statistic under some regularity assumptions. The implementation of our test is demonstrated by both simulations and a financial data example.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-liu18a,
  title = 	 {Analysis of {T}hompson Sampling for Graphical Bandits Without the Graphs},
  author =       {Liu, Fang and Zheng, Zizhan and Shroff, Ness},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {12--21},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/liu18a/liu18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/liu18a.html},
  abstract = 	 {We study multi-armed bandit problems with graph feedback, in which the decision maker is allowed to observe the neighboring actions of the chosen action, in a setting where the graph may vary over time and is never fully revealed to the decision maker. We show that when the feedback graphs are undirected, the original Thompson Sampling achieves the optimal (within logarithmic factors) regret $\tilde{}$O (p $\beta$0(G)T ) over time horizon T, where $\beta$0(G) is the average independence number of the latent graphs. To the best of our knowl- edge, this is the first result showing that the original Thompson Sampling is optimal for graphical bandits in the undirected setting. A slightly weaker regret bound of Thompson Sampling in the directed setting is also pre- sented. To fill this gap, we propose a variant of Thompson Sampling, that attains the opti- mal regret in the directed setting within a log- arithmic factor. Both algorithms can be im- plemented efficiently and do not require the knowledge of the feedback graphs at any time.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-gregorova18a,
  title = 	 {Structured nonlinear variable selection},
  author =       {Gregorova, Magda and Kalousis, Alexandros and Marchand-Maillet, Stephane},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {22--31},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/gregorova18a/gregorova18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/gregorova18a.html},
  abstract = 	 {We investigate structured sparsity methods for variable selection in regression problems where the target depends nonlinearly on the inputs. We focus on general nonlinear func- tions not limiting a priori the function space to additive models. We propose two new regu- larizers based on partial derivatives as nonlin- ear equivalents of group lasso and elastic net. We formulate the problem within the frame- work of learning in reproducing kernel Hilbert spaces and show how the variational problem can be reformulated into a more practical fi- nite dimensional equivalent. We develop a new algorithm derived from the ADMM principles that relies solely on closed forms of the proxi- mal operators. We explore the empirical prop- erties of our new algorithm for Nonlinear Vari- able Selection based on Derivatives (NVSD) on a set of experiments and confirm favourable properties of our structured-sparsity models and the algorithm in terms of both prediction and variable selection accuracy.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-pena18a,
  title = 	 {Identification of Strong Edges in {AMP} Chain Graphs},
  author =       {Pe{\~n}a, Jose M.},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {32--41},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/pena18a/pena18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/pena18a.html},
  abstract = 	 {The essential graph is a distinguished member of a Markov equivalence class of AMP chain graphs. However, the directed edges in the es- sential graph are not necessarily strong or in- variant, i.e. they may not be shared by every member of the equivalence class. Likewise for the undirected edges. In this paper, we develop a procedure for identifying which edges in an essential graph are strong. We also show how this makes it possible to bound some causal ef- fects when the true chain graph is unknown.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-lyu18a,
  title = 	 {A Univariate Bound of Area Under {ROC}},
  author =       {Lyu, Siwei and Ying, Yiming},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {42--51},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/lyu18a/lyu18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/lyu18a.html},
  abstract = 	 {Area under ROC (AUC) is an important met- ric for binary classification and bipartite rank- ing problems. However, it is difficult to di- rectly optimize AUC as a learning objective, so most existing algorithms are based on optimiz- ing a surrogate loss to AUC. One significant drawback of these surrogate losses is that they require pairwise comparisons among training data, which leads to slow running time and in- creasing local storage for online learning. In this work, we describe a new surrogate loss based on a reformulation of AUC risk, which does not require pairwise comparison but rank- ings of the predictions. We further show that the ranking operation can be avoided, and the learning objective obtained based on this sur- rogate enjoys linear complexity in time and storage. We perform experiments to demon- strate the effectiveness of the online and batch algorithms for AUC optimization based on the proposed surrogate loss.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-donner18a,
  title = 	 {Efficient {B}ayesian Inference for a {G}aussian Process Density Model},
  author =       {Donner, Christian and Opper, Manfred},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {52--61},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/donner18a/donner18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/donner18a.html},
  abstract = 	 {We reconsider a nonparametric density model based on Gaussian processes. By augmenting the model with latent P{ó}lya–Gamma random variables and a latent marked Poisson process we obtain a new likelihood which is conjugate to the model’s Gaussian process prior. The augmented posterior allows for efficient infer- ence by Gibbs sampling and an approximate variational mean field approach. For the latter we utilise sparse GP approximations to tackle the infinite dimensionality of the problem. The performance of both algorithms and compar- isons with other density estimators are demon- strated on artificial and real datasets with up to several thousand data points.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-sherstan18a,
  title = 	 {Comparing Direct and Indirect Temporal-Difference Methods for Estimating the Variance of the Return},
  author =       {Sherstan, Craig and Ashley, Dylan R. and Bennett, Brendan and Young, Kenny and White, Adam and White, Martha and Sutton, Richard S.},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {62--71},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/sherstan18a/sherstan18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/sherstan18a.html},
  abstract = 	 {Temporal-difference (TD) learning methods are widely used in reinforcement learning to estimate the expected return for each state, without a model, because of their significant advantages in computational and data effi- ciency. For many applications involving risk mitigation, it would also be useful to estimate the variance of the return by TD methods. In this paper, we describe a way of doing this that is substantially simpler than those proposed by Tamar, Di Castro, and Mannor in 2012, or those proposed by White and White in 2016. We show that two TD learners operating in series can learn expectation and variance esti- mates. The trick is to use the square of the TD error of the expectation learner as the reward of the variance learner, and the square of the ex- pectation learner’s discount rate as the discount rate of the variance learner. With these two modifications, the variance learning problem becomes a conventional TD learning problem to which standard theoretical results can be ap- plied. Our formal results are limited to the ta- ble lookup case, for which our method is still novel, but the extension to function approxi- mation is immediate, and we provide some em- pirical results for the linear function approx- imation case. Our experimental results show that our direct method behaves just as well as a comparable indirect method, but is generally more robust.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-turner18a,
  title = 	 {How well does your sampler really work?},
  author =       {Turner, Ryan and Neal, Brady},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {72--81},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/turner18a/turner18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/turner18a.html},
  abstract = 	 {We present a data-driven benchmark system to evaluate the performance of new MCMC samplers. Taking inspiration from the COCO benchmark in optimization, we view this benchmark as having critical importance to machine learning and statistics given the rate at which new samplers are proposed. The common hand-crafted examples to test new samplers are unsatisfactory; we take a meta- learning-like approach to generate realistic benchmark examples from a large corpus of data sets and models. Surrogates of posteriors found in real problems are created using highly flexible density models including modern neu- ral network models. We provide new insights into the real effective sample size of various samplers per unit time and the estimation effi- ciency of the samplers per sample. Addition- ally, we provide a meta-analysis to assess the predictive utility of various MCMC diagnos- tics and perform a nonparametric regression to combine them.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-wang18a,
  title = 	 {Learning Deep Hidden Nonlinear Dynamics from Aggregate Data},
  author =       {Wang, Yisen and Dai, Bo and Kong, Lingkai and Erfani, Sarah Monazam and Bailey, James and Zha, Hongyuan},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {82--91},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/wang18a/wang18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/wang18a.html},
  abstract = 	 {Learning nonlinear dynamics from diffusion data is a challenging problem since the individ- uals observed may be different at different time points, generally following an aggregate be- haviour. Existing work cannot handle the tasks well since they model such dynamics either di- rectly on observations or enforce the availabil- ity of complete longitudinal individual-level trajectories. However, in most of the practical applications, these requirements are unrealis- tic: the evolving dynamics may be too complex to be modeled directly on observations, and individual-level trajectories may not be avail- able due to technical limitations, experimental costs and/or privacy issues. To address these challenges, we formulate a model of diffusion dynamics as the hidden stochastic process via the introduction of hidden variables for flexi- bility, and learn the hidden dynamics directly on aggregate observations without any require- ment for individual-level trajectories. We pro- pose a dynamic generative model with Wasser- stein distance for LEarninG dEep hidden Non- linear Dynamics (LEGEND) and prove its the- oretical guarantees as well. Experiments on a range of synthetic and real-world datasets il- lustrate that LEGEND has very strong perfor- mance compared to state-of-the-art baselines.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-wang18b,
  title = 	 {Revisiting differentially private linear regression: optimal and adaptive prediction & estimation in unbounded domain},
  author =       {Wang, Yu-Xiang},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {92--102},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/wang18b/wang18b.pdf},
  url = 	 {https://proceedings.mlr.press/r16/wang18b.html},
  abstract = 	 {We revisit the problem of linear regression un- der a differential privacy constraint. By con- solidating existing pieces in the literature, we clarify the correct dependence of the feature, la- bel and coefficient domains in the optimization error and estimation error, hence revealing the delicate price of differential privacy in statis- tical estimation and statistical learning. More- over, we propose simple modifications of two existing DP algorithms: (a) posterior sampling, (b) sufficient statistics perturbation, and show that they can be upgraded into adaptive algo- rithms that are able to exploit data-dependent quantities and behave nearly optimally for every instance. Extensive experiments are conducted on both simulated data and real data, which conclude that both ADAOPS and ADASSP out- perform the existing techniques on nearly all 36 data sets that we test on.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-marchetti18a,
  title = 	 {Imaginary Kinematics},
  author =       {Marchetti, Sabina and Antonucci, Alessandro},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {103--112},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/marchetti18a/marchetti18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/marchetti18a.html},
  abstract = 	 {We introduce a novel class of adjustment rules for a collection of beliefs. This is an exten- sion of Lewis’ imaging to absorb probabilistic evidence in generalized settings. Unlike stan- dard tools for belief revision, our proposal may be used when information is inconsistent with an agent’s belief base. We show that the func- tionals we introduce are based on the imagi- nary counterpart of probability kinematics for standard belief revision, and prove that, under certain conditions, all standard postulates for belief revision are satisfied.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-rubenstein18a,
  title = 	 {From Deterministic ODEs to Dynamic Structural Causal Models},
  author =       {Rubenstein, Paul K. and Bongers, Stephan and Mooij, Joris M. and Schoelkopf, Bernhard},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {113--122},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/rubenstein18a/rubenstein18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/rubenstein18a.html},
  abstract = 	 {Structural Causal Models are widely used in causal modelling, but how they relate to other modelling tools is poorly understood. In this paper we provide a novel perspective on the re- lationship between Ordinary Differential Equa- tions and Structural Causal Models. We show how, under certain conditions, the asymptotic behaviour of an Ordinary Differential Equation under non-constant interventions can be mod- elled using Dynamic Structural Causal Models. In contrast to earlier work, we study not only the effect of interventions on equilibrium states; rather, we model asymptotic behaviour that is dynamic under interventions that vary in time, and include as a special case the study of static equilibria.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-zhao18a,
  title = 	 {{F}rank-{W}olfe Optimization for Symmetric-{NMF} under Simplicial Constraint},
  author =       {Zhao, Han and Gordon, Geoff},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {123--133},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/zhao18a/zhao18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/zhao18a.html},
  abstract = 	 {Symmetric nonnegative matrix factorization has found abundant applications in various do- mains by providing a symmetric low-rank de- composition of nonnegative matrices. In this paper we propose a Frank-Wolfe (FW) solver to optimize the symmetric nonnegative matrix factorization problem under a simplicial con- straint, which has recently been proposed for probabilistic clustering. Compared with exist- ing solutions, this algorithm is simple to imple- ment, and has no hyperparameters to be tuned. Building on the recent advances of FW algo- rithms in nonconvex optimization, we prove an O(1/$\varepsilon$2) convergence rate to $\varepsilon$-approximate KKT points, via a tight bound $\Theta$(n2) on the cur- vature constant, which matches the best known result in unconstrained nonconvex setting using gradient methods. Numerical results demon- strate the effectiveness of our algorithm. As a side contribution, we construct a simple nons- mooth convex problem where the FW algorithm fails to converge to the optimum. This result raises an interesting question about necessary conditions of the success of the FW algorithm on convex problems.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-adams18a,
  title = 	 {Learning Time Series Segmentation Models from Temporally Imprecise Labels},
  author =       {Adams, Roy and Marlin, Benjamin M.},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {134--143},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/adams18a/adams18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/adams18a.html},
  abstract = 	 {This paper considers the problem of learning time series segmentation models when the la- beled data are subject to temporal uncertainty or noise. Our approach augments the semi- Markov conditional random field (semi-CRF) model with a probabilistic model of the label observation process. This augmentation allows us to estimate the parameters of the semi-CRF from timestamps corresponding roughly to the occurrence of transitions between segments. We show how exact marginal inference can be performed in the augmented model in polyno- mial time, enabling learning based on marginal likelihood maximization. Our experiments on two activity detection problems show that the proposed approach can learn models from tem- porally imprecise labels, and can successfully refine imprecise segmentations through poste- rior inference. Finally, we show how inference complexity can be reduced by a factor of 40 using static and model-based pruning of the inference dynamic program.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-shilton18a,
  title = 	 {Multi-Target Optimisation via {B}ayesian Optimisation and Linear Programming},
  author =       {Shilton, Alistair and Rana, Santu and Gupta, Sunil and Venkatesh, Svetha},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {144--154},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/shilton18a/shilton18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/shilton18a.html},
  abstract = 	 {In Bayesian Multi-Objective optimisation, ex- pected hypervolume improvement is often used to measure the goodness of candidate solutions. However when there are many ob- jectives the calculation of expected hyper- volume improvement can become computa- tionally prohibitive. An alternative approach measures the goodness of a candidate based on the distance of that candidate from the Pareto front in objective space. In this paper we present a novel distance-based Bayesian Many-Objective optimisation algorithm. We demonstrate the efficacy of our algorithm on three problems, namely the DTLZ2 bench- mark problem, a hyper-parameter selection problem, and high-temperature creep-resistant alloy design.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-geng18a,
  title = 	 {Stochastic Learning for Sparse Discrete {M}arkov Random Fields with Controlled Gradient Approximation Error},
  author =       {Geng, Sinong and Kuang, Zhaobin and Liu, Jie and Wright, Stephen and Page, David},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {155--165},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/geng18a/geng18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/geng18a.html},
  abstract = 	 {We study the L1-regularized maximum likelihood estimator/estimation (MLE) problem for discrete Markov random fields (MRFs), where efficient and scalable learning requires both sparse regularization and approximate inference. To address these challenges, we consider a stochastic learning framework called stochastic proximal gradient (SPG; Honorio 2012a, Atchade et al. 2014, Miasojedow and Rejchel 2016). SPG is an inexact proximal gradient algorithm [Schmidt et al., 2011], whose inexactness stems from the stochastic oracle (Gibbs sampling) for gradient approximation – exact gradient evaluation is infeasible in general due to the NP-hard inference problem for discrete MRFs [Koller and Friedman, 2009]. Theoretically, we provide novel verifiable bounds to inspect and control the quality of gradient approximation. Empirically, we propose the tighten asymptotically (TAY) learning strategy based on the verifiable bounds to boost the performance of SPG.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-zheng18a,
  title = 	 {Active Information Acquisition for Linear Optimization},
  author =       {Zheng, Shuran and Waggoner, Bo and Liu, Yang and Chen, Yiling},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {166--175},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/zheng18a/zheng18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/zheng18a.html},
  abstract = 	 {We consider partially-specified optimization problems where the goal is to actively, but efficiently, acquire missing information about the problem in order to solve it. An algo- rithm designer wishes to solve a linear pro- gram (LP), max cT x s.t. Ax $\leq$b, x $\geq$0, but does not initially know some of the pa- rameters. The algorithm can iteratively choose an unknown parameter and gather information in the form of a noisy sample centered at the parameter’s (unknown) value. The goal is to find an approximately feasible and optimal so- lution to the underlying LP with high proba- bility while drawing a small number of sam- ples. We focus on two cases. (1) When the parameters b of the constraints are initially un- known, we propose an efficient algorithm com- bining techniques from the ellipsoid method for LP and confidence-bound approaches from bandit algorithms. The algorithm adaptively gathers information about constraints only as needed in order to make progress. We give sample complexity bounds for the algorithm and demonstrate its improvement over a naive approach via simulation. (2) When the param- eters c of the objective are initially unknown, we take an information-theoretic approach and give roughly matching upper and lower sam- ple complexity bounds, with an (inefficient) successive-elimination algorithm.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-kang18a,
  title = 	 {Transferable Meta Learning Across Domains},
  author =       {Kang, Bingyi and Feng, Jiashi},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {176--186},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/kang18a/kang18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/kang18a.html},
  abstract = 	 {Meta learning algorithms are effective at ob- taining meta models with the capability of solving new tasks quickly. However, they crit- ically require sufficient tasks for meta model training and the resulted model can only solve new tasks similar to the training ones. These limitations make them suffer performance de- cline in presence of insufficiency of training tasks in target domains and task heterogene- ity—the source (model training) tasks presents different characteristics from target (model ap- plication) tasks. To overcome these two signif- icant limitations of existing meta learning al- gorithms, we introduce the cross-domain meta learning framework and propose a new trans- ferable meta learning (TML) algorithm. TML performs meta task adaptation jointly with meta model learning, which effectively nar- rows divergence between source and target tasks and enables transferring source meta- knowledge to solve target tasks. Thus, the re- sulted transferable meta model can solve new learning tasks in new domains quickly. We ap- ply the proposed TML to cross-domain few- shot classification problems and evaluate its performance on multiple benchmarks. It per- forms significantly better and faster than well- established meta learning algorithms and fine- tuned domain-adapted models.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-cui18a,
  title = 	 {Learning the Causal Structure of Copula Models with Latent Variables},
  author =       {Cui, Ruifei and Groot, Perry and Schauer, Moritz and Heskes, Tom},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {187--196},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/cui18a/cui18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/cui18a.html},
  abstract = 	 {A common goal in psychometrics, sociology, and econometrics is to uncover causal rela- tions among latent variables representing hy- pothetical constructs that cannot be measured directly, such as attitude, intelligence, and motivation. Through measurement models, these constructs are typically linked to mea- surable indicators, e.g., responses to question- naire items. This paper addresses the prob- lem of causal structure learning among such la- tent variables and other observed variables. We propose the ‘Copula Factor PC’ algorithm as a novel two-step approach. It first draws samples of the underlying correlation matrix in a Gaus- sian copula factor model via a Gibbs sampler on rank-based data. These are then translated into an average correlation matrix and an ef- fective sample size, which are taken as input to the standard PC algorithm for causal discovery in the second step. We prove the consistency of our ‘Copula Factor PC’ algorithm, and demon- strate that it outperforms the PC-MIMBuild al- gorithm and a greedy step-wise approach. We illustrate our method on a real-world data set about children with Attention Deficit Hyperac- tivity Disorder.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-yuan18a,
  title = 	 {$f_{BGD}$: Learning Embeddings From Positive Unlabeled Data with {BGD}},
  author =       {YUAN, Fajie and Xin, Xin and He, Xiangnan and Guo, Guibing and Zhang, Weinan and Tat-Seng, CHUA and Jose, Joemon},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {197--206},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/yuan18a/yuan18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/yuan18a.html},
  abstract = 	 {Learning sparse features from only positive and unlabeled (PU) data is a fundamental task for problems of several domains, such as natural lan- guage processing (NLP), computer vision (CV), information retrieval (IR). Considering the nu- merous amount of unlabeled data, most prevalent methods rely on negative sampling (NS) to in- crease computational efficiency. However, sam- pling a fraction of unlabeled data as negative for training may ignore other important examples, and thus lead to non-optimal prediction perfor- mance. To address this, we present a fast and generic batch gradient descent optimizer (fBGD) to learn from all training examples without sam- pling. By leveraging sparsity in PU data, we ac- celerate fBGD by several magnitudes, making its time complexity the same level as the NS- based stochastic gradient descent method. Mean- while, we observe that the standard batch gradi- ent method suffers from gradient instability is- sues due to the sparsity property. Driven by a theoretical analysis for this potential cause, an in- tuitive solution arises naturally. To verify its effi- cacy, we perform experiments on multiple tasks with PU data across domains, and show that fBGD consistently outperforms NS-based mod- els on all tasks with comparable efficiency.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-derman18a,
  title = 	 {Soft-Robust Actor-Critic Policy-Gradient},
  author =       {Derman, Esther and Mankowitz, Daniel J and Mann, Timothy A and Mannor, Shie},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {207--217},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/derman18a/derman18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/derman18a.html},
  abstract = 	 {Robust Reinforcement Learning aims to derive an optimal behavior that accounts for model un- certainty in dynamical systems. However, pre- vious studies have shown that by considering the worst case scenario, robust policies can be overly conservative. Our soft-robust framework is an attempt to overcome this issue. In this paper, we present a novel Soft-Robust Actor- Critic algorithm (SR-AC). It learns an optimal policy with respect to a distribution over an uncertainty set and stays robust to model uncer- tainty but avoids the conservativeness of robust strategies. We show the convergence of SR-AC and test the efficiency of our approach on dif- ferent domains by comparing it against regular learning methods and their robust formulations.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-babichev18a,
  title = 	 {Constant Step Size Stochastic Gradient Descent for Probabilistic Modeling},
  author =       {Babichev, Dmitry and Bach, Francis},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {218--227},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/babichev18a/babichev18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/babichev18a.html},
  abstract = 	 {Stochastic gradient methods enable learning probabilistic models from large amounts of data. While large step-sizes (learning rates) have shown to be best for least-squares (e.g., Gaussian noise) once combined with param- eter averaging, these are not leading to con- vergent algorithms in general. In this pa- per, we consider generalized linear models, that is, conditional models based on exponen- tial families. We propose averaging moment parameters instead of natural parameters for constant-step-size stochastic gradient descent. For finite-dimensional models, we show that this can sometimes (and surprisingly) lead to better predictions than the best linear model. For infinite-dimensional models, we show that it always converges to optimal predictions, while averaging natural parameters never does. We illustrate our findings with simulations on synthetic data and classical benchmarks with many observations.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-gotovos18a,
  title = 	 {Discrete Sampling using Semigradient-based Product Mixtures},
  author =       {Gotovos, Alkis and Hassani, Hamed and Krause, Andreas and Jegelka, Stefanie},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {228--236},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/gotovos18a/gotovos18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/gotovos18a.html},
  abstract = 	 {We consider the problem of inference in dis- crete probabilistic models, that is, distributions over subsets of a finite ground set. These encompass a range of well-known models in machine learning, such as determinantal point processes and Ising models. Locally-moving Markov chain Monte Carlo algorithms, such as the Gibbs sampler, are commonly used for inference in such models, but their conver- gence is, at times, prohibitively slow. This is often caused by state-space bottlenecks that greatly hinder the movement of such samplers. We propose a novel sampling strategy that uses a specific mixture of product distributions to propose global moves and, thus, acceler- ate convergence. Furthermore, we show how to construct such a mixture using semigradi- ent information. We illustrate the effective- ness of combining our sampler with existing ones, both theoretically on an example model, as well as practically on three models learned from real-world data sets.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-aditya18a,
  title = 	 {Combining Knowledge and Reasoning through Probabilistic Soft Logic for Image Puzzle Solving},
  author =       {Aditya, Somak and Yang, Yezhou and Baral, Chitta and Aloimonos, Yiannis},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {237--247},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/aditya18a/aditya18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/aditya18a.html},
  abstract = 	 {The uncertainty associated with human per- ception is often reduced by one’s extensive prior experience and knowledge. Current datasets and systems do not emphasize the ne- cessity and benefit of using such knowledge. In this work, we propose the task of solving a genre of image-puzzles (“image riddles”) that require both capabilities involving visual de- tection (including object, activity recognition) and, knowledge-based or commonsense rea- soning. Each puzzle involves a set of images and the question “what word connects these images?”. We compile a dataset of over 3k riddles where each riddle consists of 4 im- ages and a groundtruth answer. The annota- tions are validated using crowd-sourced eval- uation. We also define an automatic evalua- tion metric to track future progress. Our task bears similarity with the commonly known IQ tasks such as analogy solving, sequence fill- ing that are often used to test intelligence. We develop a Probabilistic Reasoning-based ap- proach that utilizes commonsense knowledge about words and phrases to answer these rid- dles with a reasonable accuracy. Our approach achieves some promising results for these rid- dles and provides a strong baseline for future attempts. We make the entire dataset and re- lated materials publicly available to the com- munity (bit.ly/22f9Ala).},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-rainforth18a,
  title = 	 {Nesting Probabilistic Programs},
  author =       {Rainforth, Tom},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {248--257},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/rainforth18a/rainforth18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/rainforth18a.html},
  abstract = 	 {We formalize the notion of nesting probabilistic programming queries and investigate the result- ing statistical implications. We demonstrate that while query nesting allows the definition of models which could not otherwise be ex- pressed, such as those involving agents reason- ing about other agents, existing systems take approaches which lead to inconsistent estimates. We show how to correct this by delineating pos- sible ways one might want to nest queries and asserting the respective conditions required for convergence. We further introduce a new on- line nested Monte Carlo estimator that makes it substantially easier to ensure these conditions are met, thereby providing a simple framework for designing statistically correct inference en- gines. We prove the correctness of this online estimator and show that, when using the recom- mended setup, its asymptotic variance is always better than that of the equivalent fixed estimator, while its bias is always within a factor of two.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-tan18a,
  title = 	 {Scalable Algorithms for Learning High-Dimensional Linear Mixed Models},
  author =       {Tan, Zilong and Roche, Kimberly and Zhou, Xiang and Mukherjee, Sayan},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {258--267},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/tan18a/tan18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/tan18a.html},
  abstract = 	 {Linear mixed models (LMMs) are used exten- sively to model observations that are not in- dependent. Parameter estimation for LMMs can be computationally prohibitive on big data. State-of-the-art learning algorithms require computational complexity which depends at least linearly on the dimension p of the co- variates, and often use heuristics that do not offer theoretical guarantees. We present scal- able algorithms for learning high-dimensional LMMs with sublinear computational complex- ity dependence on p. Key to our approach are novel dual estimators which use only kernel functions of the data, and fast computational techniques based on the subsampled random- ized Hadamard transform. We provide theo- retical guarantees for our learning algorithms, demonstrating the robustness of parameter es- timation. Finally, we complement the theory with experiments on large synthetic and real data.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-forre18a,
  title = 	 {Constraint-based Causal Discovery for Non-Linear Structural Causal Models with Cycles and Latent Confounders},
  author =       {Forr{\'e}, Patrick and Mooij, Joris M.},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {268--277},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/forre18a/forre18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/forre18a.html},
  abstract = 	 {We address the problem of causal discovery from data, making use of the recently pro- posed causal modeling framework of modu- lar structural causal models (mSCM) to handle cycles, latent confounders and non-linearities. We introduce $\sigma$-connection graphs ($\sigma$-CG), a new class of mixed graphs (containing undi- rected, bidirected and directed edges) with ad- ditional structure, and extend the concept of $\sigma$-separation, the appropriate generalization of the well-known notion of d-separation in this setting, to apply to $\sigma$-CGs. We prove the closedness of $\sigma$-separation under marginalisa- tion and conditioning and exploit this to im- plement a test of $\sigma$-separation on a $\sigma$-CG. This then leads us to the first causal discovery algo- rithm that can handle non-linear functional re- lations, latent confounders, cyclic causal rela- tionships, and data from different (stochastic) perfect interventions. As a proof of concept, we show on synthetic data how well the algo- rithm recovers features of the causal graph of modular structural causal models.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-shpakova18a,
  title = 	 {Marginal Weighted Maximum Log-likelihood for Efficient Learning of Perturb-and-Map models},
  author =       {Shpakova, Tatiana and Bach, Francis and Osokin, Anton},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {278--288},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/shpakova18a/shpakova18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/shpakova18a.html},
  abstract = 	 {We consider the structured-output prediction problem through probabilistic approaches and generalize the “perturb-and-MAP” framework to more challenging weighted Hamming losses, which are crucial in applications. While in principle our approach is a straightforward marginalization, it requires solving many re- lated MAP inference problems. We show that for log-supermodular pairwise models these op- erations can be performed efficiently using the machinery of dynamic graph cuts. We also pro- pose to use double stochastic gradient descent, both on the data and on the perturbations, for efficient learning. Our framework can naturally take weak supervision (e.g., partial labels) into account. We conduct a set of experiments on medium-scale character recognition and image segmentation, showing the benefits of our algo- rithms.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-ding18a,
  title = 	 {Variational Inference for {G}aussian Processes with Panel Count Data},
  author =       {Ding, Hongyi and Lee, Young and Sato, Issei and Sugiyama, Masashi},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {289--298},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/ding18a/ding18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/ding18a.html},
  abstract = 	 {We present the first framework for Gaussian- process-modulated Poisson processes when the temporal data appear in the form of panel counts. Panel count data frequently arise when experimental subjects are observed only at dis- crete time points and only the numbers of oc- currences of the events between subsequent observation times are available. The exact occurrence timestamps of the events are un- known. The method of conducting the efficient variational inference is presented, based on the assumption of a Gaussian-process-modulated intensity function. We derive a tractable lower bound to alleviate the problems of the in- tractable evidence lower bound inherent in the variational inference framework. Our algo- rithm outperforms classical methods on both synthetic and three real panel count sets.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-monti18a,
  title = 	 {A unified probabilistic model for learning latent factors and their connectivities from high-dimensional data},
  author =       {Monti, Ricardo Pio and Hyvarinen, Aapo},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {299--308},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/monti18a/monti18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/monti18a.html},
  abstract = 	 {Connectivity estimation is challenging in the context of high-dimensional data. A useful preprocessing step is to group variables into clusters, however, it is not always clear how to do so from the perspective of connectiv- ity estimation. Another practical challenge is that we may have data from multiple related classes (e.g., multiple subjects or conditions) and wish to incorporate constraints on the simi- larities across classes. We propose a probabilis- tic model which simultaneously performs both a grouping of variables (i.e., detecting commu- nity structure) and estimation of connectivities between the groups which correspond to latent variables. The model is essentially a factor anal- ysis model where the factors are allowed to have arbitrary correlations, while the factor loading matrix is constrained to express a community structure. The model can be applied on multiple classes so that the connectivities can be differ- ent between the classes, while the community structure is the same for all classes. We pro- pose an efficient estimation algorithm based on score matching, and prove the identifiability of the model. Finally, we present an extension to directed (causal) connectivities over latent vari- ables. Simulations and experiments on fMRI data validate the practical utility of the method.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-fitzsimons18a,
  title = 	 {Improved Stochastic Trace Estimation using Mutually Unbiased Bases},
  author =       {Fitzsimons, JK and Osborne, MA and Roberts, SJ and Fitzsimons, JF},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {309--317},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/fitzsimons18a/fitzsimons18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/fitzsimons18a.html},
  abstract = 	 {The paper begins by introducing the definition and construction of mutually unbiased bases, which are a widely used concept in quantum information processing but have received lit- tle to no attention in the machine learning and statistics literature. We demonstrate their use- fulness by using them to create a new sampling technique which offers an improvement on the previously well established bounds of stochas- tic trace estimation. This approach offers a new state of the art single shot sampling vari- ance while requiring O(log(n)) random bits for x $\in$Rn which significantly improves on traditional methods such as fixed basis meth- ods, Hutchinson’s and Gaussian estimators in terms of the number of random bits required and worst case sample variance.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-huang18a,
  title = 	 {Unsupervised Multi-view Nonlinear Graph Embedding},
  author =       {Huang, Jiaming and Li, Zhao and Zheng, Vincent W. and Wen, Wen and Yang, Yifan and Chen, Yuanmi},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {318--327},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/huang18a/huang18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/huang18a.html},
  abstract = 	 {In this paper, we study the unsupervised multi-view graph embedding (UMGE) prob- lem, which aims to learn graph embedding from multiple perspectives in an unsupervised manner. However, the vast majority of multi- view learning work focuses on non-graph data, and surprisingly there are limited work on UMGE. By systematically analyzing different existing methods for UMGE, we discover that cross-view and nonlinearity play a vital role in efficiently improving graph embedding qual- ity. Motivated by this concept, we develop an unsupervised Multi-viEw nonlineaR Graph Embedding (MERGE) approach to model re- lational multi-view consistency. Experimen- tal results on five benchmark datasets demon- strate that MERGE significantly outperforms the state-of-the-art baselines in terms of accu- racy in node classification tasks without sacri- ficing the computational efficiency.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-pinot18a,
  title = 	 {Graph-based Clustering under Differential Privacy},
  author =       {Pinot, Rafael and Morvan, Anne and Yger, Florian and Gouy-Pailler, Cedric and Atif, Jamal},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {328--337},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/pinot18a/pinot18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/pinot18a.html},
  abstract = 	 {In this paper, we present the first differen- tially private clustering method for arbitrary- shaped node clusters in a graph. This algo- rithm takes as input only an approximate Min- imum Spanning Tree (MST) T released under weight differential privacy constraints from the graph. Then, the underlying nonconvex clus- tering partition is successfully recovered from cutting optimal cuts on T . As opposed to ex- isting methods, our algorithm is theoretically well-motivated. Experiments support our the- oretical findings.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-zhang18a,
  title = 	 {GaAN: Gated Attention Networks for Learning on Large and Spatiotemporal Graphs},
  author =       {Zhang, Jiani and Shi, Xingjian and Xie, Junyuan and Ma, Hao and King, Irwin and Yeung, Dit-yan},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {338--348},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/zhang18a/zhang18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/zhang18a.html},
  abstract = 	 {We propose a new network architecture, Gated Attention Networks (GaAN), for learning on graphs. Unlike the traditional multi-head at- tention mechanism, which equally consumes all attention heads, GaAN uses a convolutional sub-network to control each attention head’s importance. We demonstrate the effective- ness of GaAN on the inductive node classi- fication problem on large graphs. Moreover, with GaAN as a building block, we construct the Graph Gated Recurrent Unit (GGRU) to address the traffic speed forecasting prob- lem. Extensive experiments on three real- world datasets show that our GaAN framework achieves state-of-the-art results on both tasks.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-mogensen18a,
  title = 	 {Causal Learning for Partially Observed Stochastic Dynamical Systems},
  author =       {Mogensen, S{\o}ren Wengel and Malinsky, Daniel and Hansen, Niels Richard},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {349--359},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/mogensen18a/mogensen18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/mogensen18a.html},
  abstract = 	 {Many models of dynamical systems have causal interpretations that support reasoning about the consequences of interventions, suita- bly defined. Furthermore, local independence has been suggested as a useful independence concept for stochastic dynamical systems. There is, however, no well-developed theore- tical framework for causal learning based on this notion of independence. We study inde- pendence models induced by directed graphs (DGs) and provide abstract graphoid proper- ties that guarantee that an independence model has the global Markov property w.r.t. a DG. We apply these results to It\^{}o diffusions and event processes. For a partially observed sys- tem, directed mixed graphs (DMGs) represent the marginalized local independence model, and we develop, under a faithfulness assump- tion, a sound and complete learning algo- rithm of the directed mixed equivalence graph (DMEG) as a summary of all Markov equiva- lent DMGs.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-hegde18a,
  title = 	 {Variational zero-inflated {G}aussian processes with sparse kernels},
  author =       {Hegde, Pashupati and Heinonen, Markus and Kaski, Samuel},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {360--370},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/hegde18a/hegde18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/hegde18a.html},
  abstract = 	 {Zero-inflated datasets, which have an ex- cess of zero outputs, are commonly en- countered in problems such as climate or rare event modelling. Conventional ma- chine learning approaches tend to overesti- mate the non-zeros leading to poor perfor- mance. We propose a novel model family of zero-inflated Gaussian processes (ZiGP) for such zero-inflated datasets, produced by sparse kernels through learning a la- tent probit Gaussian process that can zero out kernel rows and columns whenever the signal is absent. The ZiGPs are particu- larly useful for making the powerful Gaus- sian process networks more interpretable. We introduce sparse GP networks where variable-order latent modelling is achieved through sparse mixing signals. We derive the non-trivial stochastic variational infer- ence tractably for scalable learning of the sparse kernels in both models. The novel output-sparse approach improves both pre- diction of zero-inflated data and inter- pretability of latent mixing models.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-garcia-duran18a,
  title = 	 {KBlrn: End-to-End Learning of Knowledge Base Representations with Latent, Relational, and Numerical Features},
  author =       {Garcia-Duran, Alberto and Niepert, Mathias},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {371--380},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/garcia-duran18a/garcia-duran18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/garcia-duran18a.html},
  abstract = 	 {We present KBLRN, a framework for end-to- end learning of knowledge base representa- tions from latent, relational, and numerical fea- tures. KBLRN integrates feature types with a novel combination of neural representation learning and probabilistic product of experts models. To the best of our knowledge, KBLRN is the first approach that learns representa- tions of knowledge bases by integrating la- tent, relational, and numerical features. We show that instances of KBLRN outperform ex- isting methods on a range of knowledge base completion tasks. We contribute a novel data set enriching commonly used knowledge base completion benchmarks with numerical fea- tures. The data sets are available under a per- missive BSD-3 license1. We also investigate the impact numerical features have on the KB completion performance of KBLRN.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-atzmon18a,
  title = 	 {Probabilistic {AND}-{OR} Attribute Grouping for Zero-Shot Learning},
  author =       {Atzmon, Yuval and Chechik, Gal},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {381--391},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/atzmon18a/atzmon18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/atzmon18a.html},
  abstract = 	 {In zero-shot learning (ZSL), a classifier is trained to recognize visual classes without any image samples. Instead, it is given seman- tic information about the class, like a textual description or a set of attributes. Learning from attributes could benefit from explicitly modeling structure of the attribute space. Un- fortunately, learning of general structure from empirical samples is hard with typical dataset sizes. Here we describe LAGO1, a probabilistic model designed to capture natural soft and- or relations across groups of attributes. We show how this model can be learned end-to- end with a deep attribute-detection model. The soft group structure can be learned from data jointly as part of the model, and can also read- ily incorporate prior knowledge about groups if available. The soft and-or structure suc- ceeds to capture meaningful and predictive structures, improving the accuracy of zero-shot learning on two of three benchmarks. Finally, LAGO reveals a unified formulation over two ZSL approaches: DAP (Lampert et al., 2009) and ESZSL (Romera-Paredes & Torr, 2015). Interestingly, taking only one sin- gleton group for each attribute, introduces a new soft-relaxation of DAP, that outperforms DAP by $\tilde$40%.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-berg18a,
  title = 	 {Sylvester Normalizing Flows for Variational Inference},
  author =       {Berg, Rianne van den and Hasenclever, Leonard and Tomczak, Jakub and Welling, Max},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {392--401},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/berg18a/berg18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/berg18a.html},
  abstract = 	 {Variational inference relies on flexible ap- proximate posterior distributions. Normaliz- ing flows provide a general recipe to con- struct flexible variational posteriors. We in- troduce Sylvester normalizing flows, which can be seen as a generalization of planar flows. Sylvester normalizing flows remove the well-known single-unit bottleneck from planar flows, making a single transformation much more flexible. We compare the performance of Sylvester normalizing flows against pla- nar flows and inverse autoregressive flows and demonstrate that they compare favorably on several datasets.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-ma18a,
  title = 	 {Holistic Representations for Memorization and Inference},
  author =       {Ma, Yunpu and Hildebrandt, Marcel and Tresp, Volker and Baier, Stephan},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {402--412},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/ma18a/ma18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/ma18a.html},
  abstract = 	 {In this paper we introduce a novel holographic memory model for the distributed storage of complex association patterns and apply it to knowledge graphs. In a knowledge graph, a la- belled link connects a subject node with an ob- ject node, jointly forming a subject-predicate- objects triple. In the presented work, nodes and links have initial random representations, plus holistic representations derived from the initial representations of nodes and links in their local neighbourhoods. A memory trace is represented in the same vector space as the holistic representations themselves. To reduce the interference between stored information, it is required that the initial random vectors should be pairwise quasi-orthogonal. We show that pairwise quasi-orthogonality can be im- proved by drawing vectors from heavy-tailed distributions, e.g., a Cauchy distribution, and, thus, memory capacity of holistic representa- tions can significantly be improved. Further- more, we show that, in combination with a simple neural network, the presented holistic representation approach is superior to other methods for link predictions on knowledge graphs.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-kyrillidis18a,
  title = 	 {Simple and practical algorithms for $\ell_p$-norm low-rank approximation},
  author =       {Kyrillidis, Anastasios},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {413--423},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/kyrillidis18a/kyrillidis18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/kyrillidis18a.html},
  abstract = 	 {We propose practical algorithms for entrywise ‘p-norm low-rank approximation, for p = 1 or p = 1. The proposed framework, which is non-convex and gradient-based, is easy to implement and typically attains better approx- imations, faster, than state of the art. From a theoretical standpoint, we show that the proposed scheme can attain (1 + ")- OPT approximations. Our algorithms are not hyperparameter-free: they achieve the desider- ata only assuming algorithm’s hyperparame- ters are known apriori—or are at least approx- imable. I.e., our theory indicates what problem quantities need to be known, in order to get a good solution within polynomial time, and does not contradict to recent inapproximabilty results, as in [46].},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-chaudhuri18a,
  title = 	 {Quantile-Regret Minimisation in Infinitely Many-Armed Bandits},
  author =       {Chaudhuri, Arghya Roy and Kalyanakrishnan, Shivaram},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {424--433},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/chaudhuri18a/chaudhuri18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/chaudhuri18a.html},
  abstract = 	 {The stochastic multi-armed bandit is a well- studied abstraction of decision making in the face of uncertainty. We consider the setting in which the number of bandit arms is much larger than the possible number of pulls, and can even be infinite. With the aim of minimising regret with respect to an optimal arm, existing methods for this set- ting either assume some structure over the set of arms (Kleinberg et al., 2008, Ray Chowdhury and Gopalan, 2017), or some property of the re- ward distribution (Wang et al., 2008). Invariably, the validity of such assumptions—and therefore the performance of the corresponding methods— depends on instance-specific parameters, which might not be known beforehand. We propose a conceptually simple, parameter-free, and practically effective alternative. Specifically we introduce a notion of regret with respect to the top quantile of a probability distribution over the expected reward of randomly drawn arms. Our main contribution is an algorithm that achieves sublinear “quantile-regret”, both (1) when it is specified a quantile, and (2) when the quantile can be any (unknown) positive value. The algorithm needs no side information about the arms or about the structure of their reward distributions: it re- lies on random sampling to reach arms in the top quantile. Experiments show that our algorithm outperforms several previous methods (in terms of conventional regret) when the latter are not tuned well, and often even when they are.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-kim18a,
  title = 	 {Variational Inference for {G}aussian Process Models for Survival Analysis},
  author =       {Kim, Minyoung and Pavlovic, Vladimir},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {434--444},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/kim18a/kim18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/kim18a.html},
  abstract = 	 {Gaussian process survival analysis model (GP- SAM) was recently proposed to address key deficiencies of the Cox proportional hazard model, namely the need to account for uncer- tainty in the hazard function modeling while, at the same time, relaxing the time-covariates factorized assumption of the Cox model. How- ever, the existing MCMC inference algorithms for GPSAM have proven to be slow in prac- tice. In this paper we propose novel and scal- able variational inference algorithms for GP- SAM that reduce the time complexity of the sampling approaches and improve scalability to large datasets. We accomplish this by em- ploying two effective strategies in scalable GP: i) using pseudo inputs and ii) approximation via random feature expansions. In both setups, we derive the full and partial likelihood formu- lations, typically considered in survival analy- sis settings. The proposed approaches are eval- uated on two clinical and a divorce-marriage benchmark datasets, where we demonstrate improvements in prediction accuracy over the existing survival analysis methods, while re- ducing the complexity of inference compared to the recent state-of-the-art MCMC-based al- gorithms.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-zhao18b,
  title = 	 {A Cost-Effective Framework for Preference Elicitation and Aggregation},
  author =       {Zhao, Zhibing and Li, Haoming and Wang, Junming and Kephart, Jeffrey O. and Mattei, Nicholas and Su, Hui and Xia, Lirong},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {445--455},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/zhao18b/zhao18b.pdf},
  url = 	 {https://proceedings.mlr.press/r16/zhao18b.html},
  abstract = 	 {We propose a cost-effective framework for preference elicitation and aggregation under the Plackett-Luce model with features. Given a budget, our framework iteratively computes the most cost-effective elicitation questions in order to help the agents make a better group decision. We illustrate the viability of the framework with experiments on Amazon Mechanical Turk, which we use to estimate the cost of answering different types of elicitation ques- tions. We compare the prediction accuracy of our framework when adopting various infor- mation criteria that evaluate the expected infor- mation gain from a question. Our experiments show carefully designed information criteria are much more efficient, i.e., they arrive at the correct answer using fewer queries, than ran- domly asking questions given the budget con- straint.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-denevi18a,
  title = 	 {Incremental Learning-to-Learn with Statistical Guarantees},
  author =       {Denevi, Giulia and Ciliberto, Carlo and Stamos, Dimitris and Pontil, Massimiliano},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {456--465},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/denevi18a/denevi18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/denevi18a.html},
  abstract = 	 {In learning-to-learn the goal is to infer a learning algorithm that works well on a class of tasks sampled from an unknown meta- distribution. In contrast to previous work on batch learning-to-learn, we consider a scenario where tasks are presented sequentially and the algorithm needs to adapt incrementally to im- prove its performance on future tasks. Key to this setting is for the algorithm to rapidly in- corporate new observations into the model as they arrive, without keeping them in memory. We focus on the case where the underlying al- gorithm is Ridge Regression parametrised by a symmetric positive semidefinite matrix. We propose to learn this matrix by applying a stochastic strategy to minimize the empirical error incurred by Ridge Regression on future tasks sampled from the meta-distribution. We study the statistical properties of the proposed algorithm and prove non-asymptotic bounds on its excess transfer risk, that is, the gener- alization performance on new tasks from the same meta-distribution. We compare our on- line learning-to-learn approach with a state-of- the-art batch method, both theoretically and empirically.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-degenne18a,
  title = 	 {Bandits with Side Observations: Bounded vs. Logarithmic Regret},
  author =       {Degenne, R{\'e}my and Garcelon, Evrard and Perchet, Vianney},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {466--475},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/degenne18a/degenne18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/degenne18a.html},
  abstract = 	 {We consider the classical stochastic multi- armed bandit but where, from time to time and roughly with frequency $\epsilon$, an extra observation is gathered by the agent for free. We prove that, no matter how small $\epsilon$ is the agent can ensure a regret uniformly bounded in time. More precisely, we construct an algorithm with a regret smaller than P i log(1/$\epsilon$) $\Delta$i , up to multi- plicative constant and log log terms. We also prove a matching lower-bound, stating that no reasonable algorithm can outperform this quantity.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-bloem-reddy18a,
  title = 	 {Sampling and Inference for Beta Neutral-to-the-Left Models of Sparse Networks},
  author =       {Bloem-Reddy, Benjamin and Foster, Adam and Mathieu, Emile and Teh, Yee Whye},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {476--485},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/bloem-reddy18a/bloem-reddy18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/bloem-reddy18a.html},
  abstract = 	 {Empirical evidence suggests that heavy-tailed degree distributions occurring in many real net- works are well-approximated by power laws with exponents $\eta$ that may take values either less than and greater than two. Models based on various forms of exchangeability are able to capture power laws with $\eta$ < 2, and admit tractable inference algorithms; we draw on pre- vious results to show that $\eta$ > 2 cannot be gen- erated by the forms of exchangeability used in existing random graph models. Preferential at- tachment models generate power law exponents greater than two, but have been of limited use as statistical models due to the inherent difficulty of performing inference in non-exchangeable models. Motivated by this gap, we design and implement inference algorithms for a recently proposed class of models that generates $\eta$ of all possible values. We show that although they are not exchangeable, these models have prob- abilistic structure amenable to inference. Our methods make a large class of previously in- tractable models useful for statistical inference.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-zhu18a,
  title = 	 {Clustered Fused Graphical Lasso},
  author =       {Zhu, Yizhi and Koyejo, Oluwasanmi},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {486--495},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/zhu18a/zhu18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/zhu18a.html},
  abstract = 	 {Estimating the dynamic connectivity struc- ture among a system of entities has gar- nered much attention in recent years. While usual methods are designed to take advantage of temporal consistency to overcome noise, they conflict with the detectability of anoma- lies. We propose Clustered Fused Graphi- cal Lasso (CFGL), a method using precom- puted clustering information to improve the signal detectability as compared to typical Fused Graphical Lasso methods. We evaluate our method in both simulated and real-world datasets and conclude that, in many cases, CFGL can significantly improve the sensitivity to signals without a significant negative effect on the temporal consistency.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-zheng18b,
  title = 	 {Unsupervised Learning of Latent Physical Properties Using Perception-Prediction Networks},
  author =       {Zheng, David and Luo, Vinson and Wu, Jiajun and Tenenbaum, Joshua},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {496--506},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/zheng18b/zheng18b.pdf},
  url = 	 {https://proceedings.mlr.press/r16/zheng18b.html},
  abstract = 	 {We propose a framework for the completely unsupervised learning of latent object prop- erties from their interactions: the perception- prediction network (PPN). Consisting of a per- ception module that extracts representations of latent object properties and a prediction module that uses those extracted properties to simulate system dynamics, the PPN can be trained in an end-to-end fashion purely from samples of object dynamics. The representations of latent object properties learned by PPNs not only are sufficient to accurately simulate the dynamics of systems comprised of previously unseen ob- jects, but also can be translated directly into human-interpretable properties (e.g. mass, co- efficient of restitution) in an entirely unsuper- vised manner. Crucially, PPNs also generalize to novel scenarios: their gradient-based training can be applied to many dynamical systems and their graph-based structure functions over sys- tems comprised of different numbers of objects. Our results demonstrate the efficacy of graph- based neural architectures in object-centric in- ference and prediction tasks, and our model has the potential to discover relevant object proper- ties in systems that are not yet well understood.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-zou18a,
  title = 	 {Subsampled Stochastic Variance-Reduced Gradient {L}angevin Dynamics},
  author =       {Zou, Difan and Xu, Pan and Gu, Quanquan},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {507--517},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/zou18a/zou18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/zou18a.html},
  abstract = 	 {Stochastic variance-reduced gradient Langevin dynamics (SVRG-LD) was recently proposed to improve the performance of stochastic gra- dient Langevin dynamics (SGLD) by reduc- ing the variance of the stochastic gradient. In this paper, we propose a variant of SVRG-LD, namely SVRG-LD+, which replaces the full gradient in each epoch with a subsampled one. We provide a nonasymptotic analysis of the convergence of SVRG-LD+ in 2-Wasserstein distance, and show that SVRG-LD+ enjoys a lower gradient complexity1 than SVRG-LD, when the sample size is large or the target ac- curacy requirement is moderate. Our analysis directly implies a sharper convergence rate for SVRG-LD, which improves the existing con- vergence rate by a factor of $\kappa$1/6n1/6, where $\kappa$ is the condition number of the log-density function and n is the sample size. Experiments on both synthetic and real-world datasets vali- date our theoretical results.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-junges18a,
  title = 	 {Finite-State Controllers of POMDPs using Parameter Synthesis},
  author =       {Junges, Sebastian and Jansen, Nils and Wimmer, Ralf and Quatmann, Tim and Winterer, Leonore and Katoen, Joost-Pieter and Becker, Bernd},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {518--528},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/junges18a/junges18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/junges18a.html},
  abstract = 	 {We study finite-state controllers (FSCs) for par- tially observable Markov decision processes (POMDPs) that are provably correct with re- spect to given specifications. The key in- sight is that computing (randomised) FSCs on POMDPs is equivalent to—and compu- tationally as hard as—synthesis for paramet- ric Markov chains (pMCs). This correspon- dence allows to use tools for synthesis in pMCs to compute correct-by-construction FSCs on POMDPs for a variety of specifications. Our experimental evaluation shows comparable per- formance to well-known POMDP solvers.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-shpitser18a,
  title = 	 {Identification of Personalized Effects Associated With Causal Pathways},
  author =       {Shpitser, Ilya and Sherman, Eli},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {529--538},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/shpitser18a/shpitser18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/shpitser18a.html},
  abstract = 	 {Unlike classical causal inference, where the goal is to estimate average causal effects within a population, in settings such as per- sonalized medicine, the goal is to map a unit’s characteristics to a treatment tailored to maxi- mize the expected outcome for that unit. Ob- taining high-quality mappings of this type is the goal of the dynamic treatment regime liter- ature. In healthcare settings, optimizing poli- cies with respect to a particular causal pathway is often of interest as well. In the context of average treatment effects, estimation of effects associated with causal pathways is considered in the mediation analysis literature. In this paper, we combine mediation analy- sis and dynamic treatment regime ideas and consider how unit characteristics may be used to tailor a treatment strategy that maximizes an effect along specified sets of causal path- ways. In particular, we define counterfactual responses to such policies, give a general iden- tification algorithm for these counterfactuals, and prove completeness of the algorithm for unrestricted policies. A corollary of our re- sults is that the identification algorithm for re- sponses to policies given in [16] is complete for arbitrary policies.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-karan18a,
  title = 	 {Fast Counting in Machine Learning Applications},
  author =       {Karan, Subhadeep and Eichhorn, Matthew and Hurlburt, Blake and Iraci, Grant and Zola, Jaroslaw},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {539--548},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/karan18a/karan18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/karan18a.html},
  abstract = 	 {We propose scalable methods to execute count- ing queries in machine learning applications. To achieve memory and computational effi- ciency, we abstract counting queries and their context such that the counts can be aggregated as a stream. We demonstrate performance and scalability of the resulting approach on random queries, and through extensive experimentation using Bayesian networks learning and associ- ation rule mining. Our methods significantly outperform commonly used ADtrees and hash tables, and are practical alternatives for process- ing large-scale data.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-dvijotham18a,
  title = 	 {A Dual Approach to Scalable Verification of Deep Networks},
  author =       {Dvijotham, Krishnamurthy and Stanforth, Robert and Gowal, Sven and Mann, Timothy and Kohli, Pushmeet},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {549--558},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/dvijotham18a/dvijotham18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/dvijotham18a.html},
  abstract = 	 {This paper addresses the problem of formally verifying desirable properties of neural net- works, i.e., obtaining provable guarantees that neural networks satisfy specifications relating their inputs and outputs (robustness to bounded norm adversarial perturbations, for example). Most previous work on this topic was lim- ited in its applicability by the size of the net- work, network architecture and the complexity of properties to be verified. In contrast, our framework applies to a general class of activa- tion functions and specifications on neural net- work inputs and outputs. We formulate verifi- cation as an optimization problem (seeking to find the largest violation of the specification) and solve a Lagrangian relaxation of the opti- mization problem to obtain an upper bound on the worst case violation of the specification be- ing verified. Our approach is anytime i.e. it can be stopped at any time and a valid bound on the maximum violation can be obtained. We de- velop specialized verification algorithms with provable tightness guarantees under special as- sumptions and demonstrate the practical sig- nificance of our general verification approach on a variety of verification tasks.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-smith18a,
  title = 	 {Understanding Measures of Uncertainty for Adversarial Example Detection},
  author =       {Smith, Lewis and Gal, Yarin},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {559--568},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/smith18a/smith18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/smith18a.html},
  abstract = 	 {Measuring uncertainty is a promising technique for detecting adversarial examples, crafted in- puts on which the model predicts an incorrect class with high confidence. There are various measures of uncertainty, including predictive entropy and mutual information, each capturing distinct types of uncertainty. We study these measures, and shed light on why mutual infor- mation seems to be effective at the task of adver- sarial example detection. We highlight failure modes for MC dropout, a widely used approach for estimating uncertainty in deep models. This leads to an improved understanding of the draw- backs of current methods, and a proposal to im- prove the quality of uncertainty estimates using probabilistic model ensembles. We give illustra- tive experiments using MNIST to demonstrate the intuition underlying the different measures of uncertainty, as well as experiments on a real- world Kaggle dogs vs cats classification dataset.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-blom18a,
  title = 	 {Causal Discovery in the Presence of Measurement Error},
  author =       {Blom, Tineke and Klimovskaia, Anna and Magliacane, Sara and Mooij, Joris M.},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {569--578},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/blom18a/blom18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/blom18a.html},
  abstract = 	 {Causal discovery algorithms infer causal re- lations from data based on several assump- tions, including notably the absence of mea- surement error. However, this assumption is most likely violated in practical applications, which may result in erroneous, irreproducible results. In this work we show how to obtain an upper bound for the variance of random mea- surement error from the covariance matrix of measured variables and how to use this up- per bound as a correction for constraint-based causal discovery. We demonstrate a practical application of our approach on both simulated data and real-world protein signaling data.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-wang18c,
  title = 	 {{IDK} Cascades: Fast Deep Learning by Learning not to Overthink},
  author =       {Wang, Xin and Luo, Yujia and Crankshaw, Daniel and Tumanov, Alexey and Yu, Fisher and Gonzalez, Joseph E.},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {579--589},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/wang18c/wang18c.pdf},
  url = 	 {https://proceedings.mlr.press/r16/wang18c.html},
  abstract = 	 {Advances in deep learning have led to substan- tial increases in prediction accuracy but have been accompanied by increases in the cost of rendering predictions. We conjecture that for a majority of real-world inputs, the recent ad- vances in deep learning have created models that effectively “over-think” on simple inputs. In this paper we revisit the classic question of building model cascades that primarily leverage class asymmetry to reduce cost. We introduce the “I Don’t Know” (IDK) prediction cascades framework, a general framework to systemat- ically compose a set of pre-trained models to accelerate inference without a loss in predic- tion accuracy. We propose two search based methods for constructing cascades as well as a new cost-aware objective within this frame- work. The proposed IDK cascade framework can be easily adopted in the existing model serving systems without additional model re- training. We evaluate the proposed techniques on a range of benchmarks to demonstrate the effectiveness of the proposed framework.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-nair18a,
  title = 	 {Learning Fast Optimizers for Contextual Stochastic Integer Programs},
  author =       {Nair, Vinod and Dvijotham, Dj and Dunning, Iain and Vinyals, Oriol},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {590--599},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/nair18a/nair18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/nair18a.html},
  abstract = 	 {We present a novel reinforcement learning (RL) approach to learning a fast and highly scalable solver for a two-stage stochastic integer pro- gram in the large-scale data setting. Mixed inte- ger programming solvers do not scale to large datasets for this problem class. Additionally, they solve each instance independently, without any knowledge transfer across instances. We address these limitations with a learnable local search solver that jointly learns two policies, one to generate an initial solution and another to iteratively improve it with local moves. The policies use contextual features for a problem instance as input, which enables learning across instances and generalization to new ones. We also propose learning a policy to compute a bound on the objective using dual decompo- sition. Benchmark results show that on test instances our approach rapidly achieves approx- imately 30% to 2000% better objective value, which a state of the art integer programming solver (SCIP) requires more than an order of magnitude more running time to match. Our approach also achieves better solution quality on seven out of eight benchmark problems than standard baselines such as Tabu Search and Pro- gressive Hedging.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-ren18a,
  title = 	 {Differential Analysis of Directed Networks},
  author =       {Ren, Min and Zhang, Dabao},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {600--609},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/ren18a/ren18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/ren18a.html},
  abstract = 	 {We developed a novel statistical method to identify structural differences between net- works characterized by structural equation models. We propose to reparameterize the model to separate the differential structures from common structures, and then design an algorithm with calibration and construction stages to identify these differential structures. The calibration stage serves to obtain con- sistent prediction by building the $\ell$2 regular- ized regression of each endogenous variables against pre-screened exogenous variables, cor- recting for potential endogeneity issue. The construction stage consistently selects and es- timates both common and differential effects by undertaking $\ell$1 regularized regression of each endogenous variable against the predicts of other endogenous variables as well as its an- choring exogenous variables. Our method al- lows easy parallel computation at each stage. Theoretical results are obtained to establish non-asymptotic error bounds of predictions and estimates at both stages, as well as the con- sistency of identified common and differential effects. Our studies on synthetic data demon- strated that our proposed method performed much better than independently constructing the networks. A real data set is analyzed to illustrate the applicability of our method.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-bixler18a,
  title = 	 {Sparse-Matrix Belief Propagation},
  author =       {Bixler, Reid and Huang, Bert},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {610--619},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/bixler18a/bixler18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/bixler18a.html},
  abstract = 	 {We propose sparse-matrix belief propagation, which executes loopy belief propagation in pair- wise Markov random fields by replacing in- dexing over graph neighborhoods with sparse- matrix operations. This abstraction allows for seamless integration with optimized sparse lin- ear algebra libraries, including those that per- form matrix and tensor operations on modern hardware such as graphical processing units (GPUs). The sparse-matrix abstraction allows the implementation of belief propagation in a high-level language (e.g., Python) that is also able to leverage the power of GPU paralleliza- tion. We demonstrate sparse-matrix belief prop- agation by implementing it in a modern deep learning framework (PyTorch), measuring the resulting massive improvement in running time, and facilitating future integration into deep learning models.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-meisami18a,
  title = 	 {Sequential Learning under Probabilistic Constraints},
  author =       {Meisami, Amirhossein and Lam, Henry and Dong, Chen and Pani, Abhishek},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {620--630},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/meisami18a/meisami18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/meisami18a.html},
  abstract = 	 {We provide the first study on online learn- ing problems under stochastic constraints that are “soft”, i.e., need to be satisfied with high probability. These constraints are imposed on all or some stages of the time horizon so that the stage decisions probabilistically satisfy some given safety conditions. The distribu- tions that govern these conditions are learned through the collected observations. Under a Bayesian framework, we introduce a scheme that provides statistical feasibility guarantees through the time horizon, by using posterior Monte Carlo samples to form sampled con- straints which leverage the scenario generation approach in chance-constrained programming. We demonstrate how our scheme can be inte- grated into Thompson sampling and illustrate it with an application in online advertisement.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-broka18a,
  title = 	 {Abstraction Sampling in Graphical Models},
  author =       {Broka, Filjor and Dechter, Rina and Ihler, Alexander and Kask, Kalev},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {631--640},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/broka18a/broka18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/broka18a.html},
  abstract = 	 {We present a new sampling scheme for approx- imating hard to compute queries over graphical models, such as computing the partition func- tion. The scheme builds upon exact algorithms that traverse a weighted directed state-space graph representing a global function over a graphical model (e.g., probability distribution). With the aid of an abstraction function and ran- domization, the state space can be compacted (or trimmed) to facilitate tractable computa- tion, yielding a Monte Carlo Estimate that is unbiased. We present the general scheme and analyze its properties analytically and empiri- cally, investigating two specific ideas for pick- ing abstractions - targeting reduction of vari- ance or search space size.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-saemundsson18a,
  title = 	 {Meta Reinforcement Learning with Latent Variable {G}aussian Processes},
  author =       {Saemundsson, Steindor and Hofmann, Katja and Deisenroth, Marc Peter},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {641--651},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/saemundsson18a/saemundsson18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/saemundsson18a.html},
  abstract = 	 {Learning from small data sets is critical in many practical applications where data col- lection is time consuming or expensive, e.g., robotics, animal experiments or drug design. Meta learning is one way to increase the data efficiency of learning algorithms by general- izing learned concepts from a set of training tasks to unseen, but related, tasks. Often, this relationship between tasks is hard coded or re- lies in some other way on human expertise. In this paper, we frame meta learning as a hi- erarchical latent variable model and infer the relationship between tasks automatically from data. We apply our framework in a model- based reinforcement learning setting and show that our meta-learning model effectively gen- eralizes to novel tasks by identifying how new tasks relate to prior ones from minimal data. This results in up to a 60% reduction in the average interaction time needed to solve tasks compared to strong baselines.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-zhang18b,
  title = 	 {Non-Parametric Path Analysis in Structural Causal Models},
  author =       {Zhang, Junzhe and Bareinboim, Elias},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {652--661},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/zhang18b/zhang18b.pdf},
  url = 	 {https://proceedings.mlr.press/r16/zhang18b.html},
  abstract = 	 {One of the fundamental tasks in causal infer- ence is to decompose the observed association between a decision X and an outcome Y into its most basic structural mechanisms. In this paper, we introduce counterfactual measures for effects along with a specific mechanism, represented as a path from X to Y in an ar- bitrary structural causal model. We derive a novel non-parametric decomposition formula that expresses the covariance of X and Y as a sum over unblocked paths from X to Y con- tained in an arbitrary causal model. This for- mula allows a fine-grained path analysis with- out requiring a commitment to any particular parametric form, and can be seen as a gen- eralization of Wright’s decomposition method in linear systems (1923,1932) and Pearl’s non- parametric mediation formula (2001).},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-lacey18a,
  title = 	 {Stochastic Layer-Wise Precision in Deep Neural Networks},
  author =       {Lacey, Griffin and Taylor, Graham W. and Areibi, Shawki},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {662--671},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/lacey18a/lacey18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/lacey18a.html},
  abstract = 	 {Low precision weights, activations, and gradi- ents have been proposed as a way to improve the computational efficiency and memory foot- print of deep neural networks. Recently, low precision networks have even shown to be more robust to adversarial attacks. How- ever, typical implementations of low precision DNNs use uniform precision across all lay- ers of the network. In this work, we explore whether a heterogeneous allocation of preci- sion across a network leads to improved per- formance, and introduce a learning scheme where a DNN stochastically explores multi- ple precision configurations through learning. This permits a network to learn an optimal pre- cision configuration. We show on convolu- tional neural networks trained on MNIST and ILSVRC12 that even though these nets learn a uniform or near-uniform allocation strat- egy respectively, stochastic precision leads to a favourable regularization effect improving generalization.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-nabi18a,
  title = 	 {Estimation of Personalized Effects Associated With Causal Pathways},
  author =       {Nabi, Razieh and Kanki, Phyllis and Shpitser, Ilya},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {672--681},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/nabi18a/nabi18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/nabi18a.html},
  abstract = 	 {The goal of personalized decision making is to map a unit’s characteristics to an action tai- lored to maximize the expected outcome for that unit. Obtaining high-quality mappings of this type is the goal of the dynamic regime lit- erature. In healthcare settings, optimizing poli- cies with respect to a particular causal pathway may be of interest as well. For example, we may wish to maximize the chemical effect of a drug given data from an observational study where the chemical effect of the drug on the outcome is entangled with the indirect effect mediated by differential adherence. In such cases, we may wish to optimize the direct ef- fect of a drug, while keeping the indirect effect to that of some reference treatment. [15] shows how to combine mediation analysis and dy- namic treatment regime ideas to defines poli- cies associated with causal pathways and coun- terfactual responses to these policies. In this paper, we derive a variety of methods for learn- ing high quality policies of this type from data, in a causal model corresponding to a longitu- dinal setting of practical importance. We illus- trate our methods via a dataset of HIV patients undergoing therapy, gathered in the Nigerian PEPFAR program.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-sajed18a,
  title = 	 {High-confidence error estimates for learned value functions},
  author =       {Sajed, Touqir and Chung, Wesley and White, Martha},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {682--691},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/sajed18a/sajed18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/sajed18a.html},
  abstract = 	 {Estimating the value function for a fixed pol- icy is a fundamental problem in reinforcement learning. Policy evaluation algorithms—to es- timate value functions—continue to be devel- oped, to improve convergence rates, improve stability and handle variability, particularly for off-policy learning. To understand the prop- erties of these algorithms, the experimenter needs high-confidence estimates of the accu- racy of the learned value functions. For en- vironments with small, finite state-spaces, like chains, the true value function can be easily computed, to compute accuracy. For large, or continuous state-spaces, however, this is no longer feasible. In this paper, we address the largely open problem of how to obtain these high-confidence estimates, for general state- spaces. We provide a high-confidence bound on an empirical estimate of the value error to the true value error. We use this bound to design an offline sampling algorithm, which stores the required quantities to repeatedly compute value error estimates for any learned value function. We provide experiments in- vestigating the number of samples required by this offline algorithm in simple benchmark re- inforcement learning domains, and highlight that there are still many open questions to be solved for this important problem.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-fiez18a,
  title = 	 {Combinatorial Bandits for Incentivizing Agents with Dynamic Preferences},
  author =       {Fiez, Tanner and Sekar, Shreyas and Zheng, Liyuan and Ratliff, Lillian},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {692--702},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/fiez18a/fiez18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/fiez18a.html},
  abstract = 	 {The design of personalized incentives or rec- ommendations to improve user engagement is gaining prominence as digital platform providers continually emerge. We propose a multi-armed bandit framework for match- ing incentives to users, whose preferences are unknown a priori and evolving dynamically in time, in a resource constrained environ- ment. We design an algorithm that com- bines ideas from three distinct domains: (i) a greedy matching paradigm, (ii) the upper confidence bound algorithm (UCB) for ban- dits, and (iii) mixing times from the theory of Markov chains. For this algorithm, we provide theoretical bounds on the regret and demon- strate its performance via both synthetic and realistic (matching supply and demand in a bike-sharing platform) examples.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-garg18a,
  title = 	 {Sparse Multi-Prototype Classification},
  author =       {Garg, Vikas K. and Xiao, Lin and Dekel, Ofer},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {703--713},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/garg18a/garg18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/garg18a.html},
  abstract = 	 {We introduce a new class of sparse multi- prototype classifiers, designed to combine the computational advantages of sparse predictors with the non-linear power of prototype-based classification techniques. This combination makes sparse multi- prototype models especially well-suited for resource constrained computational plat- forms, such as the IoT devices. We cast our supervised learning problem as a convex- concave saddle point problem and design a provably-fast algorithm to solve it. We complement our theoretical analysis with an empirical study that demonstrates the merits of our methodology.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-piatkowski18a,
  title = 	 {Fast Stochastic Quadrature for Approximate Maximum-Likelihood Estimation},
  author =       {Piatkowski, Nico and Morik, Katharina},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {714--723},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/piatkowski18a/piatkowski18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/piatkowski18a.html},
  abstract = 	 {Recent stochastic quadrature techniques for undirected graphical models rely on near- minimax degree-k polynomial approximations to the model’s potential function for inferring the partition function. While providing de- sirable statistical guarantees, typical construc- tions of such approximations are themselves not amenable to efficient inference. Here, we develop a class of Monte Carlo sampling algo- rithms for efficiently approximating the value of the partition function, as well as the asso- ciated pseudo-marginals. More precisely, for pairwise models with n vertices and m edges, the complexity can be reduced from O(dk) to O(k4 + kn + m), where d $\geq$4m is the parameter dimension. We also consider the uses of stochastic quadrature for the problem of maximum-likelihood (ML) parameter esti- mation. For completely observed data, our analysis gives rise to a probabilistic bound on the log-likelihood of the model. Maxi- mizing this bound yields an approximate ML estimate which, in analogy to the moment- matching of exact ML estimation, can be inter- preted in terms of pseudo-moment-matching. We present experimental results illustrating the behavior of this approximate ML estimator.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-lou18a,
  title = 	 {Finite-sample Bounds for Marginal {MAP}},
  author =       {Lou, Qi and Dechter, Rina and Ihler, Alexander},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {724--733},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/lou18a/lou18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/lou18a.html},
  abstract = 	 {Marginal MAP is a key task in Bayesian in- ference and decision-making, and known to be very challenging in general. In this paper, we present an algorithm that blends heuristic search and importance sampling to provide any- time finite-sample bounds for marginal MAP along with predicted MAP solutions. We con- vert bounding marginal MAP to a surrogate task of bounding a series of summation prob- lems of an augmented graphical model, and then adapt dynamic importance sampling [Lou et al., 2017b], a recent advance in bounding the partition function, to provide finite-sample bounds for the surrogate task. Those bounds are guaranteed to be tight given enough time, and the values of the predicted MAP solutions will converge to the optimum. Our algorithm runs in an anytime/anyspace manner, which gives flexible trade-offs between memory, time, and solution quality. We demonstrate the effective- ness of our approach empirically on multiple challenging benchmarks in comparison with some state-of-the-art search algorithms.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-shpitser18b,
  title = 	 {Acyclic Linear SEMs Obey the Nested {M}arkov Property},
  author =       {Shpitser, Ilya and Evans, Robin and Richardson, Thomas S.},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {734--744},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/shpitser18b/shpitser18b.pdf},
  url = 	 {https://proceedings.mlr.press/r16/shpitser18b.html},
  abstract = 	 {The conditional independence structure in- duced on the observed marginal distribution by a hidden variable directed acyclic graph (DAG) may be represented by a graphical model rep- resented by mixed graphs called maximal an- cestral graphs (MAGs). This model has a num- ber of desirable properties, in particular the set of Gaussian distributions can be parameterized by viewing the graph as a path diagram. Mod- els represented by MAGs have been used for causal discovery [22], and identification theory for causal effects [28]. In addition to ordinary conditional indepen- dence constraints, hidden variable DAGs also induce generalized independence constraints. These constraints form the nested Markov property [20]. We first show that acyclic linear SEMs obey this property. Further we show that a natural parameterization for all Gaussian dis- tributions obeying the nested Markov property arises from a generalization of maximal ances- tral graphs that we call maximal arid graphs (MArG). We show that every nested Markov model can be associated with a MArG; viewed as a path diagram this MArG parametrizes the Gaussian nested Markov model. This leads di- rectly to methods for ML fitting and computing BIC scores for Gaussian nested models.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-chen18a,
  title = 	 {A Unified Particle-Optimization Framework for Scalable {B}ayesian Sampling},
  author =       {Chen, Changyou and Zhang, Ruiyi and Wang, Wenlin and Li, Bai and Chen, Liqun},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {745--754},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/chen18a/chen18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/chen18a.html},
  abstract = 	 {There has been recent interest in developing scalable Bayesian sampling methods such as stochastic gradient MCMC (SG-MCMC) and Stein variational gradient descent (SVGD) for big-data analysis. A standard SG-MCMC algo- rithm simulates samples from a discrete-time Markov chain to approximate a target distribu- tion, thus samples could be highly correlated, an undesired property for SG-MCMC. In con- trary, SVGD directly optimizes a set of particles to approximate a target distribution, and thus is able to obtain good approximations with rela- tively much fewer samples. In this paper, we propose a principle particle-optimization frame- work based on Wasserstein gradient flows to unify SG-MCMC and SVGD, and to allow new algorithms to be developed. Our framework interprets SG-MCMC as particle optimization on the space of probability measures, revealing a strong connection between SG-MCMC and SVGD. The key component of our framework is several particle-approximate techniques to efficiently solve the original partial differential equations on the space of probability measures. Extensive experiments on both synthetic data and deep neural networks demonstrate the ef- fectiveness and efficiency of our framework for scalable Bayesian sampling.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-moore18a,
  title = 	 {An Efficient Quantile Spatial Scan Statistic for Finding Unusual Regions in Continuous Spatial Data with Covariates},
  author =       {Moore, Travis and Wong, Weng-Keen},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {755--764},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/moore18a/moore18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/moore18a.html},
  abstract = 	 {Domains such as citizen science biodiversity monitoring and real estate sales are produc- ing spatial data with a continuous response and a vector of covariates associated with each spatial data point. A common data analy- sis task involves finding unusual regions that differ from the surrounding area. Existing techniques compare regions according to the means of their distributions to measure unusu- alness. Comparing means is not only vulner- able to outliers, but it is also restrictive as an analyst may want to compare other parts of the probability distributions. For instance, an analyst interested in unusual areas for high- end homes would be more interested in the 90th percentile of home sale prices than in the mean. We introduce the Quantile Spatial Scan Statistic (QSSS), which finds unusual regions in spatial data by comparing quantiles of data distributions while accounting for covariates at each data point. We also develop an exact in- cremental update of the hypothesis test used by the QSSS, which results in a massive speedup over a naive implementation.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-zhou18a,
  title = 	 {Stable Gradient Descent},
  author =       {Zhou, Yingxue and Chen, Sheng and Banerjee, Arindam},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {765--774},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/zhou18a/zhou18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/zhou18a.html},
  abstract = 	 {The goal of many machine learning tasks is to learn a model that has small population risk. While mini-batch stochastic gradient descent (SGD) and variants are popular approaches for achieving this goal, it is hard to pre- scribe a clear stopping criterion and to establish high probability convergence bounds to the population risk. In this paper, we introduce Stable Gradient Descent which validates stochastic gra- dient computations by splitting data into training and validation sets and reuses samples using a differential pri- vate mechanism. StGD comes with a natural upper bound on the number of iterations and has high-probability convergence to the population risk. Ex- perimental results illustrate that StGD is empirically competitive and often better than SGD and GD.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-callaway18a,
  title = 	 {Learning to select computations},
  author =       {Callaway, Frederick and Gul, Sayan and Krueger, Paul M. and Griffiths, Thomas L. and Lieder, Falk},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {775--784},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/callaway18a/callaway18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/callaway18a.html},
  abstract = 	 {The efficient use of limited computational re- sources is an essential ingredient of intel- ligence. Selecting computations optimally according to rational metareasoning would achieve this, but this is computationally in- tractable. Inspired by psychology and neu- roscience, we propose the first concrete and domain-general learning algorithm for approx- imating the optimal selection of computations: Bayesian metalevel policy search (BMPS). We derive this general, sample-efficient search al- gorithm for a computation-selecting metalevel policy based on the insight that the value of information lies between the myopic value of information and the value of perfect in- formation. We evaluate BMPS on three in- creasingly difficult metareasoning problems: when to terminate computation, how to allo- cate computation between competing options, and planning. Across all three domains, BMPS achieved near-optimal performance and com- pared favorably to previously proposed metar- easoning heuristics. Finally, we demonstrate the practical utility of BMPS in an emergency management scenario, even accounting for the overhead of metareasoning.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-de-asis18a,
  title = 	 {Per-decision Multi-step Temporal Difference Learning with Control Variates},
  author =       {De Asis, Kristopher and Sutton, Richard S.},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {785--793},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/de-asis18a/de-asis18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/de-asis18a.html},
  abstract = 	 {Multi-step temporal difference (TD) learning is an important approach in reinforcement learning, as it unifies one-step TD learning with Monte Carlo methods in a way where intermediate algorithms can outperform ei- ther extreme. They address a bias-variance trade off between reliance on current estimates, which could be poor, and incorporating longer sampled reward sequences into the updates. Especially in the off-policy setting, where the agent aims to learn about a policy different from the one generating its behaviour, the vari- ance in the updates can cause learning to di- verge as the number of sampled rewards used in the estimates increases. In this paper, we in- troduce per-decision control variates for multi- step TD algorithms, and compare them to ex- isting methods. Our results show that includ- ing the control variates can greatly improve performance on both on and off-policy multi- step temporal difference learning tasks.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-tan18b,
  title = 	 {The Indian Buffet Hawkes Process to Model Evolving Latent Influences},
  author =       {Tan, Xi and Rao, Vinayak and Neville, Jennifer},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {794--803},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/tan18b/tan18b.pdf},
  url = 	 {https://proceedings.mlr.press/r16/tan18b.html},
  abstract = 	 {Temporal events in the real world often exhibit reinforcing dynamics, where earlier events trigger follow-up activity in the near future. A canonical example of modeling such dy- namics is the Hawkes process (HP). However, previous HP models do not capture the rich dynamics of real-world activity—which can be driven by multiple latent triggering factors shared by past and future events, with the la- tent features themselves exhibiting temporal dependency structures. For instance, rather than view a new document just as a response to other documents in the recent past, it is impor- tant to account for the factor-structure under- lying all previous documents. This structure itself is not fixed, with the influence of earlier documents decaying with time. To this end, we propose a novel Bayesian nonparametric stochastic point process model, the Indian Buf- fet Hawkes Processes (IBHP), to learn multiple latent triggering factors underlying streaming document/message data. The IBP facilitates the inclusion of multiple triggering factors in the HP, and the HP allows for modeling latent factor evolution in the IBP. We develop a learn- ing algorithm for the IBHP based on Sequen- tial Monte Carlo and demonstrate the effective- ness of the model. In both synthetic and real data experiments, our model achieves equiv- alent or higher likelihood and provides inter- pretable topics and shows their dynamics.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-saha18a,
  title = 	 {Battle of Bandits},
  author =       {Saha, Aadirupa and Gopalan, Aditya},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {804--813},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/saha18a/saha18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/saha18a.html},
  abstract = 	 {We introduce Battling-Bandits – an online learning framework where given a set of n arms, the learner needs to select a subset of k $\geq$2 arms in each round and subsequently observes a stochastic feedback indicating the winner of the round. This framework generalizes the stan- dard Dueling-Bandit framework which applies to several practical scenarios such as medical treatment preferences, recommender systems, search engine optimization etc., where it is eas- ier and more effective to collect feedback for multiple options simultaneously. We develop a novel class of pairwise-subset choice model, for modelling the subset-wise winner feedback and propose three algorithms - Battling-Doubler, Battling-MultiSBM and Battling-Duel: While the first two are designed for a special class of linear-link based choice models, the third one applies to a much general class of pairwise- subset choice models with Condorcet winner. We also analyzed their regret guarantees and show the optimality of Battling-Duel proving a matching regret lower bound of $\Omega$(n log T), which (perhaps surprisingly) shows that the flexibility of playing size-k subsets does not really help to gather information faster than the corresponding dueling case (k = 2), at least for the current subsetwise feedback choice model. The efficacy of our algorithms are demonstrated through extensive experimental evaluations on a variety of synthetic and real world datasets.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-le-priol18a,
  title = 	 {Adaptive Stochastic Dual Coordinate Ascent for Conditional Random Fields},
  author =       {Le Priol, R{\'e}mi and Pich{\'e}, Alexandre and Lacoste-Julien, Simon},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {814--823},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/le-priol18a/le-priol18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/le-priol18a.html},
  abstract = 	 {This work investigates the training of condi- tional random fields (CRFs) via the stochas- tic dual coordinate ascent (SDCA) algorithm of Shalev-Shwartz and Zhang (2016). SDCA enjoys a linear convergence rate and a strong empirical performance for binary classification problems. However, it has never been used to train CRFs. Yet it benefits from an “exact” line search with a single marginalization oracle call, unlike previous approaches. In this paper, we adapt SDCA to train CRFs, and we enhance it with an adaptive non-uniform sampling strategy based on block duality gaps. We perform ex- periments on four standard sequence prediction tasks. SDCA demonstrates performances on par with the state of the art, and improves over it on three of the four datasets, which have in common the use of sparse features.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-sabharwal18a,
  title = 	 {Adaptive Stratified Sampling for Precision-Recall Estimation},
  author =       {Sabharwal, Ashish and Xue, Yexiang},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {824--833},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/sabharwal18a/sabharwal18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/sabharwal18a.html},
  abstract = 	 {We propose a new algorithm for computing a constant-factor approximation of precision- recall (PR) curves for massive noisy datasets produced by generative models. Assessing va- lidity of items in such datasets requires human annotation, which is costly and must be mini- mized. Our algorithm, ADASTRAT, is the first data-aware method for this task. It chooses the next point to query on the PR curve adaptively, based on previous observations. It then selects specific items to annotate using stratified sam- pling. Under a mild monotonicity assumption, ADASTRAT outputs a guaranteed approxima- tion of the underlying precision function, while using a number of annotations that scales very slowly with N, the dataset size. For exam- ple, when the minimum precision is bounded by a constant, it issues only log log N preci- sion queries. In general, it has a regret of no more than log log N w.r.t. an oracle that is- sues queries at data-dependent (unknown) op- timal points. On a scaled-up NLP dataset of 3.5M items, ADASTRAT achieves a remark- ably close approximation of the true precision function using only 18 precision queries, 13x fewer than best previous approaches.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-guarnizo18a,
  title = 	 {Fast Kernel Approximations for Latent Force Models and Convolved Multiple-Output {G}aussian processes},
  author =       {Guarnizo, Cristian and {\'A}lvarez, Mauricio},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {834--843},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/guarnizo18a/guarnizo18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/guarnizo18a.html},
  abstract = 	 {A latent force model is a Gaussian process with a covariance function inspired by a differential operator. Such covariance function is obtained by performing convolution integrals between Green’s functions associated to the differential operators, and covariance functions associated to latent functions. In the classical formula- tion of latent force models, the covariance func- tions are obtained analytically by solving a dou- ble integral, leading to expressions that involve numerical solutions of different types of error functions. In consequence, the covariance ma- trix calculation is considerably expensive, be- cause it requires the evaluation of one or more of these error functions. In this paper, we use random Fourier features to approximate the so- lution of these double integrals obtaining sim- pler analytical expressions for such covariance functions. We show experimental results using ordinary differential operators and provide an extension to build general kernel functions for convolved multiple output Gaussian processes.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-cheng18a,
  title = 	 {Fast Policy Learning through Imitation and Reinforcement},
  author =       {Cheng, Ching-An and Yan, Xinyan and Wagener, Nolan and Boots, Byron},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {844--854},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/cheng18a/cheng18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/cheng18a.html},
  abstract = 	 {Imitation learning (IL) consists of a set of tools that leverage expert demonstrations to quickly learn policies. However, if the expert is subop- timal, IL can yield policies with inferior per- formance compared to reinforcement learning (RL). In this paper, we aim to provide an algo- rithm that combines the best aspects of RL and IL. We accomplish this by formulating sev- eral popular RL and IL algorithms in a com- mon mirror descent framework, showing that these algorithms can be viewed as a variation on a single approach. We then propose LOKI, a strategy for policy learning that first performs a small but random number of IL iterations be- fore switching to a policy gradient RL method. We show that if the switching time is prop- erly randomized, LOKI can learn to outperform a suboptimal expert and converge faster than running policy gradient from scratch. Finally, we evaluate the performance of LOKI experi- mentally in several simulated environments.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-davidson18a,
  title = 	 {Hyperspherical Variational Auto-Encoders},
  author =       {Davidson, Tim and Falorsi, Luca and De Cao, Nicola and Kipf, Thomas and Tomczak, Jakub M.},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {855--864},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/davidson18a/davidson18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/davidson18a.html},
  abstract = 	 {The Variational Auto-Encoder (VAE) is one of the most used unsupervised machine learn- ing models. But although the default choice of a Gaussian distribution for both the prior and posterior represents a mathematically con- venient distribution often leading to competi- tive results, we show that this parameterization fails to model data with a latent hyperspheri- cal structure. To address this issue we propose using a von Mises-Fisher (vMF) distribution in- stead, leading to a hyperspherical latent space. Through a series of experiments we show how such a hyperspherical VAE, or S-VAE, is more suitable for capturing data with a hyperspheri- cal latent structure, while outperforming a nor- mal, N-VAE, in low dimensions on other data types.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-chou18a,
  title = 	 {Dissociation-Based Oblivious Bounds for Weighted Model Counting},
  author =       {Chou, Li and Gatterbauer, Wolfgang and Gogate, Vibhav},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {865--874},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/chou18a/chou18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/chou18a.html},
  abstract = 	 {We consider the weighted model counting task which includes important tasks in graphical models, such as computing the partition func- tion and probability of evidence as special cases. We propose a novel partition-based bounding al- gorithm that exploits logical structure and gives rise to a set of inequalities from which upper (or lower) bounds can be derived efficiently. The bounds come with optimality guarantees under certain conditions and are oblivious in that they require only limited observations of the structure and parameters of the problem. We experimentally compare our bounds with the mini-bucket scheme (which is also oblivi- ous) and show that our new bounds are often superior and never worse on a wide variety of benchmark networks.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-izmailov18a,
  title = 	 {Averaging Weights Leads to Wider Optima and Better Generalization},
  author =       {Izmailov, Pavel and Podoprikhin, Dmitrii and Garipov, Timur and Vetrov, Dmitry and Wilson, Andrew Gordon},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {875--884},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/izmailov18a/izmailov18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/izmailov18a.html},
  abstract = 	 {Deep neural networks are typically trained by optimizing a loss function with an SGD vari- ant, in conjunction with a decaying learning rate, until convergence. We show that simple averaging of multiple points along the trajec- tory of SGD, with a cyclical or constant learn- ing rate, leads to better generalization than conventional training. We also show that this Stochastic Weight Averaging (SWA) procedure finds much broader optima than SGD, and ap- proximates the recent Fast Geometric Ensem- bling (FGE) approach with a single model. Using SWA we achieve notable improvement in test accuracy over conventional SGD train- ing on a range of state-of-the-art residual net- works, PyramidNets, DenseNets, and Shake- Shake networks on CIFAR-10, CIFAR-100, and ImageNet. In short, SWA is extremely easy to implement, improves generalization, and has almost no computational overhead.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-madan18a,
  title = 	 {Block-Value Symmetries in Probabilistic Graphical Models},
  author =       {Madan, Gagan and Anand, Ankit and Mausam and Singla, Parag},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {885--894},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/madan18a/madan18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/madan18a.html},
  abstract = 	 {One popular way for lifted inference in proba- bilistic graphical models is to first merge sym- metric states into a single cluster (orbit) and then use these for downstream inference, via variations of orbital MCMC [Niepert, 2012]. These orbits are represented compactly us- ing permutations over variables, and variable- value (VV) pairs, but they can miss several state symmetries in a domain. We define the notion of permutations over block-value (BV) pairs, where a block is a set of variables. BV strictly generalizes VV sym- metries, and can compute many more sym- metries for increasing block sizes. To opera- tionalize use of BV permutations in lifted in- ference, we describe 1) an algorithm to com- pute BV permutations given a block parti- tion of the variables, 2) BV-MCMC, an exten- sion of orbital MCMC that can sample from BV orbits, and 3) a heuristic to suggest good block partitions. Our experiments show that BV-MCMC can mix much faster compared to vanilla MCMC and orbital MCMC.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-krishnan18a,
  title = 	 {Max-margin learning with the {B}ayes factor},
  author =       {Krishnan, Rahul G. and Khandelwal, Arjun and Ranganath, Rajesh and Sontag, David},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {895--904},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/krishnan18a/krishnan18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/krishnan18a.html},
  abstract = 	 {We propose a new way to answer probabilis- tic queries that span multiple datapoints. We formalize reasoning about the similarity of dif- ferent datapoints as the evaluation of the Bayes Factor within a hierarchical deep generative model that enforces a separation between the latent variables used for representation learning and those used for reasoning. Under this model, we derive an intuitive estimator for the Bayes Factor that represents similarity as the amount of overlap in representation space shared by dif- ferent points. The estimator we derive relies on a query-conditional latent reasoning network, that parameterizes a distribution over the latent space of the deep generative model. The latent reasoning network is trained to amortize the posterior-predictive distribution under a hierar- chical model using supervised data and a max- margin learning algorithm. We explore how the model may be used to focus the data variations captured in the latent space of the deep genera- tive model and how this may be used to build new algorithms for few-shot learning.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-chen18b,
  title = 	 {Densified Winner Take All ({WTA}) Hashing for Sparse Datasets},
  author =       {Chen, Beidi and Shrivastava, Anshumali},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {905--915},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/chen18b/chen18b.pdf},
  url = 	 {https://proceedings.mlr.press/r16/chen18b.html},
  abstract = 	 {WTA (Winner Take All) hashing has been suc- cessfully applied in many large-scale vision applications. This hashing scheme was tai- lored to take advantage of the comparative rea- soning (or order based information), which showed significant accuracy improvements. In this paper, we identify a subtle issue with WTA, which grows with the sparsity of the datasets. This issue limits the discriminative power of WTA. We then propose a solution to this problem based on the idea of Densification which makes use of 2-universal hash functions in a novel way. Our experiments show that Densified WTA Hashing outperforms Vanilla WTA Hashing both in image retrieval and clas- sification tasks consistently and significantly.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-sharma18a,
  title = 	 {Lifted Marginal {MAP} Inference},
  author =       {Sharma, Vishal and Sheikh, Noman Ahmed and Mittal, Happy and Gogate, Vibhav and Singla, Parag},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {916--925},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/sharma18a/sharma18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/sharma18a.html},
  abstract = 	 {Lifted inference reduces the complexity of in- ference in relational probabilistic models by identifying groups of constants (or atoms) which behave symmetric to each other. A number of techniques have been proposed in the literature for lifting marginal as well MAP inference. We present the first application of lifting rules for marginal-MAP (MMAP), an important inference problem in models having latent (random) variables. Our main contribu- tion is two fold: (1) we define a new equiv- alence class of (logical) variables, called Sin- gle Occurrence for MAX (SOM), and show that solution lies at extreme with respect to the SOM variables, i.e., predicate groundings differing only in the instantiation of the SOM variables take the same truth value (2) we de- fine a sub-class SOM-R (SOM Reduce) and exploit properties of extreme assignments to show that MMAP inference can be performed by reducing the domain of SOM-R variables to a single constant. We refer to our lifting technique as the SOM-R rule for lifted MMAP. Combined with existing rules such as decom- poser and binomial, this results in a power- ful framework for lifted MMAP. Experiments on three benchmark domains show significant gains in both time and memory compared to ground inference as well as lifted approaches not using SOM-R.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-kuzelka18a,
  title = 	 {{PAC}-Reasoning in Relational Domains},
  author =       {Kuzelka, Ondrej and Wang, Yuyi and Davis, Jesse and Schockaert, Steven},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {926--935},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/kuzelka18a/kuzelka18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/kuzelka18a.html},
  abstract = 	 {We consider the problem of predicting plausible missing facts in relational data, given a set of imperfect logical rules. In particular, our aim is to provide bounds on the (expected) number of incorrect inferences that are made in this way. Since for classical inference it is in general impossible to bound this number in a non-trivial way, we consider two inference relations that weaken, but remain close in spirit to classical inference.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-yu18a,
  title = 	 {Pure Exploration of Multi-Armed Bandits with Heavy-Tailed Payoffs},
  author =       {Yu, Xiaotian and Shao, Han and Lyu, Michael R. and King, Irwin},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {936--945},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/yu18a/yu18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/yu18a.html},
  abstract = 	 {Inspired by heavy-tailed distributions in prac- tical scenarios, we investigate the problem on pure exploration of Multi-Armed Bandits (MAB) with heavy-tailed payoffs by breaking the assumption of payoffs with sub-Gaussian noises in MAB, and assuming that stochastic payoffs from bandits are with finite p-th mo- ments, where p $\in$(1, +$\infty$). The main contri- butions in this paper are three-fold. First, we technically analyze tail probabilities of empir- ical average and truncated empirical average (TEA) for estimating expected payoffs in se- quential decisions with heavy-tailed noises via martingales. Second, we propose two effective bandit algorithms based on different prior in- formation (i.e., fixed confidence or fixed bud- get) for pure exploration of MAB generating payoffs with finite p-th moments. Third, we derive theoretical guarantees for the proposed two bandit algorithms, and demonstrate the ef- fectiveness of two algorithms in pure explo- ration of MAB with heavy-tailed payoffs in synthetic data and real-world financial data.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-subbaswamy18a,
  title = 	 {Counterfactual Normalization: Proactively Addressing Dataset Shift Using Causal Mechanisms},
  author =       {Subbaswamy, Adarsh and Saria, Suchi},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {946--956},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/subbaswamy18a/subbaswamy18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/subbaswamy18a.html},
  abstract = 	 {Predictive models can fail to generalize from training to deployment environments because of dataset shift, posing a threat to model re- liability in practice. As opposed to previous methods which use samples from the target distribution to reactively correct dataset shift, we propose using graphical knowledge of the causal mechanisms relating variables in a pre- diction problem to proactively remove variables that participate in spurious associations with the prediction target, allowing models to gen- eralize across datasets. To accomplish this, we augment the causal graph with latent counter- factual variables that account for the underlying causal mechanisms, and show how we can es- timate these variables. In our experiments we demonstrate that models using good estimates of the latent variables instead of the observed variables transfer better from training to tar- get domains with minimal accuracy loss in the training domain.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-agrawal18a,
  title = 	 {Decentralized Planning for Non-dedicated Agent Teams with Submodular Rewards in Uncertain Environments},
  author =       {Agrawal, Pritee and Varakantham, Pradeep and Yeoh, William},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {957--966},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/agrawal18a/agrawal18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/agrawal18a.html},
  abstract = 	 {Decentralized planning under uncertainty for agent teams is a problem of interest in many domains including (but not limited to) disas- ter rescue, sensor networks and security pa- trolling. Decentralized MDPs, Dec-MDPs have traditionally been used to represent such decen- tralized planning under uncertainty problems. However, in many domains, agents may not be dedicated to the team for the entire time horizon. For instance, due to limited availabil- ity of resources, it is quite common for police personnel leaving patrolling teams to attend to accidents. Such non-dedication can arise due to the emergence of higher priority tasks or damage to existing agents. However, there is very limited literature dealing with handling of non-dedication in decentralized settings. To that end, we provide a general model to rep- resent problems dealing with cooperative and decentralized planning for non-dedicated agent teams. We also provide two greedy approaches (an offline one and an offline-online one) that are able to deal with agents leaving the team in an effective and efficient way by exploiting the submodularity property. Finally, we demon- strate that our approaches are able to obtain more than 90% of optimal solution quality on benchmark problems from the literature.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-lawton18a,
  title = 	 {A Forest Mixture Bound for Block-Free Parallel Inference},
  author =       {Lawton, Neal and Steeg, Greg Ver and Galstyan, Aram},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {967--976},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/lawton18a/lawton18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/lawton18a.html},
  abstract = 	 {Coordinate ascent variational inference is an important algorithm for inference in proba- bilistic models, but it is slow because it updates only a single variable at a time. Block coordi- nate methods perform inference faster by up- dating blocks of variables in parallel. How- ever, the speed and convergence of these algo- rithms depends on how the variables are par- titioned into blocks. In this paper, we give a convergent parallel algorithm for inference in deep exponential families that doesn’t require the variables to be partitioned into blocks. We achieve this by lower bounding the ELBO by a new objective we call the forest mixture bound (FM bound) that separates the inference prob- lem for variables within a hidden layer. We apply this to the simple case when all random variables are Gaussian and show empirically that the algorithm converges faster for models that are inherently more forest-like.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-jaber18a,
  title = 	 {Causal Identification under {M}arkov Equivalence},
  author =       {Jaber, Amin and Zhang, Jiji and Bareinboim, Elias},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {977--986},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/jaber18a/jaber18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/jaber18a.html},
  abstract = 	 {Assessing the magnitude of cause-and-effect relations is one of the central challenges found throughout the empirical sciences. The prob- lem of identification of causal effects is con- cerned with determining whether a causal ef- fect can be computed from a combination of observational data and substantive knowledge about the domain under investigation, which is formally expressed in the form of a causal graph. In many practical settings, however, the knowledge available for the researcher is not strong enough so as to specify a unique causal graph. Another line of investigation attempts to use observational data to learn a qualita- tive description of the domain called a Markov equivalence class, which is the collection of causal graphs that share the same set of ob- served features. In this paper, we marry both approaches and study the problem of causal identification from an equivalence class, repre- sented by a partial ancestral graph (PAG). We start by deriving a set of graphical properties of PAGs that are carried over to its induced sub- graphs. We then develop an algorithm to com- pute the effect of an arbitrary set of variables on an arbitrary outcome set. We show that the algorithm is strictly more powerful than the current state of the art found in the literature.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-hewitt18a,
  title = 	 {The Variational Homoencoder: Learning to learn high capacity generative models from few examples},
  author =       {Hewitt, Luke B. and Nye, Maxwell I. and Gane, Andreea and Jaakkola, Tommi and Tenenbaum, Joshua B.},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {987--996},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/hewitt18a/hewitt18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/hewitt18a.html},
  abstract = 	 {Hierarchical Bayesian methods can unify many related tasks (e.g. k-shot classification, conditional and unconditional generation) as inference within a single generative model. However, when this generative model is ex- pressed as a powerful neural network such as a PixelCNN, we show that existing learning techniques typically fail to effectively use la- tent variables. To address this, we develop a modification of the Variational Autoencoder in which encoded observations are decoded to new elements from the same class. This technique, which we call a Variational Ho- moencoder (VHE), produces a hierarchical la- tent variable model which better utilises la- tent variables. We use the VHE framework to learn a hierarchical PixelCNN on the Omniglot dataset, which outperforms all existing models on test set likelihood and achieves strong per- formance on one-shot generation and classifi- cation tasks. We additionally validate the VHE on natural images from the YouTube Faces database. Finally, we develop extensions of the model that apply to richer dataset structures such as factorial and hierarchical categories.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-salah18a,
  title = 	 {Probabilistic Collaborative Representation Learning for Personalized Item Recommendation},
  author =       {Salah, Aghiles and Lauw, Hady W.},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {997--1007},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/salah18a/salah18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/salah18a.html},
  abstract = 	 {We present Probabilistic Collaborative Repre- sentation Learning (PCRL), a new generative model of user preferences and item contexts. The latter builds on the assumption that rela- tionships among items within contexts (e.g., browsing session, shopping cart, etc.) may un- derlie various aspects that guide the choices people make. Intuitively, PCRL seeks repre- sentations of items reflecting various regulari- ties between them that might be useful at ex- plaining user preferences. Formally, it relies on Bayesian Poisson Factorization to model user-item interactions, and uses a multilayered latent variable architecture to learn represen- tations of items from their contexts. PCRL seamlessly integrates both tasks within a joint framework. However, inference and learn- ing under the proposed model are challenging due to several sources of intractability. Rely- ing on the recent advances in approximate in- ference/learning, we derive an efficient varia- tional algorithm to estimate our model from observations. We further conduct experiments on several real-world datasets to showcase the benefits of the proposed model.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-palmer18a,
  title = 	 {Reforming Generative Autoencoders via Goodness-of-Fit Hypothesis Testing},
  author =       {Palmer, Aaron and Dey, Dipak and Bi, Jinbo},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {1008--1018},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/palmer18a/palmer18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/palmer18a.html},
  abstract = 	 {Generative models, while not new, have taken the deep learning field by storm. However, the widely used training methods have not exploited the substantial statistical literature concerning parametric distributional testing. Having sound theoretical foundations, these goodness-of-fit tests enable parts of the black box to be stripped away. In this paper we use the Shapiro-Wilk and propose a new multivari- ate generalization of Shapiro-Wilk to respec- tively test for univariate and multivariate nor- mality of the code layer of a generative autoen- coder. By replacing the discriminator in tradi- tional deep models with the hypothesis tests, we gain several advantages: objectively evalu- ate whether the encoder is actually embedding data onto a normal manifold, accurately define when convergence happens, explicitly balance between reconstruction and encoding training. Not only does our method produce competitive results, but it does so in a fraction of the time. We highlight the fact that the hypothesis tests used in our model asymptotically lead to the same solution of the L2-Wasserstein distance metrics used by several generative models to- day.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-seong18a,
  title = 	 {Towards Flatter Loss Surface via Nonmonotonic Learning Rate Scheduling},
  author =       {Seong, Sihyeon and Lee, Yegang and Kee, Youngwook and Han, Dongyoon and Kim, Junmo},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {1019--1029},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/seong18a/seong18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/seong18a.html},
  abstract = 	 {Whereas optimizing deep neural networks us- ing stochastic gradient descent has shown great performances in practice, the rule for setting step size (i.e. learning rate) of gradient de- scent is not well studied. Although it appears that some intriguing learning rate rules such as ADAM (Kingma and Ba, 2014) have since been developed, they concentrated on improv- ing convergence, not on improving generaliza- tion capabilities. Recently, the improved gen- eralization property of the flat minima was re- visited, and this research guides us towards promising solutions to many current optimiza- tion problems. In this paper, we analyze the flatness of loss surfaces through the lens of ro- bustness to input perturbations and advocate that gradient descent should be guided to reach flatter region of loss surfaces to achieve gen- eralization. Finally, we suggest a learning rate rule for escaping sharp regions of loss surfaces, and we demonstrate the capacity of our ap- proach by performing numerous experiments.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-zhao18c,
  title = 	 {A {L}agrangian Perspective on Latent Variable Generative Models},
  author =       {Zhao, Shengjia and Song, Jiaming and Ermon, Stefano},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {1030--1040},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/zhao18c/zhao18c.pdf},
  url = 	 {https://proceedings.mlr.press/r16/zhao18c.html},
  abstract = 	 {A large number of objectives have been proposed to train latent variable generative models. We show that many of them are Lagrangian dual functions of the same primal optimization prob- lem. The primal problem optimizes the mutual information between latent and visible variables, subject to the constraints of accurately model- ing the data distribution and performing correct amortized inference. By choosing to maximize or minimize mutual information, and choosing different Lagrange multipliers, we obtain differ- ent objectives including InfoGAN, ALI/BiGAN, ALICE, CycleGAN, beta-VAE, adversarial au- toencoders, AVB, AS-VAE and InfoVAE. Based on this observation, we provide an exhaustive characterization of the statistical and computa- tional trade-offs made by all the training objec- tives in this class of Lagrangian duals. Next, we propose a dual optimization method where we optimize model parameters as well as the La- grange multipliers. This method achieves Pareto optimal solutions in terms of optimizing informa- tion and satisfying the constraints.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-eismann18a,
  title = 	 {{B}ayesian optimization and attribute adjustment},
  author =       {Eismann, Stephan and Levy, Daniel and Shu, Rui and Bartzsch, Stefan and Ermon, Stefano},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {1041--1051},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/eismann18a/eismann18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/eismann18a.html},
  abstract = 	 {Automatic design via Bayesian optimization holds great promise given the constant increase of available data across domains. However, it faces difficulties from high-dimensional, poten- tially discrete, search spaces. We propose to probabilistically embed inputs into a lower di- mensional, continuous latent space, where we perform gradient-based optimization guided by a Gaussian process. Building on variational au- toncoders, we use both labeled and unlabeled data to guide the encoding and increase its ac- curacy. In addition, we propose an adversar- ial extension to render the latent representa- tion invariant with respect to specific design attributes, which allows us to transfer these at- tributes across structures. We apply the frame- work both to a functional-protein dataset and to perform optimization of drag coefficients di- rectly over high-dimensional shapes without in- corporating domain knowledge or handcrafted features.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-lee18a,
  title = 	 {Join Graph Decomposition Bounds for Influence Diagrams},
  author =       {Lee, Junkyu and Ihler, Alexander and Dechter, Rina},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {1052--1061},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/lee18a/lee18a.pdf},
  url = 	 {https://proceedings.mlr.press/r16/lee18a.html},
  abstract = 	 {We introduce a new decomposition method for bounding the maximum expected utility of in- fluence diagrams. While most current schemes use reductions to the Marginal Map task over a Bayesian Network, our approach is direct, aim- ing to avoid the large explosion in the model size that often results by such reductions. In this paper, we extend to influence diagrams the principles of decomposition methods that were applied earlier to probabilistic inference, uti- lizing an algebraic framework called valuation algebra which effectively captures both multi- plicative and additive local structures present in influence diagrams. Empirical evaluation on four benchmarks demonstrates the effectiveness of our approach compared to reduction-based approaches and illustrates significant improve- ments in the upper bounds on maximum ex- pected utility.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR16-zhang18c,
  title = 	 {Causal Discovery with Linear Non-{G}aussian Models under Measurement Error: Structural Identifiability Results},
  author =       {Zhang, Kun and Gong, Mingming and Ramsey, Joseph and Batmanghelich, Kayhan and Spirtes, Peter and Glymour, Clark},
  booktitle = 	 {Proceedings of the 34th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {1062--1071},
  year = 	 {2018},
  editor = 	 {Globerson, Amir and Silva, Ricardo},
  volume = 	 {R16},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {06--10 Aug},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r16/main/assets/zhang18c/zhang18c.pdf},
  url = 	 {https://proceedings.mlr.press/r16/zhang18c.html},
  abstract = 	 {Causal discovery methods aim to recover the causal process that generated purely observa- tional data. Despite its successes on a number of real problems, the presence of measurement error in the observed data can produce seri- ous mistakes in the output of various causal discovery methods. Given the ubiquity of measurement error caused by instruments or proxies used in the measuring process, this problem is one of the main obstacles to reli- able causal discovery. It is still unknown to what extent the causal structure of relevant variables can be identified in principle. This study aims to take a step towards filling that void. We assume that the underlining pro- cess or the measurement-error free variables follows a linear, non-Guassian causal model, and show that the so-called ordered group decomposition of the causal model, which con- tains major causal information, is identifiable. The causal structure identifiability is further improved with different types of sparsity con- straints on the causal structure. Finally, we give rather mild conditions under which the whole causal structure is fully identifiable.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



