


@Proceedings{UAI2014,
  title =     {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  booktitle = {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  editor =    {Nevin L. Zhang and Jin Tian},
  publisher = {PMLR},
  series =    {Proceedings of Machine Learning Research},
  volume =    R12
}



@InProceedings{pmlr-vR12-zhang14a,
  title = 	 {The 30th Uncertainty in Artificial Intelligence Conference: Preface},
  author =       {Zhang, Nevin L. and Tian, Jin},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {1--6},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/zhang14a/zhang14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/zhang14a.html},
  abstract = 	 {vii Organizing Committee ix Acknowledgments xi Sponsors xvii Best Paper Awards xix 1 Proceedings 1 MEMR: A Margin Equipped Monotone Retargeting Framework for Ranking. Sreangsu Acharyya, Joydeep Ghosh . . . . . . . . . . . . . . . . . . . . . . . . . . . . . . . 1 On Convergence and Optimality of Best-Response Learning with Policy Types in Multiagent Systems. Stefano Albrecht, Subramanian Ramamoorthy . . . . . . . . . . . . . . . . . . . . . . . . . 12 Accelerating MCMC via Parallel Predictive Prefetching. Elaine Angelino, Eddie Kohler, Margo Seltzer, Amos Waterland, Ryan Adams . . . . . . 22 A variational approach to stable principal component pursuit. Aleksandr Aravkin, Stephen Becker, Volkan Cevher, Peder Olsen . . . . . . . . . . . . . . 32 Markov Network Structure Learning via Ensemble-of-Forests Models. Eirini Arvaniti, Manfred Claassen . . . . . . . . . . . . . . . . . . . . . . . . . . . . . . . 42 Can(Plan)+: Extending the Operational Semantics of the BDI architecture to deal with Uncertain Information. Kim Bauters, Weiru Liu, Jun Hong, Carles Sierra, Lluis Godo . . . . . . . . . . . . . . . 52 Message Passing for Soft Constraint Dual Decomposition. David Belanger, Alexandre Passos, Sebastian Riedel, Andrew McCallum . . . . . . . . . . 62 Bayesian Interactive Decision Support for Multi-Attribute Problems with Even Swaps. Debarun Bhattacharjya, Jeffrey Kephart . . . . . . . . . . . . . . . . . . . . . . . . . . . . 72 Learning to Predict from Crowdsourced Data. Wei Bi, Liwei Wang, James Kwok, Zhuowen Tu . . . . . . . . . . . . . . . . . . . . . . . 82 Lifted Tree-Reweighted Variational Inference. Hung Bui, Tuyen Huynh, David Sontag . . . . . . . . . . . . . . . . . . . . . . . . . . . . 92 Approximate Decentralized Bayesian Inference. Trevor Campbell, Jonathan How . . . . . . . . . . . . . . . . . . . . . . . . . . . . . . . . 102 Inferring latent structures via information inequalities. Rafael Chaves, Lukas Luft, Thiago Maciel, David Gross, Dominik Janzing, Bernhard Sch\"{}olkopf . . . . . . . . . . . . . . . . . . . . . . . . . . . . . . . . . . . . . . . . . . . . . 112 Near-optimal Adaptive Pool-based Active Learning with General Loss. Nguyen Viet Cuong, Wee Sun Lee, Nan Ye . . . . . . . . . . . . . . . . . . . . . . . . . . 122 A Permutation-Based Kernel Conditional Independence Test. Gary Doran, Krikamol Muandet, Kun Zhang, Bernhard Sch\"{}olkopf . . . . . . . . . . . . . 132 Parallel Markov Chain Monte Carlo for Pitman-Yor Mixture Models. Kumar Dube},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-marchant14a,
  title = 	 {Sequential {B}ayesian Optimisation for Spatial-Temporal Monitoring},
  author =       {Marchant, Roman and Ramos, Fabio and Sanner, Scott},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {7--16},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/marchant14a/marchant14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/marchant14a.html},
  abstract = 	 {Bayesian Optimisation has received considerable attention in recent years as a general methodol- ogy to find the maximum of costly-to-evaluate objective functions. Most existing BO work fo- cuses on where to gather a set of samples with- out giving special consideration to the sampling sequence, or the costs or constraints associated with that sequence. However, in real-world sequential decision problems such as robotics, the order in which samples are gathered is paramount, especially when the robot needs to optimise a temporally non-stationary objective function. Additionally, the state of the environ- ment and sensing platform determine the type and cost of samples that can be gathered. To address these issues, we formulate Sequential Bayesian Optimisation (SBO) with side-state in- formation within a Partially Observed Markov Decision Process (POMDP) framework that can accommodate discrete and continuous observa- tion spaces. We build on previous work using Monte-Carlo Tree Search (MCTS) and Upper Confidence bound for Trees (UCT) for POMDPs and extend it to work with continuous state and observation spaces. Through a series of experi- ments on monitoring a spatial-temporal process with a mobile robot, we show that our UCT- based SBO POMDP optimisation outperforms myopic and non-myopic alternatives.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-foulds14a,
  title = 	 {Annealing Paths for the Evaluation of Topic Models},
  author =       {Foulds, James and Smyth, Padhraic},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {17--26},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/foulds14a/foulds14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/foulds14a.html},
  abstract = 	 {Statistical topic models such as latent Dirich- let allocation have become enormously popu- lar in the past decade, with dozens of learning algorithms and extensions being proposed each year. As these models and algorithms continue to be developed, it becomes increasingly impor- tant to evaluate them relative to previous tech- niques. However, evaluating the predictive per- formance of a topic model is a computationally difficult task. Annealed importance sampling (AIS), a Monte Carlo technique which operates by annealing between two distributions, has pre- viously been successfully used for topic model evaluation (Wallach et al., 2009b). This tech- nique estimates the likelihood of a held-out doc- ument by simulating an annealing process from the prior to the posterior for the latent topic as- signments, and using this simulation as an im- portance sampling proposal distribution. In this paper we introduce new AIS annealing paths which instead anneal from one topic model to another, thereby estimating the relative perfor- mance of the models. This strategy can exhibit much lower empirical variance than previous ap- proaches, facilitating reliable per-documentcom- parisons of topic models. We then show how to use these paths to evaluate the predictive perfor- mance of topic model learning algorithms by effi- ciently estimating the likelihood at each iteration of the training procedure. The proposed method achieves better held-out likelihood estimates for this task than previous algorithms with, in some cases, an order of magnitude less computation.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-technology14a,
  title = 	 {Constraint-based Causal Discovery: Conflict Resolution with Answer Set Programming},
  author =       {Technology, Antti Hyttinen California Institute of and Caltech, Frederick Eberhardt and Helsinki, Matti J{\"a}rvisalo HIIT/University of},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {27--36},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/technology14a/technology14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/technology14a.html},
  abstract = 	 {Recent approaches to causal discovery based on Boolean satisfiability solvers have opened new opportunities to consider search spaces for causal models with both feedback cycles and unmea- sured confounders. However, the available meth- ods have so far not been able to provide a prin- cipled account of how to handle conflicting con- straints that arise from statistical variability. Here we present a new approach that preserves the ver- satility of Boolean constraint solving and attains a high accuracy despite the presence of statisti- cal errors. We develop a new logical encoding of (in)dependence constraints that is both well suited for the domain and allows for faster solv- ing. We represent this encoding in Answer Set Programming (ASP), and apply a state-of-the- art ASP solver for the optimization task. Based on different theoretical motivations, we explore a variety of methods to handle statistical errors. Our approach currently scales to cyclic latent variable models with up to seven observed vari- ables and outperforms the available constraint- based methods in accuracy.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-university14a,
  title = 	 {Learning from Point Sets with Observational Bias},
  author =       {University, Liang Xiong Carnegie Mellon and University, Jeff Schneider Carnegie Mellon},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {37--45},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/university14a/university14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/university14a.html},
  abstract = 	 {Many objects can be represented as sets of multi- dimensional points. A common approach to learning from these point sets is to assume that each set is an i.i.d. sample from an unknown un- derlying distribution, and then estimate the sim- ilarities between these distributions. In realistic situations, however, the point sets are often sub- ject to sampling biases due to variable or incon- sistent observation actions. These biases can fun- damentally change the observed distributions of points and distort the results of learning. In this paper we propose the use of conditional diver- gences to correct these distortions and learn from biased point sets effectively. Our empirical study shows that the proposed method can successfully correct the biases and achieve satisfactory learn- ing performance.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-university14b,
  title = 	 {{B}ayesian Optimization with Unknown Constraints},
  author =       {University, Michael Gelbart Harvard and University, Jasper Snoek Harvard and Harvard, Ryan Adams},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {46--55},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/university14b/university14b.pdf},
  url = 	 {https://proceedings.mlr.press/r12/university14b.html},
  abstract = 	 {Recent work on Bayesian optimization has shown its effectiveness in global optimization of difficult black-box objective functions. Many real-world optimization problems of interest also have constraints which are unknown a priori. In this paper, we study Bayesian optimization for constrained problems in the general case that noise may be present in the constraint func- tions, and the objective and constraints may be evaluated independently. We provide motivating practical examples, and present a general frame- work to solve such problems. We demonstrate the effectiveness of our approach on optimizing the performance of online latent Dirichlet allo- cation subject to topic sparsity constraints, tun- ing a neural network given test-time memory constraints, and optimizing Hamiltonian Monte Carlo to achieve maximal effectiveness in a fixed time, subject to passing standard convergence di- agnostics.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-dvijotham14a,
  title = 	 {Universal Convexification via Risk-Aversion},
  author =       {Dvijotham, Krishnamurthy and Fazel, Maryam and Todorov, Emanuel},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {56--65},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/dvijotham14a/dvijotham14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/dvijotham14a.html},
  abstract = 	 {We develop a framework for convexifying a general class of optimization problems. We analyze the suboptimality of the solution to the convexified problem relative to the original nonconvex problem, and prove ad- ditive approximation guarantees under some assumptions. In simple settings, the convexi- fication procedure can be applied directly and standard optimization methods can be used. In the general case we rely on stochastic gra- dient algorithms, whose convergence rate can be bounded using the convexity of the under- lying optimization problem. We then extend the framework to a general class of discrete- time dynamical systems where our convex- ification approach falls under the paradigm of risk-sensitive Markov Decision Processes. We derive the first model-based and model- free policy gradient optimization algorithms with guaranteed convergence to the optimal solution. We also present numerical results in different machine learning applications.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-nguyen14a,
  title = 	 {Collaborative Multi-output {G}aussian Processes},
  author =       {Nguyen, Trung and Australia, Edwin Bonilla National ICT},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {66--75},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/nguyen14a/nguyen14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/nguyen14a.html},
  abstract = 	 {We introduce the collaborative multi-output Gaussian process (GP) model for learning dependent tasks with very large datasets. The model fosters task correlations by mixing sparse processes and sharing multiple sets of inducing points. This facilitates the applica- tion of variational inference and the deriva- tion of an evidence lower bound that decom- poses across inputs and outputs. We learn all the parameters of the model in a sin- gle stochastic optimization framework that scales to a large number of observations per output and a large number of outputs. We demonstrate our approach on a toy prob- lem, two medium-sized datasets and a large dataset. The model achieves superior per- formance compared to single output learn- ing and previous multi-output GP models, confirming the benefits of correlating spar- sity structure of the outputs via the inducing points.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-labs14a,
  title = 	 {Matroid Bandits: Fast Combinatorial Optimization with Learning},
  author =       {Labs, Branislav Kveton Technicolor and Wen, Zheng and Ashkan, Azin and Eydgahi, Hoda and Eriksson, Brian},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {76--85},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/labs14a/labs14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/labs14a.html},
  abstract = 	 {A matroid is a notion of independence in combi- natorial optimization which is closely related to computational efficiency. In particular, it is well known that the maximum of a constrained mod- ular function can be found greedily if and only if the constraints are associated with a matroid. In this paper, we bring together the ideas of bandits and matroids, and propose a new class of combi- natorial bandits, matroid bandits. The objective in these problems is to learn how to maximize a modular function on a matroid. This function is stochastic and initially unknown. We propose a practical algorithm for solving our problem, Op- timistic Matroid Maximization (OMM); and prove two upper bounds, gap-dependent and gap-free, on its regret. Both bounds are sublinear in time and at most linear in all other quantities of inter- est. The gap-dependent upper bound is tight and we prove a matching lower bound on a partition matroid bandit. Finally, we evaluate our method on three real-world problems and show that it is practical.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-wald14a,
  title = 	 {Tightness Results for Local Consistency Relaxations in Continuous MRFs},
  author =       {Wald, Yoav and Globerson, Amir},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {86--95},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/wald14a/wald14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/wald14a.html},
  abstract = 	 {Finding the MAP assignment in graphical mod- els is a challenging task that generally requires approximations. One popular approximation ap- proach is to use linear programming relaxations that enforce local consistency. While these are commonly used for discrete variable models, they are much less understood for models with continuous variables. Here we define local consistency relaxations of MAP for continuous pairwise Markov Random Fields (MRFs), and analyze their properties. We begin by providing a characterization of models for which this relaxation is tight. These turn out to be models that can be reparameterized as a sum of local convex functions. We also provide a simple formulation of this relaxation for Gaus- sian MRFs. Next, we show how the above insights can be used to obtain optimality certificates for loopy belief propagation (LBP) in such models. Specif- ically, we show that the messages of LBP can be used to calculate upper and lower bounds on the MAP value, and that these bounds coincide at convergence, yielding a natural stopping crite- rion which was not previously available. Finally, our results illustrate a close connection between local consistency relaxations of MAP and LBP. They demonstrate that in the continu- ous case, whenever LBP is provably optimal so is the local consistency relaxation.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-chaves14a,
  title = 	 {Inferring latent structures via information inequalities},
  author =       {Chaves, Rafael and Luft, Lukas and Gerais, Thiago Maciel Federal University of Minas and Gross, David and Janzing, Dominik and Sch{\"o}lkopf, Bernhard},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {96--105},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/chaves14a/chaves14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/chaves14a.html},
  abstract = 	 {One of the goals of probabilistic inference is to decide whether an empirically observed distribution is compatible with a candidate Bayesian network. However, Bayesian net- works with hidden variables give rise to highly non-trivial constraints on the ob- served distribution. Here, we propose an information-theoretic approach, based on the insight that conditions on entropies of Bayesian networks take the form of simple linear inequalities. We describe an algorithm for deriving entropic tests for latent struc- tures. The well-known conditional indepen- dence tests appear as a special case. While the approach applies for generic Bayesian networks, we presently adopt the causal view, and show the versatility of the framework by treating several relevant problems from that domain: detecting common ancestors, quan- tifying the strength of causal influence, and inferring the direction of causation from two- variable marginals.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-marinescu14a,
  title = 	 {{AND}/{OR} Search for Marginal {MAP}},
  author =       {Marinescu, Radu and Dechter, Rina and Ihler, Alexander},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {106--115},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/marinescu14a/marinescu14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/marinescu14a.html},
  abstract = 	 {Marginal MAP problems are known to be very difficult tasks for graphical models and are so far solved exactly by systematic search guided by a join-tree upper bound. In this paper, we develop new AND/OR branch and bound algorithms for marginal MAP that use heuristics extracted from weighted mini-buckets enhanced with message- passing updates. We demonstrate the effective- ness of the resulting search algorithms against previous join-tree based approaches, which we also extend to accommodate high induced width models, through extensive empirical evaluations. Our results show not only orders-of-magnitude improvements over the state-of-the-art, but also the ability to solve problem instances well be- yond the reach of previous approaches.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-university14c,
  title = 	 {A Permutation-Based Kernel Conditional Independence Test},
  author =       {University, Gary Doran Case Western Reserve and Systems, Krikamol Muandet MPI for Intelligent and Systems, Kun Zhang MPI for Intelligent and Sch{\"o}lkopf, Bernhard},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {116--125},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/university14c/university14c.pdf},
  url = 	 {https://proceedings.mlr.press/r12/university14c.html},
  abstract = 	 {Determining conditional independence (CI) re- lationships between random variables is a chal- lenging but important task for problems such as Bayesian network learning and causal discovery. We propose a new kernel CI test that uses a sin- gle, learned permutation to convert the CI test problem into an easier two-sample test problem. The learned permutation leaves the joint distri- bution unchanged if and only if the null hypoth- esis of CI holds. Then, a kernel two-sample test, which has been studied extensively in prior work, can be applied to a permuted and an unpermuted sample to test for CI. We demonstrate that the test (1) easily allows the incorporation of prior knowledge during the permutation step, (2) has power competitive with state-of-the-art kernel CI tests, and (3) accurately estimates the null distri- bution of the test statistic, even as the dimension- ality of the conditioning variable grows.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-lattimore14a,
  title = 	 {Optimal Resource Allocation with Semi-Bandit Feedback},
  author =       {Lattimore, Tor and Crammer, Koby and Szepesvari, Csaba},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {126--135},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/lattimore14a/lattimore14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/lattimore14a.html},
  abstract = 	 {We study a sequential resource allocation prob- lem involving a fixed number of recurring jobs. At each time-step the manager should distribute available resources among the jobs in order to maximise the expected number of completed jobs. Allocating more resources to a given job in- creases the probability that it completes, but with a cut-off. Specifically, we assume a linear model where the probability increases linearly until it equals one, after which allocating additional re- sources is wasteful. We assume the difficulty of each job is unknown and present the first algo- rithm for this problem and prove upper and lower bounds on its regret. Despite its apparent sim- plicity, the problem has a rich structure: we show that an appropriate optimistic algorithm can im- prove its learning speed dramatically beyond the results one normally expects for similar problems as the problem becomes resource-laden.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-zander14a,
  title = 	 {Constructing Separators and Adjustment Sets in Ancestral Graphs},
  author =       {der Zander, Benito van and Liskiewicz, Maciej and Textor, Johannes},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {136--145},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/zander14a/zander14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/zander14a.html},
  abstract = 	 {Ancestral graphs (AGs) are graphical causal models that can represent uncertainty about the presence of latent confounders, and can be in- ferred from data. Here, we present an algo- rithmic framework for efficiently testing, con- structing, and enumerating m-separators in AGs. Moreover, we present a new constructive crite- rion for covariate adjustment in directed acyclic graphs (DAGs) and maximal ancestral graphs (MAGs) that characterizes adjustment sets as m- separators in a subgraph. Jointly, these results allow to find all adjustment sets that can iden- tify a desired causal effect with multivariate ex- posures and outcomes in the presence of latent confounding. Our results generalize and improve upon several existing solutions for special cases of these problems.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-mladenov14a,
  title = 	 {Lifted Message Passing as Reparametrization of Graphical Models},
  author =       {Mladenov, Martin and Kersting, Kristian and Globerson, Amir},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {146--155},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/mladenov14a/mladenov14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/mladenov14a.html},
  abstract = 	 {Lifted inference approaches can considerably speed up probabilistic inference in Markov ran- dom fields (MRFs) with symmetries. Given ev- idence, they essentially form a lifted, i.e., re- duced factor graph by grouping together indistin- guishable variables and factors. Typically, how- ever, lifted factor graphs are not amenable to off- the-shelf message passing (MP) approaches, and hence requires one to use either generic opti- mization tools, which would be slow for these problems, or design modified MP algorithms. Here, we demonstrate that the reliance on mod- ified MP can be eliminated for the class of MP algorithms arising from MAP-LP relaxations of pairwise MRFs. Specifically, we show that a given MRF induces a whole family of MRFs of different sizes sharing essentially the same MAP- LP solution. In turn, we give an efficient algo- rithm to compute from them the smallest one that can be solved using off-the-shelf MP. This incurs no major overhead: the selected MRF is at most twice as large as the fully lifted factor graph. This has several implications for lifted inference. For instance, running MPLP results in the first con- vergent lifted MP approach for MAP-LP relax- ations. Doing so can be faster than solving the MAP-LP using lifted linear programming. Most importantly, it suggests a novel view on lifted in- ference: it can be viewed as standard inference in a reparametrized model.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-nus14a,
  title = 	 {Near-optimal Adaptive Pool-based Active Learning with General Loss},
  author =       {NUS, Nguyen Viet Cuong and NUS, Wee Sun Lee and NUS, Nan Ye},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {156--165},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/nus14a/nus14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/nus14a.html},
  abstract = 	 {We consider adaptive pool-based active learning in a Bayesian setting. We first analyze two com- monly used greedy active learning criteria: the maximum entropy criterion, which selects the example with the highest entropy, and the least confidence criterion, which selects the example whose most probable label has the least probabil- ity value. We show that unlike the non-adaptive case, the maximum entropy criterion is not able to achieve an approximation that is within a con- stant factor of optimal policy entropy. For the least confidence criterion, we show that it is able to achieve a constant factor approximation to the optimal version space reduction in a worst-case setting, where the probability of labelings that have not been eliminated is considered as the ver- sion space. We consider a third greedy active learning criterion, the Gibbs error criterion, and generalize it to handle arbitrary loss functions be- tween labelings. We analyze the properties of the generalization and its variants, and show that they perform well in practice.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-dudik14a,
  title = 	 {Market Making with Decreasing Utility for Information},
  author =       {Dudik, Miroslav and Frongillo, Rafael and Vaughan, Jennifer Wortman},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {166--175},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/dudik14a/dudik14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/dudik14a.html},
  abstract = 	 {We study information elicitation in cost-func- tion-based combinatorial prediction markets when the market maker’s utility for information decreases over time. In the sudden revelation set- ting, it is known that some piece of information will be revealed to traders, and the market maker wishes to prevent guaranteed profits for trading on the sure information. In the gradual decrease setting, the market maker’s utility for (partial) in- formation decreases continuously over time. We design adaptive cost functions for both settings which: (1) preserve the information previously gathered in the market; (2) eliminate (or dimin- ish) rewards to traders for the publicly revealed information; (3) leave the reward structure unaf- fected for other information; and (4) maintain the market maker’s worst-case loss. Our construc- tions utilize mixed Bregman divergence, which matches our notion of utility for information.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-university14d,
  title = 	 {k-{NN} Regression on Functional Data with Incomplete Observations},
  author =       {University, Sashank J. Reddi Carnegie Mellon and Univeristy, Barnabas Poczos Carnegie Mellon},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {176--185},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/university14d/university14d.pdf},
  url = 	 {https://proceedings.mlr.press/r12/university14d.html},
  abstract = 	 {In this paper we study a general version of re- gression where each covariate itself is a func- tional data such as distributions or functions. In real applications, however, typically we do not have direct access to such data; instead only some noisy estimates of the true co- variate functions/distributions are available to us. For example, when each covariate is a distribution, then we might not be able to directly observe these distributions, but it can be assumed that i.i.d. sample sets from these distributions are available. In this pa- per we present a general framework and a k- NN based estimator for this regression prob- lem. We prove consistency of the estimator and derive its convergence rates. We further show that the proposed estimator can adapt to the local intrinsic dimension in our case and provide a simple approach for choosing k. Finally, we illustrate the applicability of our framework with numerical experiments.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-cuny14a,
  title = 	 {There {IS} a Free Lunch: Constraints for Learning {B}ayesian Networks},
  author =       {CUNY, Xiannian Fan Graduate Center and Finland", Brandon Malone "Helsinki Institute for Information Technology and York, Changhe Yuan City University of New},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {186--195},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/cuny14a/cuny14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/cuny14a.html},
  abstract = 	 {Several recent algorithms for learning Bayesian network structures first calculate potentially op- timal parent sets (POPS) for all variables and then use various optimization techniques to find a set of POPS, one for each variable, that con- stitutes an optimal network structure. This pa- per makes the observation that there is useful information implicit in the POPS. Specifically, the POPS of a variable constrain its parent can- didates. Moreover, the parent candidates of all variables together give a directed cyclic graph, which often decomposes into a set of strongly connected components (SCCs). Each SCC cor- responds to a smaller subproblem which can be solved independently of the others. Our results show that solving the constrained subproblems significantly improves the efficiency and scala- bility of heuristic search-based structure learning algorithms. Further, we show that by consider- ing only the top p POPS of each variable, we quickly find provably very high quality networks for large datasets.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-university14e,
  title = 	 {Firefly {M}onte {C}arlo: Exact {MCMC} with Subsets of Data},
  author =       {University, Dougal Maclaurin Harvard and Harvard, Ryan Adams},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {196--205},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/university14e/university14e.pdf},
  url = 	 {https://proceedings.mlr.press/r12/university14e.html},
  abstract = 	 {Markov chain Monte Carlo (MCMC) is a popular and successful general-purpose tool for Bayesian inference. However, MCMC cannot be practi- cally applied to large data sets because of the prohibitive cost of evaluating every likelihood term at every iteration. Here we present Fire- fly Monte Carlo (FlyMC) an auxiliary variable MCMC algorithm that only queries the likeli- hoods of a potentially small subset of the data at each iteration yet simulates from the exact pos- terior distribution, in contrast to recent propos- als that are approximate even in the asymptotic limit. FlyMC is compatible with a wide variety of modern MCMC algorithms, and only requires a lower bound on the per-datum likelihood fac- tors. In experiments, we find that FlyMC gen- erates samples from the posterior more than an order of magnitude faster than regular MCMC, opening up MCMC methods to larger datasets than were previously considered feasible.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-srivastava14a,
  title = 	 {First-Order Open-Universe POMDPs: Formulation and Algorithms},
  author =       {Srivastava, Siddharth and Ruan, Paul and Cheng, Xiang and Russell, Stuart},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {206--215},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/srivastava14a/srivastava14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/srivastava14a.html},
  abstract = 	 {Open-universe probability models, representable by a variety of probabilistic programming lan- guages (PPLs), handle uncertainty over the ex- istence and identity of objects—forms of uncer- tainty occurring in many real-world situations. We examine the problem of extending a declar- ative PPL to define decision problems (specifi- cally, POMDPs) and identify non-trivial repre- sentational issues in describing an agent’s ca- pability for observation and action—issues that were avoided in previous work only by making strong and restrictive assumptions. We present semantic definitions that lead to POMDP speci- fications provably consistent with the sensor and actuator capabilities of the agent. We also de- scribe a variant of point-based value iteration for solving open-universe POMDPs. Thus, we han- dle cases—such as seeing a new object and pick- ing it up—that could not previously be repre- sented or solved.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-university14f,
  title = 	 {Fast {N}ewton methods for the group fused lasso},
  author =       {University, Matt Wytock Carnegie Mellon and University, J. Zico Kolter Carnegie Mellon and University, Suvrit Sra Carnegie Mellon},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {216--225},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/university14f/university14f.pdf},
  url = 	 {https://proceedings.mlr.press/r12/university14f.html},
  abstract = 	 {We present a new algorithmic approach to the group fused lasso, a convex model that approx- imates a multi-dimensional signal via an ap- proximately piecewise-constant signal. This model has found many applications in mul- tiple change point detection, signal compres- sion, and total variation denoising, though existing algorithms typically using first-order or alternating minimization schemes. In this paper we instead develop a specialized pro- jected Newton method, combined with a pri- mal active set approach, which we show to be substantially faster that existing methods. Furthermore, we present two applications that use this algorithm as a fast subroutine for a more complex outer loop: segmenting linear regression models for time series data, and color image denoising. We show that on these problems the proposed method performs very well, solving the problems faster than state- of-the-art methods and to higher accuracy.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-university14g,
  title = 	 {Estimating Accuracy from Unlabeled Data},
  author =       {University, Emmanouil Antonios Platanios Carnegie Mellon and University, Avrim Blum Carnegie Mellon and University, Tom Mitchell Carnegie Mellon},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {226--235},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/university14g/university14g.pdf},
  url = 	 {https://proceedings.mlr.press/r12/university14g.html},
  abstract = 	 {We consider the question of how unlabeled data can be used to estimate the true accuracy of learned classifiers. This is an important question for any autonomous learning system that must es- timate its accuracy without supervision, and also when classifiers trained from one data distribu- tion must be applied to a new distribution (e.g., document classifiers trained on one text corpus are to be applied to a second corpus). We first show how to estimate error rates exactly from unlabeled data when given a collection of com- peting classifiers that make independent errors, based on the agreement rates between subsets of these classifiers. We further show that even when the competing classifiers do not make indepen- dent errors, both their accuracies and error de- pendencies can be estimated by making certain relaxed assumptions. Experiments on two data real-world data sets produce estimates within a few percent of the true accuracy, using solely un- labeled data. These results are of practical signif- icance in situations where labeled data is scarce and shed light on the more general question of how the consistency among multiple functions is related to their true accuracies.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-elkind14a,
  title = 	 {Electing the Most Probable Without Eliminating the Irrational: Voting Over Intransitive Domains},
  author =       {Elkind, Edith and University, Nisarg Shah Carnegie Mellon},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {236--245},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/elkind14a/elkind14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/elkind14a.html},
  abstract = 	 {Picking the best alternative in a given set is a well-studied problem at the core of social choice theory. In some applications, one can assume that there is an objectively correct way to compare the alternatives, which, however, cannot be ob- served directly, and individuals’ preferences over the alternatives (votes) are noisy estimates of this ground truth. The goal of voting in this case is to estimate the ground truth from the votes. In this paradigm, it is usually assumed that the ground truth is a ranking of the alternatives by their true quality. However, sometimes alterna- tives are compared using not one but multiple quality parameters, which may result in cycles in the ground truth as well as in the preferences of the individuals. Motivated by this, we provide a formal model of voting with possibly intransi- tive ground truth and preferences, and investigate the maximum likelihood approach for picking the best alternative in this case. We show that the resulting framework leads to polynomial-time al- gorithms, and also approximates the correspond- ing NP-hard problems in the classic framework.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-albrecht14a,
  title = 	 {On Convergence and Optimality of Best-Response Learning with Policy Types in Multiagent Systems},
  author =       {Albrecht, Stefano and Ramamoorthy, Subramanian},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {246--255},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/albrecht14a/albrecht14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/albrecht14a.html},
  abstract = 	 {While many multiagent algorithms are designed for homogeneous systems (i.e. all agents are iden- tical), there are important applications which re- quire an agent to coordinate its actions without knowing a priori how the other agents behave. One method to make this problem feasible is to as- sume that the other agents draw their latent policy (or type) from a specific set, and that a domain ex- pert could provide a specification of this set, albeit only a partially correct one. Algorithms have been proposed by several researchers to compute poste- rior beliefs over such policy libraries, which can then be used to determine optimal actions. In this paper, we provide theoretical guidance on two cen- tral design parameters of this method: Firstly, it is important that the user choose a posterior which can learn the true distribution of latent types, as otherwise suboptimal actions may be chosen. We analyse convergence properties of two existing posterior formulations and propose a new poste- rior which can learn correlated distributions. Sec- ondly, since the types are provided by an expert, they may be inaccurate in the sense that they do not predict the agents’ observed actions. We pro- vide a novel characterisation of optimality which allows experts to use efficient model checking al- gorithms to verify optimality of types.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-medicine14a,
  title = 	 {Efficient Inference of {G}aussian-Process-Modulated Renewal Processes with Application to Medical Event Data},
  author =       {Medicine, Thomas Lasko Vanderbilt School of},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {256--263},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/medicine14a/medicine14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/medicine14a.html},
  abstract = 	 {The episodic, irregular and asynchronous nature of medical data render them difficult substrates for standard machine learning algorithms. We would like to abstract away this difficulty for the class of time-stamped categorical variables (or events) by modeling them as a renewal pro- cess and inferring a probability density over non- parametric longitudinal intensity functions that modulate the process. Several methods exist for inferring such a density over intensity func- tions, but either their constraints prevent their use with our potentially bursty event streams, or their time complexity renders their use in- tractable on our long-duration observations of high-resolution events, or both. In this paper we present a new efficient and flexible infer- ence method that uses direct numeric integra- tion and smooth interpolation over Gaussian pro- cesses. We demonstrate that our direct method is up to twice as accurate and two orders of magni- tude more efficient than the best existing method (thinning). Importantly, our direct method can infer intensity functions over the full range of bursty to memoryless to regular events, which thinning and many other methods cannot do. Fi- nally, we apply the method to clinical event data and demonstrate a simple example application facilitated by the abstraction.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-university14h,
  title = 	 {Approximating the {B}ethe Partition Function},
  author =       {University, Adrian Weller Columbia and University, Tony Jebara Columbia},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {264--273},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/university14h/university14h.pdf},
  url = 	 {https://proceedings.mlr.press/r12/university14h.html},
  abstract = 	 {When belief propagation (BP) converges, it does so to a stationary point of the Bethe free en- ergy F, and is often strikingly accurate. How- ever, it may converge only to a local optimum or may not converge at all. An algorithm was recently introduced by Weller and Jebara for at- tractive binary pairwise MRFs which is guaran- teed to return an $\epsilon$-approximation to the global minimum of F in polynomial time provided the maximum degree $\Delta$= O(log n), where n is the number of variables. Here we extend their ap- proach and derive a new method based on an- alyzing first derivatives of F, which leads to much better performance and, for attractive mod- els, yields a fully polynomial-time approxima- tion scheme (FPTAS) without any degree restric- tion. Further, our methods apply to general (non- attractive) models, though with no polynomial time guarantee in this case, demonstrating that approximating log of the Bethe partition func- tion, log ZB = -min F, for a general model to additive $\epsilon$-accuracy may be reduced to a discrete MAP inference problem. This allows the merits of the global Bethe optimum to be tested.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-university14i,
  title = 	 {Understanding the {B}ethe approximation: when and how can it go wrong?},
  author =       {University, Adrian Weller Columbia and University, Kui Tang Columbia and University, Tony Jebara Columbia and University, David Sontag New York},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {274--283},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/university14i/university14i.pdf},
  url = 	 {https://proceedings.mlr.press/r12/university14i.html},
  abstract = 	 {Belief propagation is a remarkably effective tool for inference, even when applied to networks with cycles. It may be viewed as a way to seek the minimum of the Bethe free energy, though with no convergence guarantee in general. A variational perspective shows that, compared to exact inference, this minimization employs two forms of approximation: (i) the true entropy is approximated by the Bethe entropy, and (ii) the minimization is performed over a relaxation of the marginal polytope termed the local polytope. Here we explore when and how the Bethe ap- proximation can fail for binary pairwise models by examining each aspect of the approximation, deriving results both analytically and with new experimental methods.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-labs14b,
  title = 	 {Min-$d$-Occur: Ensuring Future Occurrences in Streaming Sets},
  author =       {Labs, Vidit Jain Yahoo and Galhotra, Sainyam},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {284--293},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/labs14b/labs14b.pdf},
  url = 	 {https://proceedings.mlr.press/r12/labs14b.html},
  abstract = 	 {Given a set of n elements and a corresponding stream of its subsets, we consider the problem of selecting k elements that should appear in at least d such subsets arriving in the “near” future with high probability. For this min-d- occur problem, we present an algorithm that provides a solution with the success proba- bility of at least 1 -O ( kd log n D + 1 n ) , where D is a known constant. Our empirical obser- vations on two streaming data sets show that this algorithm achieves high precision and re- call values. We further present a sliding win- dow adaptation of the proposed algorithm to provide a continuous selection of these ele- ments. In contrast to the existing work on predicting trends based on potential increase in popularity, our work focuses on a setting with provable guarantees.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-hci-iwr14a,
  title = 	 {Instance Label Prediction by {D}irichlet Process Multiple Instance Learning},
  author =       {HCI/IWR, Melih Kandemir Heidelberg University and HCI/IWR, Fred Hamprecht Heidelberg University},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {294--303},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/hci-iwr14a/hci-iwr14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/hci-iwr14a.html},
  abstract = 	 {We propose a generative Bayesian model that predicts instance labels from weak (bag-level) supervision. We solve this problem by simulta- neously modeling class distributions by Gaussian mixture models and inferring the class labels of positive bag instances that satisfy the multiple in- stance constraints. We employ Dirichlet process priors on mixture weights to automate model se- lection, and efficiently infer model parameters and positive bag instances by a constrained varia- tional Bayes procedure. Our method improves on the state-of-the-art of instance classification from weak supervision on 20 benchmark text catego- rization data sets and one histopathology cancer diagnosis data set.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-university14j,
  title = 	 {Transformation Based Probabilistic Clustering using Supervision},
  author =       {University, Siddharth Gopal Carnegie Mellon and University, Yiming Yang Carnegie Mellon},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {304--313},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/university14j/university14j.pdf},
  url = 	 {https://proceedings.mlr.press/r12/university14j.html},
  abstract = 	 {One of the common problems with clustering is that the generated clusters often do not match user expectations. This paper proposes a novel probabilistic framework that exploits supervised information in a discriminative and transferable manner to generate better clustering of unlabeled data. The supervision is provided by revealing the cluster assignments for some subset of the ground truth clusters and is used to learn a trans- formation of the data such that labeled instances form well-separated clusters with respect to the given clustering objective. This estimated trans- formation function enables us to fold the remain- ing unlabeled data into a space where new clus- ters hopefully match user expectations. While our framework is general, in this paper, we fo- cus on its application to Gaussian and von Mises- Fisher mixture models. Extensive testing on 23 data sets across several application domains re- vealed substantial improvement in performance over competing methods.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-china14a,
  title = 	 {Belief-Kinematics Jeffrey’s Rules  in the Theory of Evidence},
  author =       {China, Chunlai Zhou Renmin University of and University, Mingyue Wang Syracuse and China, Biao Qin Renmin University of},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {314--323},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/china14a/china14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/china14a.html},
  abstract = 	 {This paper studies the problem of revising belief- s using uncertain evidence in a framework where beliefs are represented by a belief function. We introduce two new Jeffrey’s rules for the revi- sion based on two forms of belief kinematics, an evidence-theoretic counterpart of probability kinematics. Furthermore, we provide two dis- tance measures for belief functions and show that the two belief kinematics are optimal in the sense that they minimize their corresponding distance measures.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-university14k,
  title = 	 {Inference Complexity in Continuous Time {B}ayesian Networks},
  author =       {University, Liessman Sturlaugson Montana State and University, John Sheppard Montana State},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {324--331},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/university14k/university14k.pdf},
  url = 	 {https://proceedings.mlr.press/r12/university14k.html},
  abstract = 	 {The continuous time Bayesian network (CTBN) enables temporal reasoning by rep- resenting a system as a factored, finite-state Markov process. The CTBN uses a tra- ditional Bayesian network (BN) to specify the initial distribution. Thus, the complex- ity results of Bayesian networks also apply to CTBNs through this initial distribution. However, the question remains whether prop- agating the probabilities through time is, by itself, also a hard problem. We show that exact and approximate inference in continu- ous time Bayesian networks is NP-hard even when the initial states are given.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-superieure14a,
  title = 	 {Bisimulation Metrics are Optimal Value Functions},
  author =       {Sup{\'e}rieure, Norm Ferns {\'E}cole Normale and Precup, Doina},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {332--341},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/superieure14a/superieure14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/superieure14a.html},
  abstract = 	 {Bisimulation is a notion of behavioural equiva- lence on the states of a transition system. Its defi- nition has been extended to Markov decision pro- cesses, where it can be used to aggregate states. A bisimulation metric is a quantitative analog of bisimulation that measures how similar states are from a the perspective of long-term behavior. Bisimulation metrics have been used to establish approximation bounds for state aggregation and other forms of value function approximation. In this paper, we prove that a bisimulation metric defined on the state space of a Markov decision process is the optimal value function of an opti- mal coupling of two copies of the original model. We prove the result in the general case of con- tinuous state spaces. This result has important implications in understanding the complexity of computing such metrics, and opens up the possi- bility of more efficient computational methods.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-bi14a,
  title = 	 {Learning to Predict from Crowdsourced Data},
  author =       {Bi, Wei and UIUC, Liwei Wang and Kwok, James and UCSD, Zhuowen Tu},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {342--351},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/bi14a/bi14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/bi14a.html},
  abstract = 	 {Crowdsourcing services like Amazon’s Mechan- ical Turk have facilitated and greatly expedited the manual labeling process from a large number of human workers. However, spammers are often unavoidable and the crowdsourced labels can be very noisy. In this paper, we explicitly account for four sources for a noisy crowdsourced label: worker’s dedication to the task, his/her expertise, his/her default labeling judgement, and sample difficulty. A novel mixture model is employed for worker annotations, which learns a prediction model directly from samples to labels for effi- cient out-of-sample testing. Experiments on both simulated and real-world crowdsourced data sets show that the proposed method achieves signifi- cant improvements over the state-of-the-art.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-panigrahy14a,
  title = 	 {Optimal amortized regret in every interval},
  author =       {Panigrahy, Rina and Popat, Preyas},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {352--360},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/panigrahy14a/panigrahy14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/panigrahy14a.html},
  abstract = 	 {Consider the classical problem of predicting the next bit in a sequence of bits. A standard performance measure is regret (loss in payoff) with respect to a set of experts. For exam- ple if we measure performance with respect to two constant experts one that always predicts 0’s and another that always predicts 1’s it is well known that one can get regret O( $\sqrt{}$ T) with respect to the best expert by using, say, the weighted majority algorithm [LW89]. But this algorithm does not provide performance guaran- tee in any interval. There are other algorithms (see [BM07, FSSW97, Vov99]) that ensure regret O($\sqrt{}$x log T) in any interval of length x. In this paper we show a randomized algorithm that in an amortized sense gets a regret of O($\sqrt{}$x) for any interval when the sequence is partitioned into in- tervals arbitrarily. We empirically estimated the constant in the O() for T upto 2000 and found it to be small – around 2.1. We also experimentally evaluate the efficacy of this algorithm in predict- ing high frequency stock data. $*$This work was done while this author was at Microsoft Re- search.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-university14l,
  title = 	 {Saturated Conditional Independence with Fixed and Undetermined Sets of Incomplete Random Variables},
  author =       {University, Henning Koehler Massey and Link, Sebastian},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {361--370},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/university14l/university14l.pdf},
  url = 	 {https://proceedings.mlr.press/r12/university14l.html},
  abstract = 	 {The implication problem for saturated condi- tional independence statements is studied in the presence of fixed and undetermined sets of in- complete random variables. Here, random vari- ables are termed incomplete since they admit missing data. Two different notions of implica- tion arise. In the classic notion of V -implication, a statement is implied jointly by a set of state- ments and a fixed set V of random variables. In the alternative notion of pure implication, a statement is implied by a given set of state- ments alone, leaving the set of random vari- ables undetermined. A first axiomatization for V -implication is established that distinguishes purely implied from V -implied statements. Ax- iomatic, algorithmic and logical characteriza- tions of pure implication are established. Pure implication appeals to applications in which the existence of random variables is uncertain, for example, when independence statements are in- tegrated from different sources, when random variables are unknown or shall remain hidden.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-naist14a,
  title = 	 {Latent {K}ullback {L}eibler Control for Continuous-State Systems using Probabilistic Graphical Models},
  author =       {NAIST, Takamitsu Matsubara and Nijmegen, Vicen{\c{c}} G{\'o}mez Radboud University and University, Hilbert Kappen Radboud},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {371--380},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/naist14a/naist14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/naist14a.html},
  abstract = 	 {Kullback Leibler (KL) control problems al- low for efficient computation of optimal con- trol by solving a principal eigenvector prob- lem. However, direct applicability of such framework to continuous state-action sys- tems is limited. In this paper, we propose to embed a KL control problem in a proba- bilistic graphical model where observed vari- ables correspond to the continuous (possi- bly high-dimensional) state of the system and latent variables correspond to a dis- crete (low-dimensional) representation of the state amenable for KL control computation. We present two examples of this approach. The first one uses standard hidden Markov models (HMMs) and computes exact opti- mal control, but is only applicable to low- dimensional systems. The second one uses factorial HMMs, it is scalable to higher di- mensional problems, but control computa- tion is approximate. We illustrate both ex- amples in several robot motor control tasks.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-huji14a,
  title = 	 {{HELM}: Highly Efficient Learning of Mixed copula networks},
  author =       {Huji, Yaniv Tenzer and Elidan, Gal},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {381--390},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/huji14a/huji14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/huji14a.html},
  abstract = 	 {Learning the structure of probabilistic graphi- cal models for complex real-valued domains is a formidable computational challenge. This in- evitably leads to significant modelling compro- mises such as discretization or the use of a sim- plistic Gaussian representation. In this work we address the challenge of efficiently learning truly expressive copula-based networks that facilitate a mix of varied copula families within the same model. Our approach is based on a simple but powerful bivariate building block that is used to highly efficiently perform local model selection, thus bypassing much of computational burden in- volved in structure learning. We show how this building block can be used to learn general net- works and demonstrate its effectiveness on var- ied and sizeable real-life domains. Importantly, favorable identification and generalization per- formance come with dramatic runtime improve- ments. Indeed, the benefits are such that they allow us to tackle domains that are prohibitive when using a standard learning approaches.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-nuance14a,
  title = 	 {Lifted Tree-Reweighted Variational Inference},
  author =       {Nuance, Hung Bui and City, Tuyen Huynh Jon von Neumann Institute Vietnam National University Ho Chi Minh and University, David Sontag New York},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {391--400},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/nuance14a/nuance14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/nuance14a.html},
  abstract = 	 {We analyze variational inference for highly sym- metric graphical models such as those arising from first-order probabilistic models. We first show that for these graphical models, the tree- reweighted variational objective lends itself to a compact lifted formulation which can be solved much more efficiently than the standard TRW formulation for the ground graphical model. Compared to earlier work on lifted belief prop- agation, our formulation leads to a convex op- timization problem for lifted marginal inference and provides an upper bound on the partition function. We provide two approaches for im- proving the lifted TRW upper bound. The first is a method for efficiently computing maxi- mum spanning trees in highly symmetric graphs, which can be used to optimize the TRW edge ap- pearance probabilities. The second is a method for tightening the relaxation of the marginal poly- tope using lifted cycle inequalities and novel ex- changeable cluster consistency constraints.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-labo14a,
  title = 	 {Generating structure of latent variable models for nested data},
  author =       {Labo, Masakazu Ishihata NTT Communication Science and Laboratories, Tomoharu Iwata NTT Communication Science},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {401--410},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/labo14a/labo14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/labo14a.html},
  abstract = 	 {Probabilistic latent variable models have been successfully used to capture intrinsic character- istics of various data. However, it is nontrivial to design appropriate models for given data because it requires both machine learning and domain- specific knowledge. In this paper, we focus on data with nested structure and propose a method to automatically generate a latent variable model for the given nested data, with the proposed method, the model structure is adjustable by its structural parameters. Our model can represent a wide class of hierarchical and sequential la- tent variable models including mixture models, latent Dirichlet allocation, hidden Markov mod- els and their combinations in multiple layers of the hierarchy. Even when deeply-nested data are given, where designing a proper model is diffi- cult even for experts, our method generate an ap- propriate model by extracting the essential infor- mation. We present an efficient variational in- ference method for our model based on dynamic programming on the given data structure. We ex- perimentally show that our method generates cor- rect models from artificial datasets and demon- strate that models generated by our method can extract hidden structures of blog and news article datasets.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-uiuc14a,
  title = 	 {Batch-Mode Active Learning via Error Bound Minimization},
  author =       {UIUC, Quanquan Gu CS and University, Tong Zhang Rutgers and UIUC, Jiawei Han CS},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {411--420},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/uiuc14a/uiuc14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/uiuc14a.html},
  abstract = 	 {Active learning has been proven to be quite effec- tive in reducing the human labeling efforts by ac- tively selecting the most informative examples to label. In this paper, we present a batch-mode ac- tive learning method based on logistic regression. Our key motivation is an out-of-sample bound on the estimation error of class distribution in lo- gistic regression conditioned on any fixed train- ing sample. It is different from a typical PAC- style passive learning error bound, that relies on the i.i.d. assumption of example-label pairs. In addition, it does not contain the class labels of the training sample. Therefore, it can be imme- diately used to design an active learning algo- rithm by minimizing this bound iteratively. We also discuss the connections between the pro- posed method and some existing active learn- ing approaches. Experiments on benchmark UCI datasets and text datasets demonstrate that the proposed method outperforms the state-of-the-art active learning methods significantly.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-kishimoto14a,
  title = 	 {Recursive Best-First {AND}/{OR} Search for Graphical Models},
  author =       {Kishimoto, Akihiro and Marinescu, Radu},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {421--430},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/kishimoto14a/kishimoto14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/kishimoto14a.html},
  abstract = 	 {The paper presents and evaluates the power of limited memory best-first search over AND/OR spaces for optimization tasks in graphical mod- els. We propose Recursive Best-First AND/OR Search with Overestimation (RBFAOO), a new algorithm that explores the search space in a best-first manner while operating with restricted memory. We enhance RBFAOO with a simple overestimation technique aimed at minimizing the overhead associated with re-expanding inter- nal nodes and prove correctness and complete- ness of RBFAOO. Our experiments show that RBFAOO is often superior to the current state- of-the-art approaches based on AND/OR search, especially on very hard problem instances.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-iyer14a,
  title = 	 {A Unified Approach to Fast Algorithms for Submodular Optimization based on Continuous Relaxations and Rounding},
  author =       {Iyer, Rishabh and Jegelka, Stefanie and Bilmes, Jeffrey},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {431--440},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/iyer14a/iyer14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/iyer14a.html},
  abstract = 	 {It is becoming increasingly evident that many ma- chine learning problems may be reduced to sub- modular optimization. Previous work addresses generic discrete approaches and specific relax- ations. In this work, we take a generic view from a relaxation perspective. We show a relaxation formulation and simple rounding strategy that, based on the monotone closure of relaxed con- straints, reveals analogies between minimization and maximization problems, and includes known results as special cases and extends to a wider range of settings. Our resulting approximation factors match the corresponding integrality gaps. For submodular maximization, a number of relax- ation approaches have been proposed. A critical challenge for the practical applicability of these techniques, however, is the complexity of evaluat- ing the multilinear extension. We show that this extension can be efficiently evaluated for a num- ber of useful submodular functions, thus making these otherwise impractical algorithms viable for real-world machine learning problems.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-university14m,
  title = 	 {Efficient Sparse Recovery via Adaptive Non-Convex Regularizers with Oracle Property},
  author =       {University, Ming Lin Tsinghua and University, Rong Jin Michigan State and University, Changshui Zhang Tsinghua},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {441--450},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/university14m/university14m.pdf},
  url = 	 {https://proceedings.mlr.press/r12/university14m.html},
  abstract = 	 {The main shortcoming of sparse recovery with a convex regularizer is that it is a biased esti- mator and therefore will result in a suboptimal performance in many cases. Recent studies have shown, both theoretically and empirically, that non-convex regularizer is able to overcome the biased estimation problem. Although multiple algorithms have been developed for sparse recov- ery with non-convex regularization, they are ei- ther computationally demanding or not equipped with the desired properties (i.e. optimal recovery error, selection consistency and oracle property). In this work, we develop an algorithm for effi- cient sparse recovery based on proximal gradient descent. The key feature of the proposed algo- rithm is introducing adaptive non-convex regu- larizers whose shrinking threshold vary over it- erations. The algorithm is compatible with most popular non-convex regularizers, achieves a ge- ometric convergence rate for the recovery er- ror, is selection consistent, and most importantly has the oracle property. Based on the proposed framework, we suggest to use a so–called ACCQ regularizer, which is equivalent to zero proximal projection gap adaptive hard-thresholding. Ex- periments with both synthetic data sets and real images verify both the efficiency and effective- ness of the proposed method compared to the state-of-the-art methods for sparse recovery.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-belfast14a,
  title = 	 {Can(Plan)+: Extending the Operational Semantics for the {BDI} architecture to deal with Uncertain Information},
  author =       {Belfast, Kim Bauters Queen's University and Belfast, Weiru Liu Queen's University and Belfast, Jun Hong Queen's University and CSIC, Carles Sierra IIIA and Institute, Lluis Godo Artificial Intelligence Research},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {451--460},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/belfast14a/belfast14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/belfast14a.html},
  abstract = 	 {The BDI architecture, where agents are modelled based on their beliefs, desires and intentions, pro- vides a practical approach to develop large scale systems. However, it is not well suited to model complex Supervisory Control And Data Acquisi- tion (SCADA) systems pervaded by uncertainty. In this paper we address this issue by extending the operational semantics of CAN(PLAN) into CAN(PLAN)+. We start by modelling the beliefs of an agent as a set of epistemic states where each state, possibly using a different representation, models part of the agent’s beliefs. These epis- temic states are stratified to make them commen- surable and to reason about the uncertain beliefs of the agent. The syntax and semantics of a BDI agent are extended accordingly and we identify fragments with computationally efficient seman- tics. Finally, we examine how primitive actions are affected by uncertainty and we define an ap- propriate form of lookahead planning.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-arvaniti14a,
  title = 	 {{M}arkov Network Structure Learning via Ensemble-of-Forests Models},
  author =       {Arvaniti, Eirini and Claassen, Manfred},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {461--470},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/arvaniti14a/arvaniti14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/arvaniti14a.html},
  abstract = 	 {Real world systems typically feature a variety of different dependency types and topologies that complicate model selection for probabilistic graphical models. We introduce the ensemble-of- forests model, a generalization of the ensemble- of-trees model of Meil{ă} and Jaakkola (2006). Our model enables structure learning of Markov random fields (MRF) with multiple connected components and arbitrary potentials. We present two approximate inference techniques for this model and demonstrate their performance on synthetic data. Our results suggest that the ensemble-of-forests approach can accurately re- cover sparse, possibly disconnected MRF topolo- gies, even in presence of non-Gaussian depen- dencies and/or low sample size. We applied the ensemble-of-forests model to learn the struc- ture of perturbed signaling networks of immune cells and found that these frequently exhibit non-Gaussian dependencies with disconnected MRF topologies. In summary, we expect that the ensemble-of-forests model will enable MRF structure learning in other high dimensional real world settings that are governed by non-trivial dependencies.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-lu14a,
  title = 	 {Fast Ridge Regression with Randomized Principal Component Analysis and Gradient Descent},
  author =       {Lu, Yichao and Foster, Dean},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {471--478},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/lu14a/lu14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/lu14a.html},
  abstract = 	 {We propose a new two stage algorithm LING for large scale regression problems. LING has the same risk as the well known Ridge Regres- sion under the fixed design setting and can be computed much faster. Our experiments have shown that LING performs well in terms of both prediction accuracy and computational efficiency compared with other large scale regression al- gorithms like Gradient Descent, Stochastic Gra- dient Descent and Principal Component Regres- sion on both simulated and real datasets.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-tran-thanh14a,
  title = 	 {Efficient Regret Bounds for Online Bid Optimisation in Budget-Limited Sponsored Search Auctions},
  author =       {Tran-Thanh, Long and Stavrogiannis, Lampros and Naroditskiy, Victor and Robu, Valentin and Jennings, Nicholas and Key, Peter},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {479--488},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/tran-thanh14a/tran-thanh14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/tran-thanh14a.html},
  abstract = 	 {We study the problem of an advertising agent who needs to intelligently distribute her bud- get across a sequence of online keyword bid- ding auctions. We assume the closing price of each auction is governed by the same un- known distribution, and study the problem of making provably optimal bidding deci- sions. Learning the distribution is done un- der censored observations, i.e. the closing price of an auction is revealed only if the bid we place is above it. We consider three al- gorithms, namely $\varepsilon$-First, Greedy Product- Limit (GPL) and LuekerLearn, respectively, and we show that these algorithms provably achieve Hannan-consistency. In particular, we show that the regret bound of $\varepsilon$-First is at most O(T 2 3 ) with high probability. For the other two algorithms, we first prove that, by using a censored data distribution esti- mator proposed by Zeng [19], the empirical distribution of the closing market price con- verges in probability to its true distribution with a O( 1 $\sqrt{}$ t) rate, where t is the number of updates. Based on this result, we prove that both GPL and LuekerLearn achieve O( $\sqrt{}$ T) regret bound with high probability. This in fact provides an affirmative answer to the re- search question raised in [1]. We also evalu- ate the abovementioned algorithms using real bidding data, and show that although GPL achieves the best performance on average (up to 90% of the optimal solution), its long run- ning time may limit its suitability in practice. By contrast, LuekerLearn and $\varepsilon$-First pro- posed in this paper achieve up to 85% of the optimal, but with an exponential reduction in computational complexity (a saving up to 95%, compared to GPL).},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-university14n,
  title = 	 {Correlated Compressive Sensing for Networked Data},
  author =       {University, Tianlin Shi Tsinghua and Tang, Da and Xu, Liwen and Moscibroda, Thomas},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {489--498},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/university14n/university14n.pdf},
  url = 	 {https://proceedings.mlr.press/r12/university14n.html},
  abstract = 	 {We consider the problem of recovering sparse correlated data on networks. To improve accu- racy and reduce costs, it is strongly desirable to take the potentially useful side-information of network structure into consideration. In this pa- per we present a novel correlated compressive sensing method called CorrCS for networked data. By naturally extending Bayesian compres- sive sensing, we extract correlations from net- work topology and encode them into a graphical model as prior. Then we derive posterior infer- ence algorithms for the recovery of jointly sparse and correlated networked data. First, we design algorithms to recover the data based on pairwise correlations between neighboring nodes in the network. Next, we generalize this model through a diffusion process to capture higher-order cor- relations. Both real-valued and binary data are considered. Our models are extensively tested on several real datasets from social and sensor networks and are shown to outperform baseline compressive sensing models in terms of recovery performance.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-ma14a,
  title = 	 {Adaptive Monotone Shrinkage for Regression},
  author =       {Ma, Zhuang and Foster, Dean and Stine, Robert},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {499--508},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/ma14a/ma14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/ma14a.html},
  abstract = 	 {We develop an adaptive monotone shrinkage es- timator for regression models with the following characteristics: i) dense coefficients with small but important effects; ii) a priori ordering that in- dicates the probable predictive importance of the features. We capture both properties with an em- pirical Bayes estimator that shrinks coefficients monotonically with respect to their anticipated importance. This estimator can be rapidly com- puted using a version of Pool-Adjacent-Violators algorithm. We show that the proposed monotone shrinkage approach is competitive with the class of all Bayesian estimators that share the prior in- formation. We further observe that the estima- tor also minimizes Stein’s unbiased risk estimate. Along with our key result that the estimator mim- ics the oracle Bayes rule under an order assump- tion, we also prove that the estimator is robust. Even without the order assumption, our estima- tor mimics the best performance of a large family of estimators that includes the least squares es- timator, constant-$\lambda$ ridge estimator, James-Stein estimator, etc. All the theoretical results are non- asymptotic. Simulation results and data analysis from a model for text processing are provided to support the theory.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-garnett14a,
  title = 	 {Active Learning of Linear Embeddings for {G}aussian Processes},
  author =       {Garnett, Roman and Osborne, Michael and Hennig, Philipp},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {509--518},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/garnett14a/garnett14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/garnett14a.html},
  abstract = 	 {We propose an active learning method for discovering low-dimensional structure in high- dimensional Gaussian process (GP) tasks. Such problems are increasingly frequent and impor- tant, but have hitherto presented severe practical difficulties. We further introduce a novel tech- nique for approximately marginalizing GP hyper- parameters, yielding marginal predictions robust to hyperparameter misspecification. Our method offers an efficient means of performing GP re- gression, quadrature, or Bayesian optimization in high-dimensional spaces.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-university14o,
  title = 	 {Asymptotically Exact, Embarrassingly Parallel {MCMC}},
  author =       {University, Willie Neiswanger Carnegie Mellon and University, Eric Xing Carnegie Mellon and Wang, Chong},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {519--528},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/university14o/university14o.pdf},
  url = 	 {https://proceedings.mlr.press/r12/university14o.html},
  abstract = 	 {Communication costs, resulting from synchro- nization requirements during learning, can greatly slow down many parallel machine learning algorithms. In this paper, we present a parallel Markov chain Monte Carlo (MCMC) algorithm in which subsets of data are pro- cessed independently, with very little com- munication. First, we arbitrarily partition data onto multiple machines. Then, on each machine, any classical MCMC method (e.g., Gibbs sampling) may be used to draw samples from a posterior distribution given the data subset. Finally, the samples from each ma- chine are combined to form samples from the full posterior. This embarrassingly parallel algorithm allows each machine to act inde- pendently on a subset of the data (without communication) until the final combination stage. We prove that our algorithm generates asymptotically exact samples and empirically demonstrate its ability to parallelize burn-in and sampling in several models.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-ict14a,
  title = 	 {Position-Aware ListMLE: A Sequential Learning Process for Ranking},
  author =       {ICT, Yanyan Lan and ICT, Yadong Zhu and ICT, Jiafeng Guo and ICT, Shuzi Niu and ICT, Xueqi Cheng},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {529--538},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/ict14a/ict14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/ict14a.html},
  abstract = 	 {Communication costs, resulting from synchro- nization requirements during learning, can greatly slow down many parallel machine learning algorithms. In this paper, we present a parallel Markov chain Monte Carlo (MCMC) algorithm in which subsets of data are pro- cessed independently, with very little com- munication. First, we arbitrarily partition data onto multiple machines. Then, on each machine, any classical MCMC method (e.g., Gibbs sampling) may be used to draw samples from a posterior distribution given the data subset. Finally, the samples from each ma- chine are combined to form samples from the full posterior. This embarrassingly parallel algorithm allows each machine to act inde- pendently on a subset of the data (without communication) until the final combination stage. We prove that our algorithm generates asymptotically exact samples and empirically demonstrate its ability to parallelize burn-in and sampling in several models.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-holloway14a,
  title = 	 {Venn-Abers Predictors},
  author =       {Holloway, Vladimir Vovk Royal and Petej, Ivan},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {539--548},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/holloway14a/holloway14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/holloway14a.html},
  abstract = 	 {ListMLE is a state-of-the-art listwise learning-to- rank algorithm, which has been shown to work very well in application. It defines the probabil- ity distribution based on Plackett-Luce Model in a top-down style to take into account the position information. However, both empirical contradic- tion and theoretical results indicate that ListM- LE cannot well capture the position importance, which is a key factor in ranking. To amend the problem, this paper proposes a new listwise rank- ing method, called position-aware ListMLE (p- ListMLE for short). It views the ranking prob- lem as a sequential learning process, with each step learning a subset of parameters which maxi- mize the corresponding stepwise probability dis- tribution. To solve this sequential multi-objective optimization problem, we propose to use lin- ear scalarization strategy to transform it into a single-objective optimization problem, which is efficient for computation. Our theoretical s- tudy shows that p-ListMLE is better than ListM- LE in statistical consistency with respect to typi- cal ranking evaluation measure NDCG. Further- more, our experiments on benchmark datasets demonstrate that the proposed method can sig- nificantly improve the performance of ListMLE and outperform state-of-the-art listwise learning- to-rank algorithms as well.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-upc14a,
  title = 	 {Metrics for Probabilistic Geometry},
  author =       {UPC, Alessandra Tosi and Denmar, S{\o}ren Hauberg Technical University of and UPC, Alfredo Vellido and Lawrence, Neil},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {549--557},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/upc14a/upc14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/upc14a.html},
  abstract = 	 {We investigate the geometrical structure of probabilistic generative dimensionality reduction models using the tools of Riemannian geometry. We explicitly define a distribution over the natu- ral metric given by the models. We provide the necessary algorithms to compute expected metric tensors where the distribution over mappings is given by a Gaussian process. We treat the corre- sponding latent variable model as a Riemannian manifold and we use the expectation of the met- ric under the Gaussian process prior to define in- terpolating paths and measure distance between latent points. We show how distances that respect the expected metric lead to more appropriate gen- eration of new data.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-aravkin14a,
  title = 	 {A  variational approach to stable principal component pursuit},
  author =       {Aravkin, Aleksandr and Becker, Stephen and Cevher, Volkan and Olsen, Peder},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {558--567},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/aravkin14a/aravkin14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/aravkin14a.html},
  abstract = 	 {We introduce a new convex formulation for stable principal component pursuit (SPCP) to decompose noisy signals into low-rank and sparse representations. For numerical solu- tions of our SPCP formulation, we first de- velop a convex variational framework and then accelerate it with quasi-Newton meth- ods. We show, via synthetic and real data experiments, that our approach offers advan- tages over the classical SPCP formulations in scalability and practical parameter selection.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-college14a,
  title = 	 {Model Regularization for Stable Sample Rollouts},
  author =       {College, Erik Talvitie Franklin \& Marshall},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {568--577},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/college14a/college14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/college14a.html},
  abstract = 	 {When an imperfect model is used to generate sample rollouts, its errors tend to compound – a flawed sample is given as input to the model, which causes more errors, and so on. This presents a barrier to applying rollout-based plan- ning algorithms to learned models. To ad- dress this issue, a training methodology called “hallucinated replay” is introduced, which adds samples from the model into the training data, thereby training the model to produce sensible predictions when its own samples are given as input. Capabilities and limitations of this ap- proach are studied empirically. In several exam- ples hallucinated replay allows effective planning with imperfect models while models trained us- ing only real experience fail dramatically.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-campbell14a,
  title = 	 {Approximate Decentralized {B}ayesian Inference},
  author =       {Campbell, Trevor and How, Jonathan},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {578--587},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/campbell14a/campbell14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/campbell14a.html},
  abstract = 	 {This paper presents an approximate method for performing Bayesian inference in models with conditional independence over a decentralized network of learning agents. The method first employs variational inference on each individual learning agent to generate a local approximate posterior, the agents transmit their local poste- riors to other agents in the network, and finally each agent combines its set of received local pos- teriors. The key insight in this work is that, for many Bayesian models, approximate inference schemes destroy symmetry and dependencies in the model that are crucial to the correct appli- cation of Bayes’ rule when combining the lo- cal posteriors. The proposed method addresses this issue by including an additional optimization step in the combination procedure that accounts for these broken dependencies. Experiments on synthetic and real data demonstrate that the de- centralized method provides advantages in com- putational performance and predictive test likeli- hood over previous batch and distributed meth- ods.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-landrieu14a,
  title = 	 {Continuously indexed Potts models on unoriented graphs},
  author =       {Landrieu, Loic and ParisTech, Guillaume Obozinski Ecole des Ponts},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {588--597},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/landrieu14a/landrieu14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/landrieu14a.html},
  abstract = 	 {This paper introduces an extension to undirected graphical models of the classical continuous time Markov chains. This model can be used to solve a transductive or unsupervised multi-class classi- fication problem at each point of a network de- fined as a set of nodes connected by segments of different lengths. The classification is performed not only at the nodes, but at every point of the edge connecting two nodes. This is achieved by constructing a Potts process indexed by the con- tinuum of points forming the edges of the graph. We propose a homogeneous parameterization which satisfies Kolmogorov consistency, and show that classical inference and learning algo- rithms can be applied. We then apply our model to a problem from geo- matics, namely that of labelling city blocks auto- matically with a simple typology of classes (e.g. collective housing) from simple properties of the shape and sizes of buildings of the blocks. Our experiments shows that our model outperform standard MRFs and a discriminative model like logistic regression.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-meeds14a,
  title = 	 {{GPS}-{ABC}: {G}aussian Process Surrogate Approximate {B}ayesian Computation},
  author =       {Meeds, Edward and Welling, Max},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {598--607},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/meeds14a/meeds14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/meeds14a.html},
  abstract = 	 {Scientists often express their understanding of the world through a computationally demand- ing simulation program. Analyzing the posterior distribution of the parameters given observations (the inverse problem) can be extremely chal- lenging. The Approximate Bayesian Computa- tion (ABC) framework is the standard statisti- cal tool to handle these likelihood free problems, but they require a very large number of simula- tions. In this work we develop two new ABC sampling algorithms that significantly reduce the number of simulations necessary for posterior in- ference. Both algorithms use confidence esti- mates for the accept probability in the Metropo- lis Hastings step to adaptively choose the number of necessary simulations. Our GPS-ABC algo- rithm stores the information obtained from every simulation in a Gaussian process which acts as a surrogate function for the simulated statistics. Experiments on a challenging realistic biologi- cal problem illustrate the potential of these algo- rithms.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-grizou14a,
  title = 	 {Interactive Learning from Unlabelled Instructions},
  author =       {Grizou, Jonathan and de Zaragoza, Luis Montesano Universidad and Iturrate, I{\~n}aki and Lopes, Manuel},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {608--617},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/grizou14a/grizou14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/grizou14a.html},
  abstract = 	 {Interactive learning deals with the problem of learning and solving tasks using human instruc- tions. It is common in human-robot interac- tion, tutoring systems, and in human-computer interfaces such as brain-computer ones. In most cases, learning these tasks is possible because the signals are predefined or an ad-hoc calibra- tion procedure allows to map signals to specific meanings. In this paper, we address the problem of simultaneously solving a task under human feedback and learning the associated meanings of the feedback signals. This has important practi- cal application since the user can start controlling a device from scratch, without the need of an ex- pert to define the meaning of signals or carrying out a calibration phase. The paper proposes an algorithm that simultaneously assign meanings to signals while solving a sequential task under the assumption that both, human and machine, share the same a priori on the possible instruc- tion meanings and the possible tasks. Further- more, we show using synthetic and real EEG data from a brain-computer interface that taking into account the uncertainty of the task and the signal is necessary for the machine to actively plan how to solve the task efficiently.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-salamatian14a,
  title = 	 {{SPPM}: Sparse Privacy Preserving Mappings},
  author =       {Salamatian, Salman and Technicolor, Nadia Fawaz and Labs, Branislav Kveton Technicolor and Technicolor, Nina Taft},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {618--627},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/salamatian14a/salamatian14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/salamatian14a.html},
  abstract = 	 {We study the problem of a user who has both public and private data, and wants to re- lease the public data, e.g. to a recommenda- tion service, yet simultaneously wants to pro- tect his private data from being inferred via big data analytics. This problem has previ- ously been formulated as a convex optimiza- tion problem with linear constraints where the objective is to minimize the mutual in- formation between the private and released data. This attractive formulation faces a challenge in practice because when the un- derlying alphabet of the user profile is large, there are too many potential ways to distort the original profile. We address this funda- mental scalability challenge. We propose to generate sparse privacy-preserving mappings by recasting the problem as a sequence of lin- ear programs and solving each of these in- crementally using an adaptation of Dantzig- Wolfe decomposition. We evaluate our ap- proach on several datasets and demonstrate that nearly optimal privacy-preserving map- pings can be learned quickly even at scale.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-university14p,
  title = 	 {Nonparametric Clustering with Distance Dependent Hierarchies},
  author =       {University, Soumya Ghosh Brown and Raptis, Michalis and Research, Leonid Sigal Disney and University, Erik Sudderth Brown},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {628--637},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/university14p/university14p.pdf},
  url = 	 {https://proceedings.mlr.press/r12/university14p.html},
  abstract = 	 {The distance dependent Chinese restaurant pro- cess (ddCRP) provides a flexible framework for clustering data with temporal, spatial, or other structured dependencies. Here we model mul- tiple groups of structured data, such as pixels within frames of a video sequence, or paragraphs within documents from a text corpus. We pro- pose a hierarchical generalization of the ddCRP which clusters data within groups based on dis- tances between data items, and couples clusters across groups via distances based on aggregate properties of these local clusters. Our hddCRP model subsumes previously proposed hierarchi- cal extensions to the ddCRP, and allows more flexibility in modeling complex data. This flexi- bility poses a challenging inference problem, and we derive a MCMC method that makes coordi- nated changes to data assignments both within and between local clusters. We demonstrate the effectiveness of our hddCRP on video segmenta- tion and discourse modeling tasks, achieving re- sults competitive with state-of-the-art methods.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-university14q,
  title = 	 {Modeling Citation Networks using Latent Random Offsets},
  author =       {University, Willie Neiswanger Carnegie Mellon and Wang, Chong and University, Qirong Ho Carnegie Mellon and University, Eric Xing Carnegie Mellon},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {638--647},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/university14q/university14q.pdf},
  url = 	 {https://proceedings.mlr.press/r12/university14q.html},
  abstract = 	 {Out of the many potential factors that deter- mine which links form in a document citation network, two in particular are of high impor- tance: first, a document may be cited based on its subject matter—this can be modeled by analyzing document content; second, a doc- ument may be cited based on which other documents have previously cited it—this can be modeled by analyzing citation structure. Both factors are important for users to make informed decisions and choose appropriate ci- tations as the network grows. In this paper, we present a novel model that integrates the merits of content and citation analyses into a single probabilistic framework. We demon- strate our model on three real-world citation networks. Compared with existing baselines, our model can be used to effectively explore a citation network and provide meaningful explanations for links while still maintaining competitive citation prediction performance.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-levine14a,
  title = 	 {Quantifying Nonlocal Informativeness in High-Dimensional, Loopy {G}aussian Graphical Models},
  author =       {Levine, Daniel and How, Jonathan},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {648--656},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/levine14a/levine14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/levine14a.html},
  abstract = 	 {We consider the problem of selecting informative observations in Gaussian graphical models con- taining both cycles and nuisances. More specif- ically, we consider the subproblem of quantify- ing conditional mutual information measures that are nonlocal on such graphs. The ability to effi- ciently quantify the information content of obser- vations is crucial for resource-constrained data acquisition (adaptive sampling) and data process- ing (active learning) systems. While closed- form expressions for Gaussian mutual informa- tion exist, standard linear algebraic techniques, with complexity cubic in the network size, are in- tractable for high-dimensional distributions. We investigate the use of embedded trees for com- puting nonlocal pairwise mutual information and demonstrate through numerical simulations that the presented approach achieves a significant re- duction in computational cost over inversion- based methods.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-belanger14a,
  title = 	 {Message Passing for Soft Constraint Dual Decomposition},
  author =       {Belanger, David and Passos, Alexandre and Riedel, Sebastian and McCallum, Andrew},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {657--666},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/belanger14a/belanger14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/belanger14a.html},
  abstract = 	 {Dual decomposition provides the opportunity to build complex, yet tractable, structured predic- tion models using linear constraints to link to- gether submodels that have available MAP infer- ence routines. However, since some constraints might not hold on every single example, such models can often be improved by relaxing the requirement that these constraints always hold, and instead replacing them with soft constraints that merely impose a penalty if violated. A dual objective for the resulting MAP inference prob- lem differs from the hard constraint problem’s associated dual decomposition objective only in that the dual variables are subject to box con- straints. This paper introduces a novel primal- dual block coordinate descent algorithm for min- imizing this general family of box-constrained objectives. Through experiments on two nat- ural language corpus-wide inference tasks, we demonstrate the advantages of our approach over the current alternative, based on copying vari- ables, adding auxiliary submodels and using tra- ditional dual decomposition. Our algorithm per- forms inference in the same model as was previ- ously published for these tasks, and thus is capa- ble of achieving the same accuracy, but provides a 2-10x speedup over the current state of the art.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-university14r,
  title = 	 {Learning Partial Policies to Speedup {MDP} Tree Search},
  author =       {University, Jervis Pinto Oregon State and University, Alan Fern Oregon State},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {667--676},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/university14r/university14r.pdf},
  url = 	 {https://proceedings.mlr.press/r12/university14r.html},
  abstract = 	 {A popular approach for online decision making in large MDPs is time-bounded tree search. The effectiveness of tree search, however, is largely influenced by the action branching factor, which limits the search depth given a time bound. An obvious way to reduce action branching is to consider only a subset of potentially good ac- tions at each state as specified by a provided partial policy. In this work, we consider offline learning of such partial policies with the goal of speeding up search without significantly reduc- ing decision-making quality. Our first contribu- tion is to study learning algorithms based on re- ducing our learning problem to i.i.d. supervised learning. We give a reduction-style analysis of three such algorithms, each making different as- sumptions, which relates the supervised learning objectives to the sub-optimality of search using the learned partial policies. Our second contribu- tion is to describe concrete implementations of the algorithms within the popular framework of Monte-Carlo tree search. Finally, the third con- tribution is to evaluate the learning algorithms in two challenging MDPs with large action branch- ing factors, showing that the learned partial poli- cies can significantly improve the anytime per- formance of Monte-Carlo tree search.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-wang14a,
  title = 	 {{B}ayesian Filtering with Online {G}aussian Process Latent Variable Models},
  author =       {Wang, Yali and Chicago, Marcus Brubaker Toyota Technological Institute at and Chaib-draa, Brahim and Urtasun, Raquel},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {677--685},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/wang14a/wang14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/wang14a.html},
  abstract = 	 {In this paper we present a novel non-parametric approach to Bayesian filtering, where the predic- tion and observation models are learned in an online fashion. Our approach is able to han- dle multimodal distributions over both models by employing a mixture model representation with Gaussian Processes (GP) based components. To cope with the increasing complexity of the esti- mation process, we explore two computationally efficient GP variants, sparse online GP and local GP, which help to manage computation require- ments for each mixture component. Our exper- iments demonstrate that our approach can track human motion much more accurately than exist- ing approaches that learn the prediction and ob- servation models offline and do not update these models with the incoming data stream.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-university14s,
  title = 	 {Improved Densification of One Permutation Hashing},
  author =       {University, Anshumali Shrivastava Cornell and University, Ping Li Rutgers},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {686--695},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/university14s/university14s.pdf},
  url = 	 {https://proceedings.mlr.press/r12/university14s.html},
  abstract = 	 {The existing work on densification of one permu- tation hashing [24] reduces the query processing cost of the (K, L)-parameterized Locality Sen- sitive Hashing (LSH) algorithm with minwise hashing, from O(dKL) to merely O(d + KL), where d is the number of nonzeros of the data vector, K is the number of hashes in each hash table, and L is the number of hash tables. While that is a substantial improvement, our analy- sis reveals that the existing densification scheme in [24] is sub-optimal. In particular, there is no enough randomness in that procedure, which af- fects its accuracy on very sparse datasets. In this paper, we provide a new densification pro- cedure which is provably better than the existing scheme [24]. This improvement is more signifi- cant for very sparse datasets which are common over the web. The improved technique has the same cost of O(d + KL) for query processing, thereby making it strictly preferable over the ex- isting procedure. Experimental evaluations on public datasets, in the task of hashing based near neighbor search, support our theoretical findings.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-dubey14a,
  title = 	 {Parallel {M}arkov Chain {M}onte {C}arlo for Pitman-Yor Mixture Models},
  author =       {Dubey, Kumar Avinava and Williamson, Sinead and Xing, Eric P.},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {696--705},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/dubey14a/dubey14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/dubey14a.html},
  abstract = 	 {The Pitman-Yor process provides an elegant way to cluster data that exhibit power law behavior, where the number of clusters is unknown or un- bounded. Unfortunately, inference in Pitman- Yor process-based models is typically slow and does not scale well with dataset size. In this paper we present new auxiliary-variable repre- sentations for the Pitman-Yor process and a spe- cial case of the hierarchical Pitman-Yor process that allows us to develop parallel inference algo- rithms that distribute inference both on the data space and the model space. We show that our method scales well with increasing data while avoiding any degradation in estimate quality.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-university14t,
  title = 	 {Sequential Model-Based Ensemble Optimization},
  author =       {University, Alexandre Lacoste Laval and Larochelle, Hugo and University, Mario Marchand Laval and University, Fran{\c{c}}ois Laviolette Laval},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {706--714},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/university14t/university14t.pdf},
  url = 	 {https://proceedings.mlr.press/r12/university14t.html},
  abstract = 	 {One of the most tedious tasks in the applica- tion of machine learning is model selection, i.e. hyperparameter selection. Fortunately, recent progress has been made in the automation of this process, through the use of sequential model- based optimization (SMBO) methods. This can be used to optimize a cross-validation perfor- mance of a learning algorithm over the value of its hyperparameters. However, it is well known that ensembles of learned models almost consis- tently outperform a single model, even if prop- erly selected. In this paper, we thus propose an extension of SMBO methods that automatically constructs such ensembles. This method builds on a recently proposed ensemble construction paradigm known as Agnostic Bayesian learning. In experiments on 22 regression and 39 classifi- cation data sets, we confirm the success of this proposed approach, which is able to outperform model selection with SMBO.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-cuhk14a,
  title = 	 {Nuclear Norm Regularized Least Squares Optimization on Grassmannian Manifolds},
  author =       {CUHK, Yuanyuan Liu and Shang, Fanhua and Cheng, Hong and Cheng, James},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {715--724},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/cuhk14a/cuhk14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/cuhk14a.html},
  abstract = 	 {This paper aims to address a class of nuclear norm regularized least square (NNLS) problems. By exploiting the underlying low-rank matrix manifold structure, the problem with nuclear norm regularization is cast to a Riemannian opti- mization problem over matrix manifolds. Com- pared with existing NNLS algorithms involving singular value decomposition (SVD) of large- scale matrices, our method achieves significant reduction in computational complexity. More- over, the uniqueness of matrix factorization can be guaranteed by our Grassmannian manifold method. In our solution, we first introduce the bilateral factorization into the original NNLS problem and convert it into a Grassmannian op- timization problem by using a linearized tech- nique. Then the conjugate gradient procedure on the Grassmannian manifold is developed for our method with a guarantee of local convergence. Finally, our method can be extended to address the graph regularized problem. Experimental re- sults verified both the efficiency and effective- ness of our method.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-rosenkrantz14a,
  title = 	 {{B}ayesian Inference in Treewidth-Bounded Graphical Models Without Indegree Constraints},
  author =       {Rosenkrantz, Daniel J. and Tech, Madhav V.Marathe Virginia and Ravi, S. S. and Tech, Anil K. Vullikanti Virginia},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {725--734},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/rosenkrantz14a/rosenkrantz14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/rosenkrantz14a.html},
  abstract = 	 {We present new polynomial time algorithms for inference problems in Bayesian networks (BNs) when restricted to instances that satisfy the following two conditions: they have bounded treewidth and the conditional probability table (CPT) at each node is specified concisely using an r-symmetric function for some constant r. Our polynomial time algorithms work directly on the unmoralized graph. Our results significantly ex- tend known results regarding inference problems on treewidth bounded BNs to a larger class of problem instances. We also show that relaxing either of the conditions used by our algorithms leads to computational intractability.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-el-hay14a,
  title = 	 {Structured Proportional Jump Processes},
  author =       {El-Hay, Tal and Weissbrod, Omer and Eban, Elad and Zazzi, Maurizio and Incardona, Francesca},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {735--744},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/el-hay14a/el-hay14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/el-hay14a.html},
  abstract = 	 {Learning the association between observed variables and future trajectories of continuous- time stochastic processes is a fundamental task in dynamic modeling. Often the dynamics are non-homogeneous and involve a large number of interacting components. We introduce a conditional probabilistic model that captures such dynamics, while maintaining scalability and providing an explicit way to express the interrelation between the system components. The principal idea is a factorization of the model into two distinct elements: one depends only on time and the other depends on the system configuration. We developed a learning procedure, given either full or point observations, and tested it on simulated data. We applied the proposed modeling scheme to study large cohorts of diabetes and HIV patients, and demonstrate that the factorization helps shed light on the dynamics of these diseases.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-university14u,
  title = 	 {A Spectral Algorithm for Learning Class-Based $n$-gram Models of Natural Language},
  author =       {University, Karl Stratos Columbia and Kim, Do-kyum and University, Daniel Hsu Columbia and University, Michael Collins Columbia},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {745--754},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/university14u/university14u.pdf},
  url = 	 {https://proceedings.mlr.press/r12/university14u.html},
  abstract = 	 {The Brown clustering algorithm (Brown et al., 1992) is widely used in natural language process- ing (NLP) to derive lexical representations that are then used to improve performance on vari- ous NLP problems. The algorithm assumes an underlying model that is essentially an HMM, with the restriction that each word in the vocab- ulary is emitted from a single state. A greedy, bottom-up method is then used to find the clus- tering; this method does not have a guarantee of finding the correct underlying clustering. In this paper we describe a new algorithm for clustering under the Brown et al. model. The method relies on two steps: first, the use of canonical correla- tion analysis to derive a low-dimensional repre- sentation of words; second, a bottom-up hierar- chical clustering over these representations. We show that given a sufficient number of training examples sampled from the Brown et al. model, the method is guaranteed to recover the correct clustering. Experiments show that the method recovers clusters of comparable quality to the al- gorithm of Brown et al. (1992), but is an order of magnitude more efficient.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-stanculescu14a,
  title = 	 {A Hierarchical Switching Linear Dynamical System Applied to the Detection of Sepsis in Neonatal Condition Monitoring},
  author =       {Stanculescu, Ioan and Williams, Christopher K.I. and Freer, Yvonne},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {755--764},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/stanculescu14a/stanculescu14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/stanculescu14a.html},
  abstract = 	 {In this paper we develop a Hierarchi- cal Switching Linear Dynamical System (HSLDS) for the detection of sepsis in neonates in an intensive care unit. The Fac- torial Switching LDS (FSLDS) of Quinn et al. (2009) is able to describe the observed vital signs data in terms of a number of discrete factors, which have either physiological or ar- tifactual origin. In this paper we demonstrate that by adding a higher-level discrete variable with semantics sepsis/non-sepsis we can de- tect changes in the physiological factors that signal the presence of sepsis. We demonstrate that the performance of our model for the detection of sepsis is not statistically differ- ent from the auto-regressive HMM of Stan- culescu et al. (2013), despite the fact that their model is given “ground truth” annota- tions of the physiological factors, while our HSLDS must infer them from the raw vital signs data.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-gunter14a,
  title = 	 {Efficient {B}ayesian Nonparametric Modelling of Structured Point Processes},
  author =       {Gunter, Tom and Lloyd, Chris and Roberts, Stephen and Osborne, Michael A.},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {765--774},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/gunter14a/gunter14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/gunter14a.html},
  abstract = 	 {This paper presents a Bayesian generative model for dependent Cox point processes, alongside an efficient inference scheme which scales as if the point processes were mod- elled independently. We can handle miss- ing data naturally, infer latent structure, and cope with large numbers of observed pro- cesses. A further novel contribution enables the model to work effectively in higher dimen- sional spaces. Using this method, we achieve vastly improved predictive performance on both 2D and 1D real data, validating our structured approach.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-halloran14a,
  title = 	 {Learning Peptide-Spectrum Alignment Models for Tandem Mass Spectrometry},
  author =       {Halloran, John and Bilmes, Jeffrey and Noble, William},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {775--784},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/halloran14a/halloran14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/halloran14a.html},
  abstract = 	 {We present a peptide-spectrum alignment strategy that employs a dynamic Bayesian network (DBN) for the identification of spectra produced by tan- dem mass spectrometry (MS/MS). Our method is fundamentally generative in that it models peptide fragmentation in MS/MS as a physical process. The model traverses an observed MS/MS spec- trum and a peptide-based theoretical spectrum to calculate the best alignment between the two spectra. Unlike all existing state-of-the-art meth- ods for spectrum identification that we are aware of, our method can learn alignment probabilities given a dataset of high-quality peptide-spectrum pairs. The method, moreover, accounts for noise peaks and absent theoretical peaks in the observed spectrum. We demonstrate that our method out- performs, on a majority of datasets, several widely used, state-of-the-art database search tools for spectrum identification. Furthermore, the pro- posed approach provides an extensible framework for MS/MS analysis and provides useful informa- tion that is not produced by other methods, thanks to its generative structure.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-masegosa14a,
  title = 	 {Stochastic Discriminative {EM}},
  author =       {Masegosa, Andres},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {785--794},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/masegosa14a/masegosa14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/masegosa14a.html},
  abstract = 	 {Stochastic discriminative EM (sdEM) is an online-EM-type algorithm for discriminative training of probabilistic generative models be- longing to the natural exponential family. In this work, we introduce and justify this algorithm as a stochastic natural gradient descent method, i.e. a method which accounts for the informa- tion geometry in the parameter space of the sta- tistical model. We show how this learning algo- rithm can be used to train probabilistic genera- tive models by minimizing different discrimina- tive loss functions, such as the negative condi- tional log-likelihood and the Hinge loss. The re- sulting models trained by sdEM are always gen- erative (i.e. they define a joint probability distri- bution) and, in consequence, allows to deal with missing data and latent variables in a principled way either when being learned or when making predictions. The performance of this method is illustrated by several text classification problems for which a multinomial naive Bayes and a latent Dirichlet allocation based classifier are learned using different discriminative loss functions.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-university14v,
  title = 	 {Accelerating {MCMC} via Parallel Predictive Prefetching},
  author =       {University, Elaine Angelino Harvard and University, Eddie Kohler Harvard and University, Margo Seltzer Harvard and University, Amos Waterland Harvard and Harvard, Ryan Adams},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {795--804},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/university14v/university14v.pdf},
  url = 	 {https://proceedings.mlr.press/r12/university14v.html},
  abstract = 	 {Parallel predictive prefetching is a new frame- work for accelerating a large class of widely- used Markov chain Monte Carlo (MCMC) algo- rithms. It speculatively evaluates many potential steps of an MCMC chain in parallel while ex- ploiting fast, iterative approximations to the tar- get density. This can accelerate sampling from target distributions in Bayesian inference prob- lems. Our approach takes advantage of whatever parallel resources are available, but produces re- sults exactly equivalent to standard serial execu- tion. In the initial burn-in phase of chain evalu- ation, we achieve speedup close to linear in the number of available cores.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-gribkoff14a,
  title = 	 {Understanding the Complexity of Lifted Inference and Asymmetric Weighted Model Counting},
  author =       {Gribkoff, Eric and Broeck, Guy Van Den and UW, Dan Suciu},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {805--814},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/gribkoff14a/gribkoff14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/gribkoff14a.html},
  abstract = 	 {In this paper we study lifted inference for the Weighted First-Order Model Counting prob- lem (WFOMC), which counts the assignments that satisfy a given sentence in first-order logic (FOL); it has applications in Statisti- cal Relational Learning (SRL) and Probabilis- tic Databases (PDB). We present several results. First, we describe a lifted inference algorithm that generalizes prior approaches in SRL and PDB. Second, we provide a novel dichotomy result for a non-trivial fragment of FO CNF sentences, showing that for each sentence the WFOMC problem is either in PTIME or #P- hard in the size of the input domain; we prove that, in the first case our algorithm solves the WFOMC problem in PTIME, and in the second case it fails. Third, we present several proper- ties of the algorithm. Finally, we discuss limi- tations of lifted inference for symmetric proba- bilistic databases (where the weights of ground literals depend only on the relation name, and not on the constants of the domain), and prove the impossibility of a dichotomy result for the complexity of probabilistic inference for the en- tire language FOL.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-university14w,
  title = 	 {Scalable Binary Tensor Factorization},
  author =       {University, Beyza Ermi\c{s} Bo\u{g}azi{\c{c}}i and Europe, Guillaume Bouchard Xerox Research Centre},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {815--822},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/university14w/university14w.pdf},
  url = 	 {https://proceedings.mlr.press/r12/university14w.html},
  abstract = 	 {Binary matrices and tensors are popular data structures that need to be efficiently approxi- mated by low-rank representations. A standard approach is to minimize the logistic loss, well suited for binary data. In many cases, the num- ber m of non-zero elements in the tensor is much smaller than the total number n of possible en- tries in the tensor. This creates a problem for large tensors because the computation of the lo- gistic loss has a linear time complexity with n. In this work, we show that an alternative approach is to minimize the quadratic loss (root mean square error) which leads to algorithms with a training time complexity that is reduced from O(n) to O(m), as proposed earlier in the restricted case of alternating least-square algorithms. In addi- tion, we propose and study a greedy algorithm that partitions the tensor into smaller tensors, each approximated by a quadratic upper bound. This technique provides a time-accuracy trade- off between a fast but approximate algorithm and an accurate but slow algorithm. We show that this technique leads to a considerable speedup in learning of real world tensors.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-kinathil14a,
  title = 	 {Closed-form Solutions to a Subclass of Continuous Stochastic Games via Symbolic Dynamic Programming},
  author =       {Kinathil, Shamin and Sanner, Scott and Della Penna, Nicol{\'a}s},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {823--832},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/kinathil14a/kinathil14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/kinathil14a.html},
  abstract = 	 {Zero-sum stochastic games provide a formal- ism to study competitive sequential interactions between two agents with diametrically oppos- ing goals and evolving state. A solution to such games with discrete state was presented by Littman (Littman, 1994). The continuous state version of this game remains unsolved. In many instances continuous state solutions require non- linear optimisation, a problem for which closed- form solutions are generally unavailable. We present an exact closed-form solution to a sub- class of zero-sum continuous stochastic games that can be solved as a parameterised linear pro- gram by utilising symbolic dynamic program- ming. This novel technique is applied to calcu- late exact solutions to a variety of zero-sum con- tinuous state stochastic games.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-geiger14a,
  title = 	 {Estimating causal effects by bounding confounding},
  author =       {Geiger, Philipp and Janzing, Dominik},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {833--842},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/geiger14a/geiger14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/geiger14a.html},
  abstract = 	 {Assessing the causal effect of a treatment variable X on an outcome variable Y is usually difficult due to the existence of un- observed common causes. Without further assumptions, observed dependences do not even prove the existence of a causal effect from X to Y . It is intuitively clear that strong statistical dependences between X and Y do provide evidence for X influenc- ing Y if the influence of common causes is known to be weak. We propose a framework that formalizes effect versus confounding in various ways and derive upper/lower bounds on the effect in terms of a priori given bounds on confounding. The formalization includes information theoretic quantities like informa- tion flow and causal strength, as well as other common notions like effect of treatment on the treated (ETT). We discuss several sce- narios where upper bounds on the strength of confounding can be derived. This justifies to some extent human intuition which assumes the presence of causal effect when strong (e.g. close to deterministic) statistical relations are observed.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-moore14a,
  title = 	 {Fast {G}aussian Process Posteriors with Product Trees},
  author =       {Moore, David and Russell, Stuart},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {843--852},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/moore14a/moore14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/moore14a.html},
  abstract = 	 {Gaussian processes (GP) are a powerful tool for nonparametric regression; unfortunately, calcu- lating the posterior variance in a standard GP model requires time O(n2) in the size of the training set. Previous work by Shen et al. (2006) used a k-d tree structure to approximate the pos- terior mean in certain GP models. We extend this approach to achieve efficient approximation of the posterior covariance using a tree clustering on pairs of training points, and demonstrate sig- nificant improvements in performance with neg- ligible loss of accuracy.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-acharyya14a,
  title = 	 {{MEMR}: A Margin Equipped Monotone Retargeting Framework for Ranking},
  author =       {Acharyya, Sreangsu and Ghosh, Joydeep},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {853--862},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/acharyya14a/acharyya14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/acharyya14a.html},
  abstract = 	 {We bring to bear the tools of convexity, mar- gins and the newly proposed technique of monotone retargeting upon the task of learn- ing permutations from examples. This leads to novel and efficient algorithms with guaran- teed prediction performance in the online set- ting and on global optimality and the rate of convergence in the batch setting. Monotone retargeting efficiently optimizes over all pos- sible monotone transformations as well as the finite dimensional parameters of the model. As a result we obtain an effective algorithm to learn transitive relationships over items. It captures the inherent combinatorial char- acteristics of the output space yet it has a computational burden not much more than that of a generalized linear model.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-university14x,
  title = 	 {CoRE Kernels},
  author =       {University, Ping Li Rutgers},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {863--871},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/university14x/university14x.pdf},
  url = 	 {https://proceedings.mlr.press/r12/university14x.html},
  abstract = 	 {The term “CoRE kernel” stands for correlation- resemblance kernel. In many real-world applica- tions (e.g., computer vision), the data are often high-dimensional, sparse, and non-binary. We propose two types of (nonlinear) CoRE kernels for non-binary sparse data and demonstrate the effectiveness of the new kernels through a clas- sification experiment. CoRE kernels are sim- ple with no tuning parameters. However, train- ing nonlinear kernel SVM can be costly in time and memory and may not be always suitable for truly large-scale industrial applications (e.g., search). In order to make the proposed CoRE kernels more practical, we develop basic proba- bilistic hashing (approximate) algorithms which transform nonlinear kernels into linear kernels.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-institute14a,
  title = 	 {A Consistent Estimator of the Expected Gradient Outerproduct},
  author =       {Institute, Shubhendu Trivedi Toyota Technological and Wang, Jialei and TTI-Chicago, Samory Kpotufe and TT-Chicago, Gregory Shakhnarovich},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {872--881},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/institute14a/institute14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/institute14a.html},
  abstract = 	 {In high-dimensional classification or regression problems, the expected gradient outerproduct (EGOP) of the unknown regression function f, namely EX $\nabla$f(X) \cdot $\nabla$f(X)$\top$ , is known to recover those directions v $\in$Rd most relevant to predicting the output Y . However, just as in gradient estimation, opti- mal estimators of the EGOP can be expensive in practice. We show that a simple rough estima- tor, much cheaper in practice, suffices to obtain significant improvements on real-world nonpara- metric classification and regression tasks. Fur- thermore, we prove that, despite its simplicity, this rough estimator remains statistically consis- tent under mild conditions.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-van-hasselt14a,
  title = 	 {Off-policy {TD}($ł$) with a true online equivalence},
  author =       {Van Hasselt, Hado and Mahmood, Rupam and Sutton, Rich},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {882--891},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/van-hasselt14a/van-hasselt14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/van-hasselt14a.html},
  abstract = 	 {Van Seijen and Sutton (2014) recently proposed a new version of the linear TD($\lambda$) learning algo- rithm that is exactly equivalent to an online for- ward view and that empirically performed bet- ter than its classical counterpart in both predic- tion and control problems. However, their al- gorithm is restricted to on-policy learning. In the more general case of off-policy learning, in which the policy whose outcome is predicted and the policy used to generate data may be differ- ent, their algorithm cannot be applied. One rea- son for this is that the algorithm bootstraps and thus is subject to instability problems when func- tion approximation is used. A second reason true online TD($\lambda$) cannot be used for off-policy learning is that the off-policy case requires so- phisticated importance sampling in its eligibility traces. To address these limitations, we gener- alize their equivalence result and use this gen- eralization to construct the first online algorithm to be exactly equivalent to an off-policy forward view. We show this algorithm, named true on- line GTD($\lambda$), empirically outperforms GTD($\lambda$) (Maei, 2011) which was derived from the same objective as our forward view but lacks the ex- act online equivalence. In the general theorem that allows us to derive this new algorithm, we encounter a new general eligibility-trace update.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-bhattacharjya14a,
  title = 	 {{B}ayesian Interactive Decision Support for Multi-Attribute Problems with Even Swaps},
  author =       {Bhattacharjya, Debarun and Kephart, Jeffrey},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {892--901},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/bhattacharjya14a/bhattacharjya14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/bhattacharjya14a.html},
  abstract = 	 {Even swaps is a method for solving de- terministic multi-attribute decision problems where the decision maker iteratively simpli- fies the problem until the optimal alterna- tive is revealed (Hammond et al. 1998, 1999). We present a new practical decision support system that takes a Bayesian approach to guiding the even swaps process, where the system makes queries based on its beliefs about the decision maker’s preferences and updates them as the interactive process un- folds. Through experiments, we show that it is possible to learn enough about the decision maker’s preferences to measurably reduce the cognitive burden, i.e. the number and com- plexity of queries posed by the system.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-university14y,
  title = 	 {Multi-label Image Classification with A Probabilistic Label Enhancement Model},
  author =       {University, Xin Li Temple and University, Feipeng Zhao Temple and University, Yuhong Guo Temple},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {902--911},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/university14y/university14y.pdf},
  url = 	 {https://proceedings.mlr.press/r12/university14y.html},
  abstract = 	 {In this paper, we present a novel probabilistic la- bel enhancement model to tackle multi-label im- age classification problem. Recognizing multiple objects in images is a challenging problem due to label sparsity, appearance variations of the ob- jects and occlusions. We propose to tackle these difficulties from a novel perspective by construct- ing auxiliary labels in the output space. Our idea is to exploit label combinations to enrich the la- bel space and improve the label identification ca- pacity in the original label space. In particular, we identify a set of informative label combina- tion pairs by constructing a tree-structured graph in the label space using the maximum spanning tree algorithm, which naturally forms a condi- tional random field. We then use the produced label pairs as auxiliary new labels to augment the original labels and perform piecewise train- ing under the framework of conditional random fields. In the test phase, max-product message passing is used to perform efficient inference on the tree graph, which integrates the augmented label pair classifiers and the standard individual binary classifiers for multi-label prediction. We evaluate the proposed approach on several image classification datasets. The experimental results demonstrate the superiority of our label enhance- ment model in terms of both prediction perfor- mance and running time comparing to the-state- of-the-art multi-label learning methods.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-amsterdam14a,
  title = 	 {Combining predictions from linear models when training and test inputs differ},
  author =       {Amsterdam, Thijs Van Ommen CWI},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {912--921},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/amsterdam14a/amsterdam14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/amsterdam14a.html},
  abstract = 	 {Methods for combining predictions from dif- ferent models in a supervised learning setting must somehow estimate/predict the quality of a model’s predictions at unknown future inputs. Many of these methods (often implicitly) make the assumption that the test inputs are identical to the training inputs, which is seldom reasonable. By failing to take into account that prediction will generally be harder for test inputs that did not occur in the training set, this leads to the se- lection of too complex models. Based on a novel, unbiased expression for KL divergence, we pro- pose XAIC and its special case FAIC as versions of AIC intended for prediction that use different degrees of knowledge of the test inputs. Both methods substantially differ from and may out- perform all the known versions of AIC even when the training and test inputs are iid, and are es- pecially useful for deterministic inputs and under covariate shift. Our experiments on linear models suggest that if the test and training inputs differ substantially, then XAIC and FAIC predictively outperform AIC, BIC and several other methods including Bayesian model averaging.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



@InProceedings{pmlr-vR12-berlin14a,
  title = 	 {A {B}ayesian Nonparametric Model for Spectral Estimation of Metastable Systems},
  author =       {Berlin, Hao Wu Free University of},
  booktitle = 	 {Proceedings of the 30th Conference on Uncertainty in Artificial Intelligence},
  pages = 	 {922--931},
  year = 	 {2014},
  editor = 	 {Zhang, Nevin L. and Tian, Jin},
  volume = 	 {R12},
  series = 	 {Proceedings of Machine Learning Research},
  month = 	 {23--27 Jul},
  publisher =    {PMLR},
  pdf = 	 {https://raw.githubusercontent.com/mlresearch/r12/main/assets/berlin14a/berlin14a.pdf},
  url = 	 {https://proceedings.mlr.press/r12/berlin14a.html},
  abstract = 	 {The identification of eigenvalues and eigenfunc- tions from simulation or experimental data is a fundamental and important problem for anal- ysis of metastable systems, because the domi- nant spectral components usually contain a lot of essential information of the metastable dy- namics on slow timescales. It has been shown that the dynamics of a strongly metastable sys- tem can be equivalently described as a hidden Markov model (HMM) under some technical as- sumptions and the spectral estimation can be performed through HMM learning. However, the spectral estimation with unknown number of dominant spectra is still a challenge in the framework of traditional HMMs, and the infi- nite HMMs developed based on stick-breaking processes cannot satisfactorily solved this prob- lem either. In this paper, we analyze the diffi- culties of spectral estimation for infinite HMMs, and present a new nonparametric model called stick-breaking half-weighted model (SB-HWM) to address this problem. The SB-HWM defines a sparse prior of eigenvalues and can be applied to Bayesian inference of dominant eigenpairs of metastable systems in a nonparametric manner. We demonstrate by simulations the advantages of applying SB-HWM to spectral estimation.},
  note =         {Reissued by PMLR on 04 October 2026.}
}



