\documentclass[11pt]{article} \usepackage{amssymb} \usepackage{amsfonts} \usepackage{amsmath} \usepackage{enumitem} \usepackage{bbold} \usepackage{bm} \parindent0in \pagestyle{myheadings} \DeclareSymbolFont{AMSb}{U}{msb}{m}{n} \DeclareMathSymbol{\N}{\mathbin}{AMSb}{"4E} \DeclareMathSymbol{\Z}{\mathbin}{AMSb}{"5A} \DeclareMathSymbol{\R}{\mathbin}{AMSb}{"52} \DeclareMathSymbol{\Q}{\mathbin}{AMSb}{"51} \DeclareMathSymbol{\I}{\mathbin}{AMSb}{"49} \DeclareMathSymbol{\C}{\mathbin}{AMSb}{"43} \newtheorem{theorem}{Theorem} \newenvironment{proof}[1][Proof]{\flushleft \textbf{#1.} \text }{ \hfill $\blacksquare$} \newtheorem{example0}{Example}[section] \newtheorem{lemma0}{Lemma}[section] \newtheorem{theorem0}{Theorem} \newenvironment{definition}[1][Definition]{\flushleft \textbf{#1 :}}{ } \newenvironment{example}{\medskip \flushleft \begin{example0}\rm}{\end{example0}} \newenvironment{lemma}{\medskip \flushleft \begin{lemma0}\rm}{\end{lemma0}} \renewenvironment{theorem}{\medskip \flushleft \begin{theorem0} }{\end{theorem0}} \newenvironment{Note}[1][NOTE]{\flushleft\textbf{#1 :} }{ } \renewenvironment{definition}[1][Definition]{\flushleft \textbf{#1 : }}{ } \renewenvironment{example}[1][Example]{\flushleft\textbf{#1}}{} \newenvironment{corollary}[1][Corollary :]{\flushleft \textbf{#1} }{} \newenvironment{mytheorem}[1][Theorem]{\flushleft \textbf{#1} %\textbf{} \hspace{0.05 in}\it }{ \rm } \usepackage{../Math598} \def\X{\mathbb{X}} \def\bmthzero{\bm{\theta_0}} \def\bmthone{\bm{\theta_1}} \def\bmthtwo{\bm{\theta_2}} \def\th{\theta} \def\bmphi{\bm{\phi}} \def\bmphizero{\bm{\phi_0}} \def\bmth{\bm{\theta}} \def\bmphi{\bm{\phi}} \def\bmx{\bm{x}} \def\bmg{\bm{g}} \def\transpose{{\textsf{T}}} \def\bmZn{\bm{Z}_n} \def\bmZ{\bm{Z}} \def\bmD{\bm{D}} \def\dddot#1{\stackrel{\ldots}{#1}} \def\lra{\longrightarrow} \def\Lra{\Longrightarrow} \def\Llra{\Longleftrightarrow} \def\CiP{\stackrel{p}{\lra}} \def\Cas{\stackrel{a.s.}{\lra}} \def\CiL{\stackrel{\mathfrak{L}}{\lra}} \def\eqdef{\stackrel{def}{=}} \def\tildethn{\skew2\tilde{\bmth}_n} \def\hatthn{\skew2\hat{\bmth}_n} \def\tildephin{\skew2\tilde{\bmphi}_n} \def\thnstar{\bm{\theta}_n^\star} \def\thtilde{\widetilde{\bmth}_n} \def\thhat{\widehat{\bmth}_n} \def\dotlike{\dot{\bm{l}}_n} \def\ddotlike{\ddot{\bm{l}}_n} \def\thhatNewt{\widehat{\bmth}^{(1)}} \def\thhatScor{\widehat{\bmth}^{\star}} \def\d{\textsf{d}} \def\action{\textsf{a}} \def\Loss{L} \begin{document} <>= library(knitr) # global chunk options opts_chunk$set(cache=TRUE, autodep=TRUE, fig.align='center') options(scipen=999) options(repos=c(CRAN="https://cloud.r-project.org/")) inline_hook <- function (x) { if (is.numeric(x)) { # ifelse does a vectorized comparison # If integer, print without decimal; otherwise print two places res <- ifelse(x == round(x), sprintf("%.3f", x), sprintf("%.3f", x) ) paste(res, collapse = ", ") } } #knit_hooks$set(inline = inline_hook) @ <>= myrnd <- function(x,d) {formatC(x, format="f", digits=d)} set.seed(10101) @ \begin{center} \textsclarge{MATH 559: Bayesian Theory and Methods} \vspace{0.1 in} \textsc{Decision Theory: Two Special Cases} \end{center} The Bayesian decision for loss function $\Loss(\d(\by),\theta)$ and posterior distribution $\pi_n(\theta)$ is obtained as \[ \min_d \int \Loss(\d(\by),\theta) \pi_n(\theta) \ d \theta. \] We examine the solution to two specific decision problems. \begin{enumerate}[label=\arabic*.] \item \textbf{Credible intervals:} Suppose $0 < \kappa < 1$, and consider the loss function for $\theta \in \R$ defined as \[ \Loss(\d,\theta) = \kappa(\d-\theta )\mathbb{1}(\theta < \d) + (1-\kappa)(\theta - \d)\mathbb{1}(\theta \ge \d) = \Loss_{1-\kappa}(\d,\theta) \] say. The expected posterior loss for possible decision $t$ is then \begin{align*} \int \Loss(t,\theta) \pi_n(\theta) \ d \theta & = \int \kappa(t-\theta )\mathbb{1}(\theta < t) \pi_n(\theta) \ d \theta + \int (1-\kappa)(\theta - t)\mathbb{1}(\theta \ge t) \pi_n(\theta) \ d \theta \\[6pt] & = \kappa \int_{-\infty}^{t} (t - \theta) \pi_n(\theta) \ d \theta + (1-\kappa) \int_{t}^\infty (\theta-t) \pi_n(\theta) \ d \theta \end{align*} Differentiating with respect to $t$ we obtain by the chain rule and fundamental rule of calculus \begin{itemize} \item First term: \[ \kappa \left[ \int_{-\infty}^t \pi_n(\theta) \ d \theta + t \pi_n(t) - t \pi_n(t) \right] = \kappa \int_{-\infty}^t \pi_n(\theta) \ d \theta \] \item Second term: \[ (1-\kappa) \left[ -t \pi_n(t) - \int_{t}^\infty \pi_n(\theta) \ d \theta + t \pi_n(t) \right] = -(1-\kappa) \int_{t}^\infty \pi_n(\theta) \ d \theta \] \end{itemize} and so the minimizing value of $t$ solves \[ \kappa \int_{-\infty}^t \pi_n(\theta) \ d \theta = (1-\kappa) \int_{t}^\infty \pi_n(\theta) \ d \theta \] so that \[ \kappa = \int_{t}^\infty \pi_n(\theta) \ d \theta. \] Thus the solution to the loss minimization problem is the $1-\kappa$ quantile of $\pi_n(\theta)$. Now consider the decision $\d = (\d_1,\d_2)$ that comprises two components, with loss function \[ \Loss((\d_1,\d_2),\theta) = \Loss_{1-\kappa}(\d_1,\theta) + \Loss_{\kappa}(\d_2,\theta). \] By the above calculation, we obtain the solution \[ (\d_1(\by), \d_2(\by)) \] as the $\kappa$ and $1-\kappa$ posterior quantiles respectively, so that the interval between the two values contains posterior probability $1-2 \kappa$, and is the $100 (1-2 \kappa)$\% equal-tailed credible interval. \medskip The other versions of the Bayesian credible interval can also be obtained as loss minimization decisions. \item \textbf{Predictive Distribution:} Suppose our task is to estimate the unknown conditional density $f_Y(y;\theta_0)$. We aim to construct a Bayesian estimate of $f_Y(y;\theta_0)$ using the Kullback-Leibler loss, that is, we aim to minimize \[ \int \left\{ \int \log \frac{f_Y(y;\theta)}{f(y)} f_Y(y;\theta) \ d y \right\} \pi_n(\theta) \ d \theta \] with respect to the function $f(\cdot)$, that is, to maximize \[ \iint \log f(y) f_Y(y;\theta) \pi_n(\theta) \ d \theta \ d y. \] Re-writing the double integral \[ \int \log f(y) \left\{\int f_Y(y;\theta) \pi_n(\theta) \ d \theta \right\} \ dy = \int \log f(y) \ p_n(y) \ dy \] and maximizing this quantity is the same as minimizing \[ \int \log \frac{p_n(y)}{f(y)} p_n(y) \ d y. \] Thus by properties of the KL divergence, the optimal $f$ is $p_n(y)$ \[ p_n(y) = \int f_Y(y;\theta) \pi_n(\theta) \ d \theta . \] that is, the posterior predictive distribution. \end{enumerate} \end{document}