You signed in with another tab or window. Reload to refresh your session.You signed out in another tab or window. Reload to refresh your session.You switched accounts on another tab or window. Reload to refresh your session.Dismiss alert
{{ message }}
Repository navigation
Commit 84ffc3d
Browse filesBrowse the repository at this point in the historyBrowse files
For such a statement to make sense, we need a probability space on which the whole sequence of actions and observations is defined as a random variable.
74
74
75
75
We denote by $P[X \mid Y]$ the conditional distribution of a random variable $X$ given another random variable $Y$ under a probability measure $P$.
76
+
When we write that $P[X \mid Y] = \kappa$, or that $X$ has conditional distribution $\kappa$ given $Y$, the equality should be understood as holding $Y_* P$-almost surely.
$(\mathcal{F}_t)_{t \in\mathbb{N}}$ is the canonical filtration on $\Omega_{\mathcal{T}}$, and is the natural filtration for the canonical process $(X_t)_{t \in\mathbb{N}}$.
For any $t \in\mathbb{N}$, the conditional distribution $P_{\mathcal{T}}\left[A_{t+1} \mid H_t\right]$ is $((H_t)_* P_{\mathcal{T}})$-almost surely equal to $\pi_t$.
310
311
\end{lemma}
311
312
312
313
\begin{proof}\leanok
313
-
\uses{lem:condDistrib_X_add_one}
314
-
By Lemma~\ref{lem:condDistrib_X_add_one}, $P_{\mathcal{T}}\left[X_{t+1} \mid H_t\right]$ is $((H_t)_* P_{\mathcal{T}})$-almost surely equal to $\kappa_t = \pi_t \otimes\nu_t$.
314
+
\uses{lem:IT.condDistrib_X_add_one}
315
+
By Lemma~\ref{lem:IT.condDistrib_X_add_one}, $P_{\mathcal{T}}\left[X_{t+1} \mid H_t\right]$ is $((H_t)_* P_{\mathcal{T}})$-almost surely equal to $\kappa_t = \pi_t \otimes\nu_t$.
315
316
Since $A_{t+1}$ is the projection of $X_{t+1}$ on $\mathcal{A}_{t+1}$, $P_{\mathcal{T}}\left[A_{t+1} \mid H_t\right]$ is $((H_t)_* P_{\mathcal{T}})$-almost surely equal to the projection of $\kappa_t$ on $\mathcal{A}_{t+1}$, which is $\pi_t$.
For any $t \in\mathbb{N}$, the conditional distribution $P_{\mathcal{T}}\left[R_{t+1} \mid H_t, A_{t+1}\right]$ is $((H_t, A_{t+1})_* P_{\mathcal{T}})$-almost surely equal to $\nu_t$.
It suffices to show that $((H_t, A_{t+1})_* P_{\mathcal{T}}) \otimes\nu_t = (H_t, A_{t+1}, R_{t+1})_* P_{\mathcal{T}} = (H_t, X_{t+1})_* P_{\mathcal{T}}$.
329
-
By Lemma~\ref{lem:condDistrib_X_add_one}, $P_{\mathcal{T}}\left[X_{t+1} \mid H_t\right]$ is $((H_t)_* P_{\mathcal{T}})$-almost surely equal to $\kappa_t = \pi_t \otimes\nu_t$.
330
+
By Lemma~\ref{lem:IT.condDistrib_X_add_one}, $P_{\mathcal{T}}\left[X_{t+1} \mid H_t\right]$ is $((H_t)_* P_{\mathcal{T}})$-almost surely equal to $\kappa_t = \pi_t \otimes\nu_t$.
We thus have to prove that $((H_t)_* P_{\mathcal{T}}) \otimes (\pi_t \otimes\nu_t) = ((H_t, A_{t+1})_* P_{\mathcal{T}}) \otimes\nu_t$.
333
334
334
-
By Lemma~\ref{lem:condDistrib_A_add_one}, $(H_t, A_{t+1})_* P_{\mathcal{T}} = (H_t)_* P_{\mathcal{T}} \otimes\pi_t$, and replacing this in the right-hand side gives the left-hand side (using associativity of the composition-product).
335
+
By Lemma~\ref{lem:IT.condDistrib_A_add_one}, $(H_t, A_{t+1})_* P_{\mathcal{T}} = (H_t)_* P_{\mathcal{T}} \otimes\pi_t$, and replacing this in the right-hand side gives the left-hand side (using associativity of the composition-product).
The law of $A_0$ under $P_{\mathcal{T}}$ is $\alpha_0$.
343
344
\end{lemma}
344
345
345
346
\begin{proof}\leanok
346
-
\uses{lem:law_X_zero}
347
+
\uses{lem:IT.law_X_zero}
347
348
$X_0$ has law $\mu = \alpha_0\otimes\nu'_0$. $A_0$ is the projection of $X_0$ on the first space $\mathcal{A}_0$ and $\nu_0'$ is Markov, so $A_0$ has law $\alpha_0$.
The conditional distribution $P_{\mathcal{T}}\left[R_0\mid A_0\right]$ is $(A_{0*} P_{\mathcal{T}})$-almost surely equal to $\nu'_0$.
356
357
\end{lemma}
357
358
358
359
\begin{proof}\leanok
359
-
\uses{lem:law_X_zero}
360
+
\uses{lem:IT.law_X_zero}
360
361
To prove almost sure equality, it is enough to prove that $(A_{0*} P_{\mathcal{T}}) \otimes P_{\mathcal{T}}\left[R_0\mid A_0\right] = (A_{0*} P_{\mathcal{T}}) \otimes\nu'_0$.
361
362
By definition of the conditional distribution, we have $(A_{0*} P_{\mathcal{T}}) \otimes P_{\mathcal{T}}\left[R_0\mid A_0\right] = (A_0, R_0)_* P_{\mathcal{T}} = X_{0*} P_{\mathcal{T}}$.
362
-
By Lemma~\ref{lem:law_X_zero}, $X_{0*} P_{\mathcal{T}} = \mu = \alpha_0\otimes\nu'_0$.
363
-
By Lemma~\ref{lem:law_A_zero}, $A_{0*} P_{\mathcal{T}} = \alpha_0$.
363
+
By Lemma~\ref{lem:IT.law_X_zero}, $X_{0*} P_{\mathcal{T}} = \mu = \alpha_0\otimes\nu'_0$.
364
+
By Lemma~\ref{lem:IT.law_A_zero}, $A_{0*} P_{\mathcal{T}} = \alpha_0$.
364
365
Thus the two sides are equal.
365
366
\end{proof}
366
367
@@ -373,8 +374,8 @@ \subsection{Case of an algorithm-environment interaction}
The four conditions of Definition~\ref{def:IsAlgEnvSeq} are exactly the statements of Lemmas~\ref{lem:law_A_zero}, \ref{lem:condDistrib_R_zero}, \ref{lem:condDistrib_A_add_one} and \ref{lem:condDistrib_R_add_one}.
The four conditions of Definition~\ref{def:IsAlgEnvSeq} are exactly the statements of Lemmas~\ref{lem:IT.law_A_zero}, \ref{lem:IT.condDistrib_R_zero}, \ref{lem:IT.condDistrib_A_add_one} and \ref{lem:IT.condDistrib_R_add_one}.
378
379
\end{proof}
379
380
380
381
@@ -510,14 +511,14 @@ \section{Finitely many actions}
By Lemma~\ref{lem:sumRewards_bestArm_le_of_arm_mul_eq},
63
61
\begin{align*}
64
62
\mathbb{P}(\hat{A}_m^* = a)
65
63
&\le\mathbb{P}(S_{Km, a} \ge S_{Km, a^*})
66
64
\: .
67
65
\end{align*}
68
-
By Lemma~\ref{lem:sum_rewardByCount}, $S_{Km, a} = \sum_{i=1}^m Y_{a,i}$ and $S_{Km, a^*} = \sum_{i=1}^m Y_{a^*,i}$.
69
-
And by Lemma~\ref{lem:identDistrib_sum_Icc_rewardByCount} and the independence lemma~\ref{lem:independent_rewardByCount}, the pair $(\sum_{i=1}^m Y_{a,i}, \sum_{i=1}^m Y_{a^*,i})$ has the same distribution as $(\sum_{i=0}^{m-1} Z_{i,a}, \sum_{i=0}^{m-1} Z_{i,a^*})$.
The random variables $Z_{i,a}$ and $Z_{i,a^*}$ are i.i.d. with distributions $\nu(a)$ and $\nu(a^*)$ respectively, and 1-sub-Gaussian by assumption.
77
-
They satisfy the hypotheses of Lemma~\ref{lem:measure_sum_le_sum_le'}, and we thus obtain
66
+
By Lemma~\ref{lem:prob_sumRewards_le_sumRewards_le}, and then the concentration inequality of Lemma~\ref{lem:probReal_sum_le_sum_streamMeasure} we have
Since $N_{n,a} \le n$ (Lemma~\ref{lem:pullCount_basic}), there exists $k \in [1, n]$ such that $N_{n,a} = k$ and $\hat{\mu}_{n,a} = \frac{1}{k} \sum_{m=1}^k Y_{m,a}$.
0 commit comments