r""" An epsilon-greedy policy for multi-armed bandit problems. Notes ----- Epsilon-greedy policies greedily select the arm with the highest expected payoff with probability :math:`1-\epsilon`, and selects an arm uniformly at random with probability :m
(self, epsilon=0.05, ev_prior=0.5)
| 100 | |
| 101 | class EpsilonGreedy(BanditPolicyBase): |
| 102 | def __init__(self, epsilon=0.05, ev_prior=0.5): |
| 103 | r""" |
| 104 | An epsilon-greedy policy for multi-armed bandit problems. |
| 105 | |
| 106 | Notes |
| 107 | ----- |
| 108 | Epsilon-greedy policies greedily select the arm with the highest |
| 109 | expected payoff with probability :math:`1-\epsilon`, and selects an arm |
| 110 | uniformly at random with probability :math:`\epsilon`: |
| 111 | |
| 112 | .. math:: |
| 113 | |
| 114 | P(a) = \left\{ |
| 115 | \begin{array}{lr} |
| 116 | \epsilon / N + (1 - \epsilon) &\text{if } |
| 117 | a = \arg \max_{a' \in \mathcal{A}} |
| 118 | \mathbb{E}_{q_{\hat{\theta}}}[r \mid a']\\ |
| 119 | \epsilon / N &\text{otherwise} |
| 120 | \end{array} |
| 121 | \right. |
| 122 | |
| 123 | where :math:`N = |\mathcal{A}|` is the number of arms, |
| 124 | :math:`q_{\hat{\theta}}` is the estimate of the arm payoff |
| 125 | distribution under current model parameters :math:`\hat{\theta}`, and |
| 126 | :math:`\mathbb{E}_{q_{\hat{\theta}}}[r \mid a']` is the expected |
| 127 | reward under :math:`q_{\hat{\theta}}` of receiving reward `r` after |
| 128 | taking action :math:`a'`. |
| 129 | |
| 130 | Parameters |
| 131 | ---------- |
| 132 | epsilon : float in [0, 1] |
| 133 | The probability of taking a random action. Default is 0.05. |
| 134 | ev_prior : float |
| 135 | The starting expected payoff for each arm before any data has been |
| 136 | observed. Default is 0.5. |
| 137 | """ |
| 138 | super().__init__() |
| 139 | self.epsilon = epsilon |
| 140 | self.ev_prior = ev_prior |
| 141 | self.pull_counts = defaultdict(lambda: 0) |
| 142 | |
| 143 | @property |
| 144 | def parameters(self): |