import numpy as np import matplotlib.pyplot as pl float_formatter = "{:.3f}".format np.set_printoptions(formatter={'float_kind': float_formatter}) K = 10 def bandit(bias): zn = np.random.uniform() + bias return zn if __name__ == '__main__': epsilon = 0.0 # Anti-greediness rho = 0.01 # reduce epsilon with age qu = np.zeros(K) nu = np.zeros(K) r_vec = [] # init bandits with different biases for shifting reward probability # -> expected reward q*(a) ql_star = [-0.4, -0.3, -0.2, -0.1, 0.0, +0.1, +0.2, +0.3, +0.4, +0.5] N = 1000 n_ages = 1000 for age in range(0, n_ages): r_sum = 0 for j in range(0, N): # choose action z = np.random.uniform() if z <= epsilon: # Choose random action a = np.random.randint(low=0, high=K) else: # Choose best action a = np.argmax(qu) # get reward from bandit r = bandit(ql_star[a]) nu[a] = nu[a] + 1 qu[a] = qu[a] + (r - qu[a])/nu[a] r_sum += r r_vec.append(r_sum/N) # Reduce tendency to explore with number of steps (or with age for humans) epsilon = epsilon*(1-rho) print(f"ql_star = {ql_star}") print(f"qu = {qu}") print(f"nu = {nu}") pl.plot(np.array(r_vec)) pl.grid() pl.show()