caffeine-potent/softmax_annealing_algorithm.py

## softmax_annealing_algorithm.py
import math
import random


def categorical_draw(probs):
    '''
    if
    P(A) = .5
    P(B) = .2
    P(C) = .3
    how do I uniformly sample this probabilit distribution?
    '''
    z = random.random()
    cum_prob = 0.0
    for i in range(len(probs)):
        prob = probs[i]
        cum_prob += prob
        if cum_prob > z:
            return i

class Softmax_Annealing_Bandit_Algorithm:
    def __init__(self, temperature, counts, values):
        self.counts = counts
        self.values = values
        return

    def initialize(self, n_arms):
        self.counts = [0 for col in range(n_arms)]
        self.values = [0.0 for col in range(n_armsb)]
        return

    def select_arm(self):
        t = sum(self.values) + 1
        temperature = 1/math.log(t + 0.0000001)
        z = sum([math.exp(v/ temperature) for v in self.values])
        probs = [math.exp(v/temperature)/z for v in self.values]
        return categorical_draw(probs)

    def update(self,chosen_arm,reward):
        self.counts[chosen_arm] = self.counts[chosen_arm] + 1
        n = self.counts[chosen_arm]

        value = self.values[chosen_arm]
        new_value = ((n-1) / float(n)) * value + (1/float(n)) * reward
        self.values[chosen_arm] = new_value
        return
	import math
	import random


	def categorical_draw(probs):
	'''
	if
	P(A) = .5
	P(B) = .2
	P(C) = .3
	how do I uniformly sample this probabilit distribution?
	'''
	z = random.random()
	cum_prob = 0.0
	for i in range(len(probs)):
	prob = probs[i]
	cum_prob += prob
	if cum_prob > z:
	return i

	class Softmax_Annealing_Bandit_Algorithm:
	def __init__(self, temperature, counts, values):
	self.counts = counts
	self.values = values
	return

	def initialize(self, n_arms):
	self.counts = [0 for col in range(n_arms)]
	self.values = [0.0 for col in range(n_armsb)]
	return

	def select_arm(self):
	t = sum(self.values) + 1
	temperature = 1/math.log(t + 0.0000001)
	z = sum([math.exp(v/ temperature) for v in self.values])
	probs = [math.exp(v/temperature)/z for v in self.values]
	return categorical_draw(probs)

	def update(self,chosen_arm,reward):
	self.counts[chosen_arm] = self.counts[chosen_arm] + 1
	n = self.counts[chosen_arm]

	value = self.values[chosen_arm]
	new_value = ((n-1) / float(n)) * value + (1/float(n)) * reward
	self.values[chosen_arm] = new_value
	return