- numGibbs no longer part of EntityParameter

- refactored forward and reconstruct
- conditionally use optimizer for training
This commit is contained in:
2025-12-19 17:15:10 +01:00
parent a47922cb1c
commit d3c9fe4681
4 changed files with 124 additions and 41 deletions
+85 -11
View File
@@ -15,6 +15,7 @@ class TrainingParams:
self.do_gibbs_sample_visible = False
self.do_gibbs_sample_hidden = False
self.do_batch_sample = False
self.num_gibbs_samples = 1
@classmethod
def from_dict(cls, params: dict):
@@ -28,18 +29,19 @@ class TrainingParams:
obj.do_gibbs_sample_visible = params["gibbsDoSampleVisible"]
obj.do_gibbs_sample_hidden = params["gibbsDoSampleHidden"]
obj.do_batch_sample = params["doSampleBatch"]
obj.num_gibbs_samples = params["numGibbs"]
return obj
def cd_jens(entity: Entity, v_states: Mat, params: TrainingParams):
v_probs = prob(v_states)
h_states = entity.v_to_ph(v_states)
h_states = entity.forward(v_states)
h_probs = h_states
if entity.params.do_gaussian_hidden:
h_states += gaussian(h_states.shape)
else:
h_probs = entity.v_to_ph(v_states)
h_probs = entity.forward(v_states)
if params.do_rao_blackwell:
h_states = h_probs
else:
@@ -51,24 +53,24 @@ def cd_jens(entity: Entity, v_states: Mat, params: TrainingParams):
dbh = np.sum(h_states, 0)
# Gibbs sampling with training params
for i in range(entity.params.num_gibbs_samples):
for i in range(params.num_gibbs_samples):
if params.do_gibbs_sample_hidden:
v_probs = entity.h_to_pv(sample(h_probs))
v_probs = entity.reconstruct(sample(h_probs))
else:
v_probs = entity.h_to_pv(h_probs)
v_probs = entity.reconstruct(h_probs)
# Create hidden representation given v
if params.do_gibbs_sample_visible:
h_probs = entity.v_to_ph(sample(v_probs))
h_probs = entity.forward(sample(v_probs))
else:
h_probs = entity.v_to_ph(v_probs)
h_probs = entity.forward(v_probs)
# Update weights (negative phase)
dw -= np.dot(np.transpose(v_probs), h_probs)
dbv -= np.sum(v_probs, 0)
dbh -= np.sum(h_probs, 0)
return dw, dbv, dbh
return dw, dbv, dbh, h_probs
def train(entity: Entity, batch: Mat, params: TrainingParams, status: Status, cd_func: Callable = cd_jens):
training_remain = batch.shape[0]
@@ -97,7 +99,7 @@ def train(entity: Entity, batch: Mat, params: TrainingParams, status: Status, cd
for epochs in range(params.num_epochs):
# Contrastive divergence learning: calculate gradients
dwhv, dbv, dbh = cd_func(entity, v_states, params)
dwhv, dbv, dbh, _ = cd_func(entity, v_states, params)
# Adjust weight and biases
kl = params.learning_rate/batch_size
@@ -112,7 +114,7 @@ def train(entity: Entity, batch: Mat, params: TrainingParams, status: Status, cd
# check if status update is needed
if status.want_report(round(progress)):
# Calculate error
err_rms = rms_error_accu(mini_batch - entity.h_to_pv(entity.v_to_ph(v_states)))
err_rms = rms_error_accu(batch - entity.reconstruct(entity.forward(batch)))
if not status.on_change({"progress": {"value": round(progress), "unit": "%"},
"err_rms": {"value": err_rms, "unit": ""}}):
keep_running = False
@@ -121,6 +123,78 @@ def train(entity: Entity, batch: Mat, params: TrainingParams, status: Status, cd
progress += d_progress*batch_size_remain
# Update final status
err_rms = rms_error_accu(batch - entity.h_to_pv(entity.v_to_ph(batch)))
err_rms = rms_error_accu(batch - entity.reconstruct(entity.forward(batch)))
status.on_change({"progress": {"value": round(progress), "unit": "%"},
"err_rms_total": {"value": err_rms, "unit": ""}})
class Optimizer:
def __init__(self, entity: Entity, params: TrainingParams, cd_func: Callable = cd_jens):
self.inc_bv = np.zeros(entity.state.b_v.shape)
self.inc_bh = np.zeros(entity.state.b_h.shape)
self.inc_whv = np.zeros(entity.state.w_hv.shape)
self.entity = entity
self.params = params
self.cd = cd_func
def __call__(self, batch: Mat, status: Status):
return self.process(batch, status)
def process(self, batch: Mat, status: Status):
status.on_change()
d_progress = 100.0 / self.params.num_epochs
progress = 0
v_states = batch
if self.params.do_batch_sample:
v_states = sample(batch)
forward = None
for epochs in range(self.params.num_epochs):
forward = self.epoch(v_states)
# check if status update is needed
if status.want_report(round(progress)):
# Calculate error
result = self.entity.reconstruct(forward)
err_rms = rms_error_accu(batch - result)
if not status.on_change({"progress": {"value": round(progress), "unit": "%"},
"err_rms": {"value": err_rms, "unit": ""}}):
break
progress += d_progress
err_rms = rms_error_accu(batch - self.entity.reconstruct(forward))
status.on_change({"progress": {"value": round(progress), "unit": "%"},
"err_rms": {"value": err_rms, "unit": ""}})
return forward
def epoch(self, batch: Mat):
forward = None
training_remain = batch.shape[0]
batch_row_index = 0
while training_remain > 0:
batch_size = min(self.params.mini_batch_size, training_remain)
if batch_size == 0:
batch_size = training_remain
mini_batch = batch[batch_row_index:batch_row_index + batch_size]
# Contrastive divergence learning: calculate gradients
dwhv, dbv, dbh, forward = self.cd(self.entity, mini_batch, self.params)
# Adjust weight and biases
kl = self.params.learning_rate / batch_size
inc_bv = self.params.momentum * self.inc_bv + kl * dbv
inc_bh = self.params.momentum * self.inc_bh + kl * dbh
inc_whv = self.params.momentum * self.inc_whv + kl * dwhv - self.params.weight_decay * self.entity.state.w_hv
self.entity.state.b_v += inc_bv
self.entity.state.b_h += inc_bh
self.entity.state.w_hv += inc_whv
training_remain -= batch_size
batch_row_index += batch_size
return forward