It is difficult to know which move provides the most useful information when facing a complex situation.
It looks at different possible actions and calculates which one will most effectively clear up uncertainty about the current situation.
It allows for making decisions based on what will actually teach you the most about your next steps.
It was run inside an isolated container with no network access. This is the exact command and the real output it produced — captured process output, not written by a model.
$ python3 dsa_module_v2.py Best action from state A: 1
A screenshot of that run.
A clean run proves this does what is shown above, in a CPU-only sandbox. It is a small research demo — not a production tool, and nothing here was published anywhere.
All of it — 61 lines, one file, standard library only.
# dsa_module_v2.py
import math
def entropy(probabilities):
"""Calculate Shannon entropy for a probability distribution."""
return -sum(p * math.log2(p) for p in probabilities if p > 0)
def dynamic_state_action_entropy(transition_model, current_state):
"""Calculate entropy for each action and return the best one."""
best_action = None
min_entropy = float('inf')
# Find all available actions for the current state
actions = set((state, action) for (state, action) in transition_model.keys() if state == current_state)
for state_action in actions:
action = state_action[1]
next_state_probs = list(transition_model[state_action].values())
# Handle zero probabilities and validate distribution
if len(next_state_probs) == 0 or abs(sum(next_state_probs) - 1) > 1e-6:
continue
current_entropy = entropy(next_state_probs)
if current_entropy < min_entropy:
min_entropy = current_entropy
best_action = action
return best_action
def calculate_einformation_gain(transition_model, current_state):
"""Calculate Expected Information Gain (EIG) for each action."""
action_eig = {}
# Prior entropy of current state (assuming uniform distribution for demonstration)
prior_entropy = entropy([1.0]) # Current state is known, so prior entropy is 0
actions = set(a[1] for a in transition_model.keys() if a[0] == current_state)
for action in actions:
state_action = (current_state, action)
next_state_probs = list(transition_model[state_action].values())
if not next_state_probs or abs(sum(next_state_probs) - 1) > 1e-6:
continue
# Calculate posterior entropy for this action's next states
posterior_entropy = entropy(next_state_probs)
# Expected Information Gain = Prior Entropy - Expected Posterior Entropy
# Since prior_entropy is 0 (current state is known), EIG = -posterior_entropy
eig = -posterior_entropy
action_eig[action] = eig
return action_eig
if __name__ == "__main__":
# Sample transition probability model
transition_model = {
('A', 0): {'A': 0.7, 'B': 0.3},
('A', 1): {'B': 0.9, 'A': 0.1},
('B', 0): {'A': 0.4, 'B': 0.6},
('B', 1): {'A': 0.5, 'B': 0.5},
}
current_state = 'A'
# Original entropy-based best action
best_entropy_action = dynamic_state_action_entropy(transition_model, current_state)
# New EIG-based calculation
eig_scores = calculate_einformation_gain(transition_model, current_state)
best_eig_action = max(eig_scores, key=eig_scores.get)
print(f"Best action (entropy minimization): {best_entropy_action}")
print(f"Best action (EIG maximization): {best_eig_action}")
print(f"EIG scores: {eig_scores}")