diff --git a/pearl/policy_learners/contextual_bandits/linear_bandit.py b/pearl/policy_learners/contextual_bandits/linear_bandit.py index ff0e1cd8..606011d8 100644 --- a/pearl/policy_learners/contextual_bandits/linear_bandit.py +++ b/pearl/policy_learners/contextual_bandits/linear_bandit.py @@ -150,7 +150,7 @@ def learn_batch(self, batch: TransitionBatch) -> dict[str, Any]: if batch.weight is not None else torch.ones_like(expected_values) ) - x = torch.cat([batch.state, batch.action], dim=1) + x = torch.cat([batch.state, torch.squeeze(batch.action, dim=1)], dim=1) self.model.learn_batch( x=x, y=batch.reward,