From 16a83aebaed410c771747bcc7ecca6aebe2d02fb Mon Sep 17 00:00:00 2001 From: ZainnQureshii <43629888+ZainnQureshii@users.noreply.github.com> Date: Mon, 28 Sep 2026 14:12:47 +0200 Subject: [PATCH] fix: make q_learning choose_action doctest deterministic The doctest assigned EPSILON = 0.0 in doctest's copy of the module globals, so choose_action() still read the module's EPSILON = 0.2 and explored at random about one run in ten. Patch random.random instead so the greedy branch is always taken. Co-Authored-By: Claude Opus 5.5 --- machine_learning/q_learning.py | 8 +++----- 1 file changed, 3 insertions(+), 5 deletions(-) diff --git a/machine_learning/q_learning.py b/machine_learning/q_learning.py index 4b10737b945f..4f9cee09eb00 100644 --- a/machine_learning/q_learning.py +++ b/machine_learning/q_learning.py @@ -63,14 +63,12 @@ def choose_action(state: State, available_actions: list[int]) -> int: """ Choose action using epsilon-greedy policy. + >>> from unittest.mock import patch >>> q_table.clear() - >>> old_epsilon = EPSILON - >>> EPSILON = 0.0 >>> q_table[(0, 0)][1] = 1.0 >>> q_table[(0, 0)][2] = 0.5 - >>> result = choose_action((0, 0), [1, 2]) - >>> EPSILON = old_epsilon # Restore - >>> result + >>> with patch.object(random, "random", return_value=0.99): # never explore + ... choose_action((0, 0), [1, 2]) 1 """ global EPSILON