diff --git a/fenn/reproducibility.py b/fenn/reproducibility.py index ef4a11b..52c524b 100644 --- a/fenn/reproducibility.py +++ b/fenn/reproducibility.py @@ -105,9 +105,11 @@ def generate_session_id() -> str: "grass", ] - # Select words - adj = random.choice(adjectives) - noun = random.choice(nouns) + # Select words with `secrets`, not `random`: a session id must identify a + # run, and `set_seed()` reseeds the global `random` state, which would + # otherwise make every run under the same seed pick the same adjective/noun. + adj = secrets.choice(adjectives) + noun = secrets.choice(nouns) # Add a secure hex suffix (2 bytes = 4 hex chars) to ensure uniqueness hex_suffix = secrets.token_hex(2) diff --git a/tests/unit/test_reproducibility.py b/tests/unit/test_reproducibility.py index db2a41b..cb60139 100644 --- a/tests/unit/test_reproducibility.py +++ b/tests/unit/test_reproducibility.py @@ -123,6 +123,21 @@ def test_uniqueness(self): # With 2-byte hex suffix, collisions should be extremely rare assert len(ids) > 45 + def test_word_choice_is_independent_of_the_global_seed(self): + # `set_seed()` reseeds `random`; a session id must still vary run to run. + set_seed(1234) + first = generate_session_id().rsplit("_", 1)[0] + set_seed(1234) + second = generate_session_id().rsplit("_", 1)[0] + + seen = set() + for _ in range(30): + set_seed(1234) + seen.add(generate_session_id().rsplit("_", 1)[0]) + # first/second are the timestamp+words; more than one distinct value + # means the words are not pinned by the seed + assert len(seen) > 1 or first != second + def test_uses_secrets_for_hex(self): with patch( "fenn.reproducibility.secrets.token_hex", return_value="abcd"