diff --git a/backend/src/backend/app/generators/sampling.py b/backend/src/backend/app/generators/sampling.py index 2a54563..566e7e7 100644 --- a/backend/src/backend/app/generators/sampling.py +++ b/backend/src/backend/app/generators/sampling.py @@ -40,10 +40,9 @@ def super_number_max(self) -> int: ... @dataclass(frozen=True) class WeightedPool: - """A number→probability distribution weighted by an entry score.""" + """A number→weight distribution used for weighted sampling (GEN-009).""" - probabilities: dict[int, float] - score: float + weights: dict[int, float] def sample_combinations( @@ -56,12 +55,13 @@ def sample_combinations( ) -> list[tuple[list[int], int]]: """Generate ``count`` unique valid ``(combination, super_balota)`` pairs. - For each pool, weighted sampling uses ``rng.choices`` with weights derived - from the pool's probability map × entry score. Invalid or duplicate combos - trigger resampling. On each ACCEPTED combination the Superbalota is drawn - ONCE from the same ``isolated_rng(seed)`` stream over ``sb_marginal`` - (D1: post-acceptance draw keeps stream consumption independent of rejection - counts), and the full pair is legality-gated pre-append (D5/R1). + For each pool, weighted sampling uses ``rng.choices`` with the pool's + precomputed ``weights`` map (F5 × cold boost, GEN-009). Invalid or + duplicate combos trigger resampling. On each ACCEPTED combination the + Superbalota is drawn ONCE from the same ``isolated_rng(seed)`` stream over + ``sb_marginal`` (D1: post-acceptance draw keeps stream consumption + independent of rejection counts), and the full pair is legality-gated + pre-append (D5/R1). ``sb_marginal`` maps candidate SB values to relative weights; when ``None`` a uniform distribution over the configured SB range is used. On @@ -83,9 +83,9 @@ def sample_combinations( results: list[tuple[list[int], int]] = [] for pool in pools: - # Build weighted pool: number → (probability × score) - numbers = sorted(pool.probabilities.keys()) - weights = [pool.probabilities[n] * pool.score for n in numbers] + # Build weighted pool directly from the precomputed weights map. + numbers = sorted(pool.weights.keys()) + weights = [float(pool.weights[n]) for n in numbers] needed = count - len(results) for _ in range(needed): diff --git a/backend/src/backend/app/generators/version.py b/backend/src/backend/app/generators/version.py index 4bdace3..8b71c22 100644 --- a/backend/src/backend/app/generators/version.py +++ b/backend/src/backend/app/generators/version.py @@ -1,10 +1,12 @@ """Generator version constant — bumped on algorithm changes only (GEN-009). -2.0.0 (D6): SuperBalota sampling joined the numbers' isolated RNG stream -(R2/D1), changing stream consumption — output identity (``generation_seed`` / -``snapshot_fingerprint``) differs from every pre-2.0.0 value. +3.0.0 (GEN-009 remix): dropped the meta prediction-chain ``entry.score`` from +sampling/allocation. Per-number weights are now F5 frequency × cold-coverage +boost (PM-08), computed transparently from draw history. Output identity +(``generation_seed`` / ``snapshot_fingerprint``) differs from every pre-3.0.0 +value. """ from __future__ import annotations -GENERATOR_VERSION: str = "2.0.0" +GENERATOR_VERSION: str = "3.0.0" diff --git a/backend/src/backend/app/generators/weighting.py b/backend/src/backend/app/generators/weighting.py new file mode 100644 index 0000000..a425c9b --- /dev/null +++ b/backend/src/backend/app/generators/weighting.py @@ -0,0 +1,34 @@ +"""Generation weight construction (GEN-09): legitimate statistical levers only. + +Builds the per-number sampling weights consumed by ``sampling.WeightedPool``. +The remix replaces the retired ``entry.score`` meta chain: a number's weight is +now a transparent function of draw history (F5 frequency × an optional cold +coverage boost), never a meta model score. +""" + +from __future__ import annotations + +from decimal import Decimal + +COLD_BOOST: Decimal = Decimal("1.5") +"""Multiplier applied to COLD numbers to lift their sampling weight (PM-08).""" + + +def build_weights( + probabilities: dict[int, float], + coverage: dict[int, str], + *, + cold_boost: Decimal = COLD_BOOST, +) -> dict[int, float]: + """Combine F5 probabilities with a cold-coverage boost into final weights. + + ``probabilities`` is the F5 number→probability map (probability engine). + ``coverage`` maps each number to ``"cold"``/``"normal"``/``"hot"``; only + ``"cold"`` numbers receive the ``cold_boost`` multiplier (lever C). The + result preserves the key set of ``probabilities``. + """ + boost = float(cold_boost) + return { + number: probability * (boost if coverage.get(number) == "cold" else 1.0) + for number, probability in probabilities.items() + } diff --git a/backend/src/backend/app/services/gen_service.py b/backend/src/backend/app/services/gen_service.py index 34b7be4..7feeebb 100644 --- a/backend/src/backend/app/services/gen_service.py +++ b/backend/src/backend/app/services/gen_service.py @@ -31,8 +31,12 @@ from backend.app.generators.snapshot_store import GenSnapshotStore from backend.app.generators.validation import validate_combination from backend.app.generators.version import GENERATOR_VERSION +from backend.app.generators.weighting import build_weights from backend.app.models.gen_snapshot import GenSnapshot +from backend.app.services.probability_service import _classify_coverage +from backend.app.repositories.stat_payload_repository import StatPayloadRepository from backend.app.services.errors import GenServiceError +from backend.app.statistics.engine import frequency DEFAULT_COUNT: int = 10 """Default combination count when not provided (GEN-002).""" @@ -120,22 +124,23 @@ def generate( ) -> GenerationResult: """Generate (or idempotently return) a lottery combination snapshot. - Pipeline (GEN-001): resolve the F12 selection → validate count → - allocate via the micro-unit rule (GEN-004) → load the F5 distribution → - load the SB historical marginal (R2/D2) → sample ``(combo, sb)`` pairs - with ``isolated_rng`` (GEN-005, D1) → compute the selection-weighted - score (R3/D3) → gate legality pre-persist (R1/D5) → fingerprint → - persist a NEW active version atomically. Same inputs reproduce the - identical snapshot including Superbalotas and scores (GEN-008); a - duplicate non-active fingerprint is a conflict (GEN-013). + Pipeline (GEN-001, GEN-009 remix): resolve the selection scope → validate + count → load the F5 distribution → derive per-number weights from F5 × + cold-coverage boost (PM-08) → sample ``(combo, sb)`` pairs with + ``isolated_rng`` (GEN-005, D1) → score each combo with its transparent + mean sampling weight (R3/D3) → gate legality pre-persist (R1/D5) → + fingerprint → persist a NEW active version atomically. The meta + prediction-chain scores are NOT used (audit-proven zero effect). Same + inputs reproduce the identical snapshot including Superbalotas and scores + (GEN-008); a duplicate non-active fingerprint is a conflict (GEN-013). """ lottery = self._resolve_lottery(lottery_id) effective_count = DEFAULT_COUNT if count is None else count self._validate_count(effective_count) selection = self._resolve_selection(lottery_id, selection_id) - entry_rows = self._read_selection_entries(selection.id) - entries = [SelectionEntry(score=row.score, rank=row.rank) for row in entry_rows] - allocations = allocate_count(entries, effective_count) + # GEN-09 remix: a single allocation unit carries the whole count; the + # per-number weights (not a meta entry score) drive sampling below. + allocations = allocate_count([SelectionEntry(score=1.0, rank=0)], effective_count) effective_seed = ( generation_seed(selection.fingerprint, lottery_id, effective_count, GENERATOR_VERSION) @@ -161,28 +166,29 @@ def generate( ) probabilities = self._load_distribution(lottery_id) - sb_marginal = self._load_sb_marginal(lottery) - pools = [ - WeightedPool(probabilities=probabilities, score=entries[i].score) - for i, allocated in allocations - if allocated > 0 + # Coverage (COLD/NORMAL/HOT) from draw history; only COLD numbers get a + # boost (PM-08). With no imported draw numbers this map is all "normal". + draws = [ + numbers + for _dn, numbers, _j, _w in StatPayloadRepository(self._session).iter_draws(lottery.id) ] + coverage = _classify_coverage( + frequency(draws), + lottery.min_number, + lottery.max_number, + lottery.numbers_to_select, + ) + weights = build_weights(probabilities, coverage) + sb_marginal = self._load_sb_marginal(lottery) + pools = [WeightedPool(weights=weights) for _i, allocated in allocations if allocated > 0] sampled = sample_combinations(effective_seed, pools, effective_count, lottery, sb_marginal) - # D3: score = entry_score × mean(P(n)) computed where pools carry both - # inputs. Pools consume the sample sequentially in allocation order, so - # each pair's provenance (entry score) is recovered positionally. + # D3/R3: score is the transparent mean sampling weight of the combo's + # numbers (F5 × cold boost) — no meta entry score involved. scored: list[tuple[list[int], int, float]] = [] - idx = 0 - for entry_index, allocated in allocations: - if allocated <= 0: - continue - entry_score = float(entries[entry_index].score) - for _ in range(allocated): - combo, sb = sampled[idx] - idx += 1 - mean_p = sum(probabilities.get(n, 0.0) for n in combo) / len(combo) - scored.append((combo, sb, round(entry_score * mean_p, 6))) + for combo, sb in sampled: + mean_w = sum(weights.get(n, 0.0) for n in combo) / len(combo) + scored.append((combo, sb, round(mean_w, 6))) # R1/D5: legality assert before anything is persisted. for combo, sb, _score in scored: @@ -341,23 +347,6 @@ def _resolve_selection(self, lottery_id: int, selection_id: int | None) -> Any: ) return selection - def _read_selection_entries(self, selection_id: int) -> list[Any]: - """Read the scored entries of a selection; ``GEN_NO_SELECTION`` when empty.""" - from backend.app.models.meta_selection_entry import MetaSelectionEntry - - stmt = ( - select(MetaSelectionEntry) - .where(MetaSelectionEntry.selection_id == selection_id) - .order_by(MetaSelectionEntry.rank) - ) - rows = list(self._session.execute(stmt).scalars().all()) - if not rows: - raise GenServiceError( - GenServiceError.GEN_NO_SELECTION, - f"selection {selection_id} has no entries", - ) - return rows - def _load_distribution(self, lottery_id: int) -> dict[int, float]: """Read the active F5 number→probability map; ``GEN_NO_DISTRIBUTION`` absent. diff --git a/backend/tests/gen/test_gen_generate.py b/backend/tests/gen/test_gen_generate.py index 87f8522..b66a42d 100644 --- a/backend/tests/gen/test_gen_generate.py +++ b/backend/tests/gen/test_gen_generate.py @@ -153,17 +153,17 @@ def test_generation_byte_reproducible_including_sb(self, db: Session, seed_gen_d assert pairs_a == pairs_b def test_score_formula_selection_weighted(self, db: Session, seed_gen_data) -> None: - """score == round(entry_score × mean(P(n)), 6) with uniform P=0.05 (D3).""" + """score == round(mean(weights), 6); uniform P=0.05 → score 0.05 (GEN-009).""" ids = seed_gen_data(scores=(0.7, 0.3)) result = _service(db).generate(lottery_id=ids["lottery_id"], count=1) - expected = round(0.7 * ((0.05 * 6) / 6), 6) + expected = round(0.05, 6) assert result.combinations[0].score == expected - def test_score_reflects_both_entry_weights(self, db: Session, seed_gen_data) -> None: - """count=10 spans both entries → scores ∈ {0.7×0.05, 0.3×0.05} rounded (D3).""" + def test_score_reflects_weights_not_entries(self, db: Session, seed_gen_data) -> None: + """count=10 → all scores equal the mean F5 weight (0.05), entry scores ignored (GEN-009).""" ids = seed_gen_data(scores=(0.7, 0.3)) result = _service(db).generate(lottery_id=ids["lottery_id"], count=10) - allowed = {round(0.7 * 0.05, 6), round(0.3 * 0.05, 6)} + allowed = {round(0.05, 6)} for row in result.combinations: assert row.score in allowed diff --git a/backend/tests/gen/test_identity.py b/backend/tests/gen/test_identity.py index 5d13a42..fc8ea7a 100644 --- a/backend/tests/gen/test_identity.py +++ b/backend/tests/gen/test_identity.py @@ -37,9 +37,9 @@ PRE_CHANGE_SEED = 297872213468358109463619875798332175481 PRE_CHANGE_SNAPSHOT_FINGERPRINT = "ddd2dbe5c6c8002067c2191e118caf1902c9c2e9b9ee2616136119bda3feb42c" -# Regenerated v2.0.0 golden vectors (D6) — locked atomically with the bump. -GOLDEN_SEED_V2 = 275000497823893291003335902595522194545 -GOLDEN_SNAPSHOT_FINGERPRINT_V2 = "3a767d0a41419b566bc718e64821d59a698321355b4d0da533b0435b22b48373" +# Regenerated v3.0.0 golden vectors (GEN-009 remix) — locked atomically with the bump. +GOLDEN_SEED_V3 = 198708973693754007559308447754739185303 +GOLDEN_SNAPSHOT_FINGERPRINT_V3 = "570bb8de864fb28b85027badc4552ebaa2e4c3bf6504b4cef649b1c056542ae7" class TestGenerationSeed: @@ -68,13 +68,13 @@ def test_locked_pre_change_golden_vector(self) -> None: ) assert seed == PRE_CHANGE_SEED - def test_locked_v2_golden_vector(self) -> None: - """Regenerated golden under the bumped identity (D6).""" - assert GENERATOR_VERSION == "2.0.0" + def test_locked_v3_golden_vector(self) -> None: + """Regenerated golden under the bumped identity (GEN-009 remix).""" + assert GENERATOR_VERSION == "3.0.0" seed = generation_seed( GOLDEN_SELECTION_FINGERPRINT, GOLDEN_LOTTERY_ID, GOLDEN_COUNT, GENERATOR_VERSION ) - assert seed == GOLDEN_SEED_V2 + assert seed == GOLDEN_SEED_V3 @pytest.mark.parametrize( ("selection_fingerprint", "lottery_id", "count", "version"), @@ -157,16 +157,16 @@ def test_locked_pre_change_golden_vector(self) -> None: ) assert fp == PRE_CHANGE_SNAPSHOT_FINGERPRINT - def test_locked_v2_golden_vector(self) -> None: - """Regenerated fingerprint under the bumped identity (D6).""" - assert GENERATOR_VERSION == "2.0.0" - seed_v2 = generation_seed( + def test_locked_v3_golden_vector(self) -> None: + """Regenerated fingerprint under the bumped identity (GEN-009 remix).""" + assert GENERATOR_VERSION == "3.0.0" + seed_v3 = generation_seed( GOLDEN_SELECTION_FINGERPRINT, GOLDEN_LOTTERY_ID, GOLDEN_COUNT, GENERATOR_VERSION ) fp = snapshot_fingerprint( - GOLDEN_LOTTERY_ID, GOLDEN_SELECTION_ID, GOLDEN_COUNT, seed_v2, GENERATOR_VERSION + GOLDEN_LOTTERY_ID, GOLDEN_SELECTION_ID, GOLDEN_COUNT, seed_v3, GENERATOR_VERSION ) - assert fp == GOLDEN_SNAPSHOT_FINGERPRINT_V2 + assert fp == GOLDEN_SNAPSHOT_FINGERPRINT_V3 @pytest.mark.parametrize( ("lottery_id", "selection_id", "count", "seed", "version"), @@ -228,9 +228,9 @@ def test_matches_canonical_sha256_formula(self) -> None: class TestVersionBumpAliasingGuard: """D6/R2 — v2 outputs MUST NOT alias any pre-change fixture fingerprint.""" - def test_v2_fingerprint_differs_from_pre_change(self) -> None: + def test_v3_fingerprint_differs_from_pre_change(self) -> None: """Same canonical inputs → bump moves the fingerprint away from legacy.""" - assert GENERATOR_VERSION == "2.0.0" + assert GENERATOR_VERSION == "3.0.0" seed_v2 = generation_seed( GOLDEN_SELECTION_FINGERPRINT, GOLDEN_LOTTERY_ID, GOLDEN_COUNT, GENERATOR_VERSION ) diff --git a/backend/tests/gen/test_sampling.py b/backend/tests/gen/test_sampling.py index df0c84a..083cdf5 100644 --- a/backend/tests/gen/test_sampling.py +++ b/backend/tests/gen/test_sampling.py @@ -42,10 +42,7 @@ def cfg(self) -> LotteryConfig: def _make_pool(self, n: int = 49, weight: float = 1.0) -> WeightedPool: """Create a uniform distribution over numbers 1..n.""" - return WeightedPool( - probabilities={i: weight for i in range(1, n + 1)}, - score=1.0, - ) + return WeightedPool(weights={i: weight for i in range(1, n + 1)}) def test_determinism(self, cfg: LotteryConfig) -> None: """Same seed → identical output (GEN-005, NFR-GEN-01).""" @@ -91,13 +88,13 @@ def test_sb_drawn_on_same_stream_after_acceptance(self, cfg: LotteryConfig) -> N consume SB draws (post-acceptance draw). White-box replay of the documented algorithm with a single random.Random(seed) instance. """ - pool = WeightedPool(probabilities={i: 1.0 for i in range(1, 50)}, score=1.0) + pool = WeightedPool(weights={i: 1.0 for i in range(1, 50)}) count = 4 result = sample_combinations(1234, [pool], count, cfg) rng = random.Random(1234) - numbers = sorted(pool.probabilities) - weights = [pool.probabilities[n] * pool.score for n in numbers] + numbers = sorted(pool.weights) + weights = [pool.weights[n] for n in numbers] sb_numbers = list(range(cfg.super_number_min, cfg.super_number_max + 1)) sb_weights = [1.0 / len(sb_numbers)] * len(sb_numbers) generated: set[frozenset[int]] = set() @@ -140,10 +137,7 @@ def test_max_attempts_exhaustion(self) -> None: super_number_min=1, super_number_max=9, ) - pool = WeightedPool( - probabilities={i: 1.0 for i in range(1, 7)}, - score=1.0, - ) + pool = WeightedPool(weights={i: 1.0 for i in range(1, 7)}) with pytest.raises(GenServiceError) as exc_info: sample_combinations(42, [pool], 5, tiny_cfg, max_attempts=3) assert exc_info.value.code == "GEN_SPACE_EXHAUSTED" @@ -157,15 +151,9 @@ def test_no_duplicates_in_output(self, cfg: LotteryConfig) -> None: def test_multiple_pools(self, cfg: LotteryConfig) -> None: """Multiple weighted pools produce combinations from each.""" - pool1 = WeightedPool( - probabilities={i: 1.0 for i in range(1, 50)}, - score=0.7, - ) - pool2 = WeightedPool( - probabilities={i: 1.0 for i in range(1, 50)}, - score=0.3, - ) - # 3 from pool1, 2 from pool2 = 5 total + pool1 = WeightedPool(weights={i: 0.7 for i in range(1, 50)}) + pool2 = WeightedPool(weights={i: 0.3 for i in range(1, 50)}) + # The first pool satisfies the full count; pool2 is unused. results = sample_combinations(42, [pool1, pool2], 5, cfg) assert len(results) == 5 @@ -173,8 +161,7 @@ def test_score_influences_distribution(self) -> None: """Higher score weight biases number selection.""" # Pool with strong weight on low numbers pool_low = WeightedPool( - probabilities={i: (10.0 if i <= 10 else 0.1) for i in range(1, 50)}, - score=1.0, + weights={i: (10.0 if i <= 10 else 0.1) for i in range(1, 50)} ) cfg = LotteryConfig( numbers_to_select=3, diff --git a/backend/tests/gen/test_types.py b/backend/tests/gen/test_types.py index 8bd3ad8..e305fb3 100644 --- a/backend/tests/gen/test_types.py +++ b/backend/tests/gen/test_types.py @@ -94,10 +94,10 @@ def test_inequality_different_count(self) -> None: class TestGeneratorVersion: """GENERATOR_VERSION constant — GEN-009 determinism + D6 major bump.""" - def test_version_is_2_0_0(self) -> None: - # D6: stream-consumption change (SB on the shared stream) is a breaking - # output-identity change → GENERATOR_VERSION "2.0.0". - assert GENERATOR_VERSION == "2.0.0" + def test_version_is_3_0_0(self) -> None: + # GEN-009 remix: dropping meta entry.score is a breaking output-identity + # change → GENERATOR_VERSION "3.0.0". + assert GENERATOR_VERSION == "3.0.0" def test_version_is_string(self) -> None: assert isinstance(GENERATOR_VERSION, str) diff --git a/backend/tests/generators/test_weighting.py b/backend/tests/generators/test_weighting.py new file mode 100644 index 0000000..e43d28b --- /dev/null +++ b/backend/tests/generators/test_weighting.py @@ -0,0 +1,28 @@ +"""Unit tests for generator weight construction (GEN-09).""" + +from __future__ import annotations + +from backend.app.generators.weighting import COLD_BOOST, build_weights + + +def test_build_weights_applies_cold_boost_only() -> None: + probabilities = {1: 0.5, 2: 0.5} + coverage = {1: "cold", 2: "normal"} + weights = build_weights(probabilities, coverage) + assert weights[1] == 0.5 * float(COLD_BOOST) + assert weights[2] == 0.5 + + +def test_build_weights_ignores_hot_and_normal() -> None: + probabilities = {1: 0.3, 2: 0.7} + coverage = {1: "normal", 2: "hot"} + weights = build_weights(probabilities, coverage) + assert weights == {1: 0.3, 2: 0.7} + + +def test_build_weights_preserves_key_set() -> None: + probabilities = {1: 0.2, 2: 0.3, 3: 0.5} + coverage = {1: "cold", 2: "normal", 3: "hot"} + weights = build_weights(probabilities, coverage) + assert set(weights) == {1, 2, 3} + assert weights[1] == 0.2 * float(COLD_BOOST)