Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
24 changes: 12 additions & 12 deletions backend/src/backend/app/generators/sampling.py
Original file line number Diff line number Diff line change
Expand Up @@ -40,10 +40,9 @@ def super_number_max(self) -> int: ...

@dataclass(frozen=True)
class WeightedPool:
"""A number→probability distribution weighted by an entry score."""
"""A number→weight distribution used for weighted sampling (GEN-009)."""

probabilities: dict[int, float]
score: float
weights: dict[int, float]


def sample_combinations(
Expand All @@ -56,12 +55,13 @@ def sample_combinations(
) -> list[tuple[list[int], int]]:
"""Generate ``count`` unique valid ``(combination, super_balota)`` pairs.

For each pool, weighted sampling uses ``rng.choices`` with weights derived
from the pool's probability map × entry score. Invalid or duplicate combos
trigger resampling. On each ACCEPTED combination the Superbalota is drawn
ONCE from the same ``isolated_rng(seed)`` stream over ``sb_marginal``
(D1: post-acceptance draw keeps stream consumption independent of rejection
counts), and the full pair is legality-gated pre-append (D5/R1).
For each pool, weighted sampling uses ``rng.choices`` with the pool's
precomputed ``weights`` map (F5 × cold boost, GEN-009). Invalid or
duplicate combos trigger resampling. On each ACCEPTED combination the
Superbalota is drawn ONCE from the same ``isolated_rng(seed)`` stream over
``sb_marginal`` (D1: post-acceptance draw keeps stream consumption
independent of rejection counts), and the full pair is legality-gated
pre-append (D5/R1).

``sb_marginal`` maps candidate SB values to relative weights; when ``None``
a uniform distribution over the configured SB range is used. On
Expand All @@ -83,9 +83,9 @@ def sample_combinations(
results: list[tuple[list[int], int]] = []

for pool in pools:
# Build weighted pool: number → (probability × score)
numbers = sorted(pool.probabilities.keys())
weights = [pool.probabilities[n] * pool.score for n in numbers]
# Build weighted pool directly from the precomputed weights map.
numbers = sorted(pool.weights.keys())
weights = [float(pool.weights[n]) for n in numbers]

needed = count - len(results)
for _ in range(needed):
Expand Down
10 changes: 6 additions & 4 deletions backend/src/backend/app/generators/version.py
Original file line number Diff line number Diff line change
@@ -1,10 +1,12 @@
"""Generator version constant — bumped on algorithm changes only (GEN-009).

2.0.0 (D6): SuperBalota sampling joined the numbers' isolated RNG stream
(R2/D1), changing stream consumption — output identity (``generation_seed`` /
``snapshot_fingerprint``) differs from every pre-2.0.0 value.
3.0.0 (GEN-009 remix): dropped the meta prediction-chain ``entry.score`` from
sampling/allocation. Per-number weights are now F5 frequency × cold-coverage
boost (PM-08), computed transparently from draw history. Output identity
(``generation_seed`` / ``snapshot_fingerprint``) differs from every pre-3.0.0
value.
"""

from __future__ import annotations

GENERATOR_VERSION: str = "2.0.0"
GENERATOR_VERSION: str = "3.0.0"
34 changes: 34 additions & 0 deletions backend/src/backend/app/generators/weighting.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,34 @@
"""Generation weight construction (GEN-09): legitimate statistical levers only.

Builds the per-number sampling weights consumed by ``sampling.WeightedPool``.
The remix replaces the retired ``entry.score`` meta chain: a number's weight is
now a transparent function of draw history (F5 frequency × an optional cold
coverage boost), never a meta model score.
"""

from __future__ import annotations

from decimal import Decimal

COLD_BOOST: Decimal = Decimal("1.5")
"""Multiplier applied to COLD numbers to lift their sampling weight (PM-08)."""


def build_weights(
probabilities: dict[int, float],
coverage: dict[int, str],
*,
cold_boost: Decimal = COLD_BOOST,
) -> dict[int, float]:
"""Combine F5 probabilities with a cold-coverage boost into final weights.

``probabilities`` is the F5 number→probability map (probability engine).
``coverage`` maps each number to ``"cold"``/``"normal"``/``"hot"``; only
``"cold"`` numbers receive the ``cold_boost`` multiplier (lever C). The
result preserves the key set of ``probabilities``.
"""
boost = float(cold_boost)
return {
number: probability * (boost if coverage.get(number) == "cold" else 1.0)
for number, probability in probabilities.items()
}
81 changes: 35 additions & 46 deletions backend/src/backend/app/services/gen_service.py
Original file line number Diff line number Diff line change
Expand Up @@ -31,8 +31,12 @@
from backend.app.generators.snapshot_store import GenSnapshotStore
from backend.app.generators.validation import validate_combination
from backend.app.generators.version import GENERATOR_VERSION
from backend.app.generators.weighting import build_weights
from backend.app.models.gen_snapshot import GenSnapshot
from backend.app.services.probability_service import _classify_coverage
from backend.app.repositories.stat_payload_repository import StatPayloadRepository
from backend.app.services.errors import GenServiceError
from backend.app.statistics.engine import frequency

DEFAULT_COUNT: int = 10
"""Default combination count when not provided (GEN-002)."""
Expand Down Expand Up @@ -120,22 +124,23 @@ def generate(
) -> GenerationResult:
"""Generate (or idempotently return) a lottery combination snapshot.

Pipeline (GEN-001): resolve the F12 selection → validate count →
allocate via the micro-unit rule (GEN-004) → load the F5 distribution →
load the SB historical marginal (R2/D2) → sample ``(combo, sb)`` pairs
with ``isolated_rng`` (GEN-005, D1) → compute the selection-weighted
score (R3/D3) → gate legality pre-persist (R1/D5) → fingerprint →
persist a NEW active version atomically. Same inputs reproduce the
identical snapshot including Superbalotas and scores (GEN-008); a
duplicate non-active fingerprint is a conflict (GEN-013).
Pipeline (GEN-001, GEN-009 remix): resolve the selection scope → validate
count → load the F5 distribution → derive per-number weights from F5 ×
cold-coverage boost (PM-08) → sample ``(combo, sb)`` pairs with
``isolated_rng`` (GEN-005, D1) → score each combo with its transparent
mean sampling weight (R3/D3) → gate legality pre-persist (R1/D5) →
fingerprint → persist a NEW active version atomically. The meta
prediction-chain scores are NOT used (audit-proven zero effect). Same
inputs reproduce the identical snapshot including Superbalotas and scores
(GEN-008); a duplicate non-active fingerprint is a conflict (GEN-013).
"""
lottery = self._resolve_lottery(lottery_id)
effective_count = DEFAULT_COUNT if count is None else count
self._validate_count(effective_count)
selection = self._resolve_selection(lottery_id, selection_id)
entry_rows = self._read_selection_entries(selection.id)
entries = [SelectionEntry(score=row.score, rank=row.rank) for row in entry_rows]
allocations = allocate_count(entries, effective_count)
# GEN-09 remix: a single allocation unit carries the whole count; the
# per-number weights (not a meta entry score) drive sampling below.
allocations = allocate_count([SelectionEntry(score=1.0, rank=0)], effective_count)

effective_seed = (
generation_seed(selection.fingerprint, lottery_id, effective_count, GENERATOR_VERSION)
Expand All @@ -161,28 +166,29 @@ def generate(
)

probabilities = self._load_distribution(lottery_id)
sb_marginal = self._load_sb_marginal(lottery)
pools = [
WeightedPool(probabilities=probabilities, score=entries[i].score)
for i, allocated in allocations
if allocated > 0
# Coverage (COLD/NORMAL/HOT) from draw history; only COLD numbers get a
# boost (PM-08). With no imported draw numbers this map is all "normal".
draws = [
numbers
for _dn, numbers, _j, _w in StatPayloadRepository(self._session).iter_draws(lottery.id)
]
coverage = _classify_coverage(
frequency(draws),
lottery.min_number,
lottery.max_number,
lottery.numbers_to_select,
)
weights = build_weights(probabilities, coverage)
sb_marginal = self._load_sb_marginal(lottery)
pools = [WeightedPool(weights=weights) for _i, allocated in allocations if allocated > 0]
sampled = sample_combinations(effective_seed, pools, effective_count, lottery, sb_marginal)

# D3: score = entry_score × mean(P(n)) computed where pools carry both
# inputs. Pools consume the sample sequentially in allocation order, so
# each pair's provenance (entry score) is recovered positionally.
# D3/R3: score is the transparent mean sampling weight of the combo's
# numbers (F5 × cold boost) — no meta entry score involved.
scored: list[tuple[list[int], int, float]] = []
idx = 0
for entry_index, allocated in allocations:
if allocated <= 0:
continue
entry_score = float(entries[entry_index].score)
for _ in range(allocated):
combo, sb = sampled[idx]
idx += 1
mean_p = sum(probabilities.get(n, 0.0) for n in combo) / len(combo)
scored.append((combo, sb, round(entry_score * mean_p, 6)))
for combo, sb in sampled:
mean_w = sum(weights.get(n, 0.0) for n in combo) / len(combo)
scored.append((combo, sb, round(mean_w, 6)))

# R1/D5: legality assert before anything is persisted.
for combo, sb, _score in scored:
Expand Down Expand Up @@ -341,23 +347,6 @@ def _resolve_selection(self, lottery_id: int, selection_id: int | None) -> Any:
)
return selection

def _read_selection_entries(self, selection_id: int) -> list[Any]:
"""Read the scored entries of a selection; ``GEN_NO_SELECTION`` when empty."""
from backend.app.models.meta_selection_entry import MetaSelectionEntry

stmt = (
select(MetaSelectionEntry)
.where(MetaSelectionEntry.selection_id == selection_id)
.order_by(MetaSelectionEntry.rank)
)
rows = list(self._session.execute(stmt).scalars().all())
if not rows:
raise GenServiceError(
GenServiceError.GEN_NO_SELECTION,
f"selection {selection_id} has no entries",
)
return rows

def _load_distribution(self, lottery_id: int) -> dict[int, float]:
"""Read the active F5 number→probability map; ``GEN_NO_DISTRIBUTION`` absent.

Expand Down
10 changes: 5 additions & 5 deletions backend/tests/gen/test_gen_generate.py
Original file line number Diff line number Diff line change
Expand Up @@ -153,17 +153,17 @@ def test_generation_byte_reproducible_including_sb(self, db: Session, seed_gen_d
assert pairs_a == pairs_b

def test_score_formula_selection_weighted(self, db: Session, seed_gen_data) -> None:
"""score == round(entry_score × mean(P(n)), 6) with uniform P=0.05 (D3)."""
"""score == round(mean(weights), 6); uniform P=0.05 → score 0.05 (GEN-009)."""
ids = seed_gen_data(scores=(0.7, 0.3))
result = _service(db).generate(lottery_id=ids["lottery_id"], count=1)
expected = round(0.7 * ((0.05 * 6) / 6), 6)
expected = round(0.05, 6)
assert result.combinations[0].score == expected

def test_score_reflects_both_entry_weights(self, db: Session, seed_gen_data) -> None:
"""count=10 spans both entries → scores ∈ {0.7×0.05, 0.3×0.05} rounded (D3)."""
def test_score_reflects_weights_not_entries(self, db: Session, seed_gen_data) -> None:
"""count=10 → all scores equal the mean F5 weight (0.05), entry scores ignored (GEN-009)."""
ids = seed_gen_data(scores=(0.7, 0.3))
result = _service(db).generate(lottery_id=ids["lottery_id"], count=10)
allowed = {round(0.7 * 0.05, 6), round(0.3 * 0.05, 6)}
allowed = {round(0.05, 6)}
for row in result.combinations:
assert row.score in allowed

Expand Down
30 changes: 15 additions & 15 deletions backend/tests/gen/test_identity.py
Original file line number Diff line number Diff line change
Expand Up @@ -37,9 +37,9 @@
PRE_CHANGE_SEED = 297872213468358109463619875798332175481
PRE_CHANGE_SNAPSHOT_FINGERPRINT = "ddd2dbe5c6c8002067c2191e118caf1902c9c2e9b9ee2616136119bda3feb42c"

# Regenerated v2.0.0 golden vectors (D6) — locked atomically with the bump.
GOLDEN_SEED_V2 = 275000497823893291003335902595522194545
GOLDEN_SNAPSHOT_FINGERPRINT_V2 = "3a767d0a41419b566bc718e64821d59a698321355b4d0da533b0435b22b48373"
# Regenerated v3.0.0 golden vectors (GEN-009 remix) — locked atomically with the bump.
GOLDEN_SEED_V3 = 198708973693754007559308447754739185303
GOLDEN_SNAPSHOT_FINGERPRINT_V3 = "570bb8de864fb28b85027badc4552ebaa2e4c3bf6504b4cef649b1c056542ae7"


class TestGenerationSeed:
Expand Down Expand Up @@ -68,13 +68,13 @@ def test_locked_pre_change_golden_vector(self) -> None:
)
assert seed == PRE_CHANGE_SEED

def test_locked_v2_golden_vector(self) -> None:
"""Regenerated golden under the bumped identity (D6)."""
assert GENERATOR_VERSION == "2.0.0"
def test_locked_v3_golden_vector(self) -> None:
"""Regenerated golden under the bumped identity (GEN-009 remix)."""
assert GENERATOR_VERSION == "3.0.0"
seed = generation_seed(
GOLDEN_SELECTION_FINGERPRINT, GOLDEN_LOTTERY_ID, GOLDEN_COUNT, GENERATOR_VERSION
)
assert seed == GOLDEN_SEED_V2
assert seed == GOLDEN_SEED_V3

@pytest.mark.parametrize(
("selection_fingerprint", "lottery_id", "count", "version"),
Expand Down Expand Up @@ -157,16 +157,16 @@ def test_locked_pre_change_golden_vector(self) -> None:
)
assert fp == PRE_CHANGE_SNAPSHOT_FINGERPRINT

def test_locked_v2_golden_vector(self) -> None:
"""Regenerated fingerprint under the bumped identity (D6)."""
assert GENERATOR_VERSION == "2.0.0"
seed_v2 = generation_seed(
def test_locked_v3_golden_vector(self) -> None:
"""Regenerated fingerprint under the bumped identity (GEN-009 remix)."""
assert GENERATOR_VERSION == "3.0.0"
seed_v3 = generation_seed(
GOLDEN_SELECTION_FINGERPRINT, GOLDEN_LOTTERY_ID, GOLDEN_COUNT, GENERATOR_VERSION
)
fp = snapshot_fingerprint(
GOLDEN_LOTTERY_ID, GOLDEN_SELECTION_ID, GOLDEN_COUNT, seed_v2, GENERATOR_VERSION
GOLDEN_LOTTERY_ID, GOLDEN_SELECTION_ID, GOLDEN_COUNT, seed_v3, GENERATOR_VERSION
)
assert fp == GOLDEN_SNAPSHOT_FINGERPRINT_V2
assert fp == GOLDEN_SNAPSHOT_FINGERPRINT_V3

@pytest.mark.parametrize(
("lottery_id", "selection_id", "count", "seed", "version"),
Expand Down Expand Up @@ -228,9 +228,9 @@ def test_matches_canonical_sha256_formula(self) -> None:
class TestVersionBumpAliasingGuard:
"""D6/R2 — v2 outputs MUST NOT alias any pre-change fixture fingerprint."""

def test_v2_fingerprint_differs_from_pre_change(self) -> None:
def test_v3_fingerprint_differs_from_pre_change(self) -> None:
"""Same canonical inputs → bump moves the fingerprint away from legacy."""
assert GENERATOR_VERSION == "2.0.0"
assert GENERATOR_VERSION == "3.0.0"
seed_v2 = generation_seed(
GOLDEN_SELECTION_FINGERPRINT, GOLDEN_LOTTERY_ID, GOLDEN_COUNT, GENERATOR_VERSION
)
Expand Down
31 changes: 9 additions & 22 deletions backend/tests/gen/test_sampling.py
Original file line number Diff line number Diff line change
Expand Up @@ -42,10 +42,7 @@ def cfg(self) -> LotteryConfig:

def _make_pool(self, n: int = 49, weight: float = 1.0) -> WeightedPool:
"""Create a uniform distribution over numbers 1..n."""
return WeightedPool(
probabilities={i: weight for i in range(1, n + 1)},
score=1.0,
)
return WeightedPool(weights={i: weight for i in range(1, n + 1)})

def test_determinism(self, cfg: LotteryConfig) -> None:
"""Same seed → identical output (GEN-005, NFR-GEN-01)."""
Expand Down Expand Up @@ -91,13 +88,13 @@ def test_sb_drawn_on_same_stream_after_acceptance(self, cfg: LotteryConfig) -> N
consume SB draws (post-acceptance draw). White-box replay of the documented
algorithm with a single random.Random(seed) instance.
"""
pool = WeightedPool(probabilities={i: 1.0 for i in range(1, 50)}, score=1.0)
pool = WeightedPool(weights={i: 1.0 for i in range(1, 50)})
count = 4
result = sample_combinations(1234, [pool], count, cfg)

rng = random.Random(1234)
numbers = sorted(pool.probabilities)
weights = [pool.probabilities[n] * pool.score for n in numbers]
numbers = sorted(pool.weights)
weights = [pool.weights[n] for n in numbers]
sb_numbers = list(range(cfg.super_number_min, cfg.super_number_max + 1))
sb_weights = [1.0 / len(sb_numbers)] * len(sb_numbers)
generated: set[frozenset[int]] = set()
Expand Down Expand Up @@ -140,10 +137,7 @@ def test_max_attempts_exhaustion(self) -> None:
super_number_min=1,
super_number_max=9,
)
pool = WeightedPool(
probabilities={i: 1.0 for i in range(1, 7)},
score=1.0,
)
pool = WeightedPool(weights={i: 1.0 for i in range(1, 7)})
with pytest.raises(GenServiceError) as exc_info:
sample_combinations(42, [pool], 5, tiny_cfg, max_attempts=3)
assert exc_info.value.code == "GEN_SPACE_EXHAUSTED"
Expand All @@ -157,24 +151,17 @@ def test_no_duplicates_in_output(self, cfg: LotteryConfig) -> None:

def test_multiple_pools(self, cfg: LotteryConfig) -> None:
"""Multiple weighted pools produce combinations from each."""
pool1 = WeightedPool(
probabilities={i: 1.0 for i in range(1, 50)},
score=0.7,
)
pool2 = WeightedPool(
probabilities={i: 1.0 for i in range(1, 50)},
score=0.3,
)
# 3 from pool1, 2 from pool2 = 5 total
pool1 = WeightedPool(weights={i: 0.7 for i in range(1, 50)})
pool2 = WeightedPool(weights={i: 0.3 for i in range(1, 50)})
# The first pool satisfies the full count; pool2 is unused.
results = sample_combinations(42, [pool1, pool2], 5, cfg)
assert len(results) == 5

def test_score_influences_distribution(self) -> None:
"""Higher score weight biases number selection."""
# Pool with strong weight on low numbers
pool_low = WeightedPool(
probabilities={i: (10.0 if i <= 10 else 0.1) for i in range(1, 50)},
score=1.0,
weights={i: (10.0 if i <= 10 else 0.1) for i in range(1, 50)}
)
cfg = LotteryConfig(
numbers_to_select=3,
Expand Down
8 changes: 4 additions & 4 deletions backend/tests/gen/test_types.py
Original file line number Diff line number Diff line change
Expand Up @@ -94,10 +94,10 @@ def test_inequality_different_count(self) -> None:
class TestGeneratorVersion:
"""GENERATOR_VERSION constant — GEN-009 determinism + D6 major bump."""

def test_version_is_2_0_0(self) -> None:
# D6: stream-consumption change (SB on the shared stream) is a breaking
# output-identity change → GENERATOR_VERSION "2.0.0".
assert GENERATOR_VERSION == "2.0.0"
def test_version_is_3_0_0(self) -> None:
# GEN-009 remix: dropping meta entry.score is a breaking output-identity
# change → GENERATOR_VERSION "3.0.0".
assert GENERATOR_VERSION == "3.0.0"

def test_version_is_string(self) -> None:
assert isinstance(GENERATOR_VERSION, str)
Expand Down
Loading
Loading