Skip to content
Merged
24 changes: 12 additions & 12 deletions backend/src/backend/app/generators/sampling.py
Original file line number Diff line number Diff line change
Expand Up @@ -40,10 +40,9 @@ def super_number_max(self) -> int: ...

@dataclass(frozen=True)
class WeightedPool:
"""A number→probability distribution weighted by an entry score."""
"""A number→weight distribution used for weighted sampling (GEN-009)."""

probabilities: dict[int, float]
score: float
weights: dict[int, float]


def sample_combinations(
Expand All @@ -56,12 +55,13 @@ def sample_combinations(
) -> list[tuple[list[int], int]]:
"""Generate ``count`` unique valid ``(combination, super_balota)`` pairs.

For each pool, weighted sampling uses ``rng.choices`` with weights derived
from the pool's probability map × entry score. Invalid or duplicate combos
trigger resampling. On each ACCEPTED combination the Superbalota is drawn
ONCE from the same ``isolated_rng(seed)`` stream over ``sb_marginal``
(D1: post-acceptance draw keeps stream consumption independent of rejection
counts), and the full pair is legality-gated pre-append (D5/R1).
For each pool, weighted sampling uses ``rng.choices`` with the pool's
precomputed ``weights`` map (F5 × cold boost, GEN-009). Invalid or
duplicate combos trigger resampling. On each ACCEPTED combination the
Superbalota is drawn ONCE from the same ``isolated_rng(seed)`` stream over
``sb_marginal`` (D1: post-acceptance draw keeps stream consumption
independent of rejection counts), and the full pair is legality-gated
pre-append (D5/R1).

``sb_marginal`` maps candidate SB values to relative weights; when ``None``
a uniform distribution over the configured SB range is used. On
Expand All @@ -83,9 +83,9 @@ def sample_combinations(
results: list[tuple[list[int], int]] = []

for pool in pools:
# Build weighted pool: number → (probability × score)
numbers = sorted(pool.probabilities.keys())
weights = [pool.probabilities[n] * pool.score for n in numbers]
# Build weighted pool directly from the precomputed weights map.
numbers = sorted(pool.weights.keys())
weights = [float(pool.weights[n]) for n in numbers]

needed = count - len(results)
for _ in range(needed):
Expand Down
10 changes: 6 additions & 4 deletions backend/src/backend/app/generators/version.py
Original file line number Diff line number Diff line change
@@ -1,10 +1,12 @@
"""Generator version constant — bumped on algorithm changes only (GEN-009).

2.0.0 (D6): SuperBalota sampling joined the numbers' isolated RNG stream
(R2/D1), changing stream consumption — output identity (``generation_seed`` /
``snapshot_fingerprint``) differs from every pre-2.0.0 value.
3.0.0 (GEN-009 remix): dropped the meta prediction-chain ``entry.score`` from
sampling/allocation. Per-number weights are now F5 frequency × cold-coverage
boost (PM-08), computed transparently from draw history. Output identity
(``generation_seed`` / ``snapshot_fingerprint``) differs from every pre-3.0.0
value.
"""

from __future__ import annotations

GENERATOR_VERSION: str = "2.0.0"
GENERATOR_VERSION: str = "3.0.0"
34 changes: 34 additions & 0 deletions backend/src/backend/app/generators/weighting.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,34 @@
"""Generation weight construction (GEN-09): legitimate statistical levers only.

Builds the per-number sampling weights consumed by ``sampling.WeightedPool``.
The remix replaces the retired ``entry.score`` meta chain: a number's weight is
now a transparent function of draw history (F5 frequency × an optional cold
coverage boost), never a meta model score.
"""

from __future__ import annotations

from decimal import Decimal

COLD_BOOST: Decimal = Decimal("1.5")
"""Multiplier applied to COLD numbers to lift their sampling weight (PM-08)."""


def build_weights(
probabilities: dict[int, float],
coverage: dict[int, str],
*,
cold_boost: Decimal = COLD_BOOST,
) -> dict[int, float]:
"""Combine F5 probabilities with a cold-coverage boost into final weights.

``probabilities`` is the F5 number→probability map (probability engine).
``coverage`` maps each number to ``"cold"``/``"normal"``/``"hot"``; only
``"cold"`` numbers receive the ``cold_boost`` multiplier (lever C). The
result preserves the key set of ``probabilities``.
"""
boost = float(cold_boost)
return {
number: probability * (boost if coverage.get(number) == "cold" else 1.0)
for number, probability in probabilities.items()
}
108 changes: 108 additions & 0 deletions backend/src/backend/app/services/ev_service.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,108 @@
"""Expected-value analysis for generated combinations (EV-15).

Lever A from the number-generation-remix design: surface the *payout-if-win*
expectation of a ticket so the UI can be honest about odds. The app models a
single jackpot tier (Draw.jackpot) with parimutuel split by winners, so the only
data-driven lever is the current jackpot magnitude relative to ticket cost and
historical mean winners. Without per-combination sales data (lever B is
out-of-scope), every combination has the same expected value; `combination_ev`
still accepts an `expected_winners` so popularity-differentiated ranking can be
added later without an API change.
"""

from __future__ import annotations

import math
from decimal import ROUND_HALF_UP, Decimal

from sqlalchemy.orm import Session

from backend.app.repositories.lottery_repository import LotteryRepository
from backend.app.repositories.stat_payload_repository import StatPayloadRepository
from backend.app.services.errors import NotFoundError


class EVService:
"""Compute expected value of a ticket / combination for a lottery."""

def __init__(self, session: Session) -> None:
self._session = session
self._lotteries = LotteryRepository(session)
self._payloads = StatPayloadRepository(session)

def _resolve_lottery(self, *, lottery_code: str | None, lottery_id: int | None):
"""Resolve the lottery from ``code`` or ``id``; 404-style when absent."""
lottery = None
if lottery_code is not None:
lottery = self._lotteries.get_by_code(lottery_code)
elif lottery_id is not None:
lottery = self._lotteries.get(lottery_id)
if lottery is None:
raise NotFoundError("lottery does not exist")
return lottery

def combinations_count(
self, *, lottery_code: str | None = None, lottery_id: int | None = None
) -> int:
"""Number of possible combinations C(universe, numbers_to_select)."""
lottery = self._resolve_lottery(lottery_code=lottery_code, lottery_id=lottery_id)
universe = lottery.max_number - lottery.min_number + 1
return math.comb(universe, lottery.numbers_to_select)

def _latest_jackpot_and_avg_winners(self, lottery_id: int) -> tuple[Decimal, Decimal]:
jackpots: list[Decimal] = []
winners: list[Decimal] = []
for _draw_number, _numbers, jackpot, winner in self._payloads.iter_draws(lottery_id):
if jackpot is not None:
jackpots.append(Decimal(str(jackpot)))
if winner is not None:
winners.append(Decimal(str(winner)))
latest = jackpots[-1] if jackpots else Decimal(0)
avg = (sum(winners) / Decimal(len(winners))) if winners else Decimal(1)
return latest, avg

@staticmethod
def combination_ev(
combo,
jackpot: Decimal | float | str,
ticket_cost: Decimal | float | str,
combinations: int,
expected_winners: Decimal | float | str = 1,
) -> Decimal:
"""Expected value of playing one combination.

``EV = (jackpot / expected_winners) / combinations - ticket_cost``.
The ``combo`` argument is accepted for API symmetry (per-combination
winner estimates can differentiate EV later) but is not yet used.
"""
jackpot_d = Decimal(str(jackpot))
cost = Decimal(str(ticket_cost))
combos = Decimal(combinations)
winners = Decimal(str(expected_winners)) if expected_winners else Decimal(1)
expected_share = jackpot_d / winners
ev = expected_share / combos - cost
return ev.quantize(Decimal("0.0001"), rounding=ROUND_HALF_UP)

def estimate_ticket_ev(
self,
ticket_cost: Decimal | float | str,
*,
lottery_code: str | None = None,
lottery_id: int | None = None,
) -> Decimal:
"""EV of a single ticket using the latest jackpot and historical mean winners."""
lottery = self._resolve_lottery(lottery_code=lottery_code, lottery_id=lottery_id)
jackpot, avg_winners = self._latest_jackpot_and_avg_winners(lottery.id)
combos = self.combinations_count(lottery_id=lottery.id)
return self.combination_ev(None, jackpot, ticket_cost, combos, avg_winners)

def is_high_ev_window(
self,
ticket_cost: Decimal | float | str,
*,
lottery_code: str | None = None,
lottery_id: int | None = None,
) -> bool:
"""True when a ticket's EV is positive (rare; only on large rollovers)."""
ev = self.estimate_ticket_ev(ticket_cost, lottery_code=lottery_code, lottery_id=lottery_id)
return ev > 0
81 changes: 35 additions & 46 deletions backend/src/backend/app/services/gen_service.py
Original file line number Diff line number Diff line change
Expand Up @@ -31,8 +31,12 @@
from backend.app.generators.snapshot_store import GenSnapshotStore
from backend.app.generators.validation import validate_combination
from backend.app.generators.version import GENERATOR_VERSION
from backend.app.generators.weighting import build_weights
from backend.app.models.gen_snapshot import GenSnapshot
from backend.app.services.probability_service import _classify_coverage
from backend.app.repositories.stat_payload_repository import StatPayloadRepository
from backend.app.services.errors import GenServiceError
from backend.app.statistics.engine import frequency

DEFAULT_COUNT: int = 10
"""Default combination count when not provided (GEN-002)."""
Expand Down Expand Up @@ -120,22 +124,23 @@ def generate(
) -> GenerationResult:
"""Generate (or idempotently return) a lottery combination snapshot.

Pipeline (GEN-001): resolve the F12 selection → validate count →
allocate via the micro-unit rule (GEN-004) → load the F5 distribution →
load the SB historical marginal (R2/D2) → sample ``(combo, sb)`` pairs
with ``isolated_rng`` (GEN-005, D1) → compute the selection-weighted
score (R3/D3) → gate legality pre-persist (R1/D5) → fingerprint →
persist a NEW active version atomically. Same inputs reproduce the
identical snapshot including Superbalotas and scores (GEN-008); a
duplicate non-active fingerprint is a conflict (GEN-013).
Pipeline (GEN-001, GEN-009 remix): resolve the selection scope → validate
count → load the F5 distribution → derive per-number weights from F5 ×
cold-coverage boost (PM-08) → sample ``(combo, sb)`` pairs with
``isolated_rng`` (GEN-005, D1) → score each combo with its transparent
mean sampling weight (R3/D3) → gate legality pre-persist (R1/D5) →
fingerprint → persist a NEW active version atomically. The meta
prediction-chain scores are NOT used (audit-proven zero effect). Same
inputs reproduce the identical snapshot including Superbalotas and scores
(GEN-008); a duplicate non-active fingerprint is a conflict (GEN-013).
"""
lottery = self._resolve_lottery(lottery_id)
effective_count = DEFAULT_COUNT if count is None else count
self._validate_count(effective_count)
selection = self._resolve_selection(lottery_id, selection_id)
entry_rows = self._read_selection_entries(selection.id)
entries = [SelectionEntry(score=row.score, rank=row.rank) for row in entry_rows]
allocations = allocate_count(entries, effective_count)
# GEN-09 remix: a single allocation unit carries the whole count; the
# per-number weights (not a meta entry score) drive sampling below.
allocations = allocate_count([SelectionEntry(score=1.0, rank=0)], effective_count)

effective_seed = (
generation_seed(selection.fingerprint, lottery_id, effective_count, GENERATOR_VERSION)
Expand All @@ -161,28 +166,29 @@ def generate(
)

probabilities = self._load_distribution(lottery_id)
sb_marginal = self._load_sb_marginal(lottery)
pools = [
WeightedPool(probabilities=probabilities, score=entries[i].score)
for i, allocated in allocations
if allocated > 0
# Coverage (COLD/NORMAL/HOT) from draw history; only COLD numbers get a
# boost (PM-08). With no imported draw numbers this map is all "normal".
draws = [
numbers
for _dn, numbers, _j, _w in StatPayloadRepository(self._session).iter_draws(lottery.id)
]
coverage = _classify_coverage(
frequency(draws),
lottery.min_number,
lottery.max_number,
lottery.numbers_to_select,
)
weights = build_weights(probabilities, coverage)
sb_marginal = self._load_sb_marginal(lottery)
pools = [WeightedPool(weights=weights) for _i, allocated in allocations if allocated > 0]
sampled = sample_combinations(effective_seed, pools, effective_count, lottery, sb_marginal)

# D3: score = entry_score × mean(P(n)) computed where pools carry both
# inputs. Pools consume the sample sequentially in allocation order, so
# each pair's provenance (entry score) is recovered positionally.
# D3/R3: score is the transparent mean sampling weight of the combo's
# numbers (F5 × cold boost) — no meta entry score involved.
scored: list[tuple[list[int], int, float]] = []
idx = 0
for entry_index, allocated in allocations:
if allocated <= 0:
continue
entry_score = float(entries[entry_index].score)
for _ in range(allocated):
combo, sb = sampled[idx]
idx += 1
mean_p = sum(probabilities.get(n, 0.0) for n in combo) / len(combo)
scored.append((combo, sb, round(entry_score * mean_p, 6)))
for combo, sb in sampled:
mean_w = sum(weights.get(n, 0.0) for n in combo) / len(combo)
scored.append((combo, sb, round(mean_w, 6)))

# R1/D5: legality assert before anything is persisted.
for combo, sb, _score in scored:
Expand Down Expand Up @@ -341,23 +347,6 @@ def _resolve_selection(self, lottery_id: int, selection_id: int | None) -> Any:
)
return selection

def _read_selection_entries(self, selection_id: int) -> list[Any]:
"""Read the scored entries of a selection; ``GEN_NO_SELECTION`` when empty."""
from backend.app.models.meta_selection_entry import MetaSelectionEntry

stmt = (
select(MetaSelectionEntry)
.where(MetaSelectionEntry.selection_id == selection_id)
.order_by(MetaSelectionEntry.rank)
)
rows = list(self._session.execute(stmt).scalars().all())
if not rows:
raise GenServiceError(
GenServiceError.GEN_NO_SELECTION,
f"selection {selection_id} has no entries",
)
return rows

def _load_distribution(self, lottery_id: int) -> dict[int, float]:
"""Read the active F5 number→probability map; ``GEN_NO_DISTRIBUTION`` absent.

Expand Down
8 changes: 8 additions & 0 deletions backend/src/backend/app/services/meta_service.py
Original file line number Diff line number Diff line change
Expand Up @@ -10,6 +10,7 @@

import json
from dataclasses import dataclass, field
from datetime import UTC, datetime
from typing import Any

from sqlalchemy.orm import Session
Expand Down Expand Up @@ -136,6 +137,13 @@ def rank(
# 5. Idempotency check
existing = self._store.find_by_fingerprint(fp)
if existing is not None:
# D8 healing: refresh created_at so a rerank against a newer
# bt_snapshot is not blocked by a stale timestamp. The ranking
# content (fingerprint) is identical, so it stays valid for the
# current backtest context.
existing.created_at = datetime.now(UTC)
self._session.add(existing)
self._session.commit()
return RankingResult(
ranking_id=existing.id,
lottery_id=lottery_id,
Expand Down
Loading
Loading