diff --git a/.env.example b/.env.example index dd70afa..aaaf1db 100644 --- a/.env.example +++ b/.env.example @@ -92,7 +92,14 @@ TARGETED_NEWS_REFRESH_HOURS=6 # Forecasting signals (Jev vs market price) PREDICTION_AUTO=false PREDICTION_MAX_PER_RUN=10 -MODEL_WEIGHT_MAX=0.5 +# Jev's maximum weight against the price, and how far from the price (log-odds) before it shrinks +MODEL_WEIGHT_MAX=0.25 +MODEL_DISAGREEMENT_LOGIT=2.0 +# A forecast is valid for buying for this many hours, and while the price moves less than this (log-odds) +FORECAST_MAX_AGE_HOURS=6 +FORECAST_MAX_PRICE_MOVE=0.5 +# Markets decided by an asset's price (e.g. "Bitcoin above $84,000 on September 24"): no bets, no paid forecasts +EXCLUDE_PRICE_MARKETS=true MIN_EDGE=0.05 MIN_EVIDENCE=0.5 KELLY_FRACTION=0.25 diff --git a/backend/alerts/service.py b/backend/alerts/service.py index ce14de2..8f4c148 100644 --- a/backend/alerts/service.py +++ b/backend/alerts/service.py @@ -27,6 +27,7 @@ from backend.betting.clv import summarize as clv_summary from backend.markets.matching import source_quality from backend.i18n import side as side_label, tr +from backend.markets.kinds import skip_paid_forecast logger = logging.getLogger(__name__) @@ -227,6 +228,8 @@ async def _run_alerts(db: AsyncSession) -> dict: market = await db.get(Market, trig["market_id"], populate_existing=True) if market is None or market.closed: continue + if skip_paid_forecast(market.question): # decided by an asset's price: no bet can come of it + continue await _refresh_price(market) price = market.yes_price try: diff --git a/backend/api/routes/status.py b/backend/api/routes/status.py index b56eb81..63f7f3b 100644 --- a/backend/api/routes/status.py +++ b/backend/api/routes/status.py @@ -26,6 +26,10 @@ async def count(stmt): "min_edge": settings.MIN_EDGE, "min_evidence": settings.MIN_EVIDENCE, "model_weight_max": settings.MODEL_WEIGHT_MAX, + "model_disagreement_logit": settings.MODEL_DISAGREEMENT_LOGIT, + "forecast_max_age_hours": settings.FORECAST_MAX_AGE_HOURS, + "forecast_max_price_move": settings.FORECAST_MAX_PRICE_MOVE, + "exclude_price_markets": settings.EXCLUDE_PRICE_MARKETS, "blend_method": settings.BLEND_METHOD, "jev_calib_a": settings.JEV_CALIB_A, "jev_calib_b": settings.JEV_CALIB_B, diff --git a/backend/betting/economics.py b/backend/betting/economics.py index aa78edf..abcc33a 100644 --- a/backend/betting/economics.py +++ b/backend/betting/economics.py @@ -237,7 +237,10 @@ def evaluate( exposure: Exposure, liquidity: float, risk_free_rate: float, + hours_to_end: Optional[float] = None, + extra_reasons: Optional[list] = None, ) -> Evaluation: + """extra_reasons: blocking reasons found by the caller (stale forecast, price market).""" side = "NO" if signal == "BUY_NO" else "YES" p_side = p_yes if side == "YES" else 1.0 - p_yes p_cons = max(0.0, p_side - profile.z * sigma) @@ -254,6 +257,13 @@ def evaluate( notes.append(tr("Book non disponibile: book stimato a 4 livelli, da prezzo medio + metà spread in su, con la profondità dalla liquidità dichiarata.", "Book not available: book estimated at 4 levels, from mid price + half spread upwards, with depth from the declared liquidity.")) + reasons.extend(extra_reasons or []) + if hours_to_end is not None and hours_to_end < profile.min_hours_to_end: + left = tr(f"{hours_to_end:.0f} ore", f"{hours_to_end:.0f} hours") if hours_to_end >= 1 else \ + tr(f"{hours_to_end * 60:.0f} minuti", f"{hours_to_end * 60:.0f} minutes") + reasons.append(Reason("too_close", tr( + f"Si risolve tra {left}: il prezzo sa già quasi tutto, il preset chiede almeno {profile.min_hours_to_end:.0f} ore.", + f"It resolves in {left}: the price already knows almost everything, the preset asks for at least {profile.min_hours_to_end:.0f} hours."))) if signal == "HOLD": reasons.append(Reason("no_signal", tr("Il modello non vede una differenza sufficiente rispetto al prezzo.", "The model does not see a large enough difference from the price."))) @@ -317,6 +327,16 @@ def evaluate( f"Prudent annualised return {_num(apr * 100)}%, below the {_num(hurdle * 100)}% threshold " f"(risk-free rate {_num(risk_free_rate * 100)}% + preset premium)."))) + # Absolute return: annualizing a short bet makes any small margin look huge, so the prudent + # return on the money spent must also clear a minimum per bet + roi_cons = (exp_profit_cons / outlay) if outlay > 0 else \ + ((p_cons / (best + fee_per_share(best, quote.fee_bps)) - 1) if best is not None else None) + if roi_cons is not None and roi_cons < profile.min_roi and signal != "HOLD" \ + and not any(r.code in ("edge_after_costs", "return_too_low") for r in reasons): + reasons.append(Reason("roi_too_low", tr( + f"Rendimento prudente della scommessa {_num(roi_cons * 100, 1)}%: il preset ne chiede almeno {_num(profile.min_roi * 100)}%.", + f"Prudent return of the bet {_num(roi_cons * 100, 1)}%: the preset asks for at least {_num(profile.min_roi * 100)}%."))) + blocking = [r for r in reasons if r.blocking] if blocking: verdict = "NO" diff --git a/backend/betting/plans.py b/backend/betting/plans.py index ed3170c..49e200d 100644 --- a/backend/betting/plans.py +++ b/backend/betting/plans.py @@ -25,17 +25,37 @@ def half_spread(market: Market) -> float: return settings.DEFAULT_SPREAD / 2 +def is_outcome(prediction) -> bool: + """A view of one outcome of a multi-outcome forecast (its blend comes from the whole distribution).""" + return getattr(prediction, "multi_prediction_id", None) is not None + + def forecast_of(prediction, sigma: float) -> Forecast: - """The forecast as it was computed: calibrated Jev, its weight, the pooling method. - Forecasts made before calibration and log-odds pooling were linear and uncalibrated.""" + """The forecast with today's parameters, ready to be pooled with any price: Jev calibrated + with the current calibration, the base weight (MODEL_WEIGHT_MAX × evidence; the reduction + when Jev is far from the price is applied by the pooling, at each price). + Forecasts made before log-odds pooling were linear and uncalibrated; outcomes of + multi-outcome events keep the probability given by the distribution.""" method = getattr(prediction, "blend_method", None) or "linear" - cal = getattr(prediction, "calibrated_probability", None) - if cal is None: - cal = prediction.model_probability if method == "linear" else calibrate(prediction.model_probability) - w = prediction.model_weight if prediction.model_weight is not None else model_weight(prediction.evidence_strength) + if method == "linear" or is_outcome(prediction): + cal = getattr(prediction, "calibrated_probability", None) or prediction.model_probability + else: + cal = calibrate(prediction.model_probability) + w = model_weight(prediction.evidence_strength) return Forecast(model=cal, weight=w, evidence=prediction.evidence_strength, sigma=sigma, method=method) +def signal_at(prediction, price: Optional[float]) -> str: + """The forecast's signal at `price`: the blend is recomputed there, as when the forecast was made. + Outcomes of multi-outcome events and markets without a price keep the stored signal.""" + if price is None or is_outcome(prediction): + return prediction.signal + edge = forecast_of(prediction, 0).p_yes(price) - price + if prediction.evidence_strength < settings.MIN_EVIDENCE or abs(edge) < settings.MIN_EDGE: + return "HOLD" + return "BUY_YES" if edge > 0 else "BUY_NO" + + async def best_bid(market: Market, side: str) -> Optional[float]: """What one share of `side` would fetch now: the book's best bid, else Gamma's top of book.""" token = market.yes_token_id if side == "YES" else market.no_token_id @@ -104,7 +124,7 @@ async def plan_for(db: AsyncSession, market: Market, prediction, ev: Evaluation, fee_bps = market.taker_fee_bps if market.taker_fee_bps is not None else fees.category_rate(market.category) * 10_000 sell_bid = await best_bid(market, position.side) if position else None return build_plan( - ev=ev.as_dict(), fc=forecast_of(prediction, ev.sigma), signal=prediction.signal, + ev=ev.as_dict(), fc=forecast_of(prediction, ev.sigma), signal=signal_at(prediction, market.yes_price), market_price=market.yes_price, profile=profile, fee_bps=fee_bps, days=ev.days, min_edge=settings.MIN_EDGE, risk_free=settings.RISK_FREE_RATE, position={"side": position.side, "shares": position.shares, "avg_price": position.avg_price} if position else None, @@ -124,6 +144,7 @@ def exit_plan(prediction, market: Market, side: str, profile: RiskProfile, days: bid = market.best_bid if market.best_bid is not None else market.yes_price else: bid = 1 - market.best_ask if market.best_ask is not None else (1 - market.yes_price if market.yes_price is not None else None) - flipped = (prediction.signal == "BUY_NO" and side == "YES") or (prediction.signal == "BUY_YES" and side == "NO") + signal = signal_at(prediction, market.yes_price) + flipped = (signal == "BUY_NO" and side == "YES") or (signal == "BUY_YES" and side == "NO") action = "SELL" if flipped or (bid is not None and target is not None and bid >= target) else "HOLD" return {"action": action, "sell_above": target, "bid": bid, "flipped": flipped} diff --git a/backend/betting/portfolio.py b/backend/betting/portfolio.py index 0059182..b69ff4a 100644 --- a/backend/betting/portfolio.py +++ b/backend/betting/portfolio.py @@ -115,14 +115,24 @@ async def calibration_factor(db: AsyncSession) -> float: .subquery() ) rows = (await db.execute( - select(Market.resolved_yes, latest.c.blended_probability) + select(Market.resolved_yes, latest.c.blended_probability, latest.c.market_probability) .join(latest, latest.c.market_id == Market.id).where(Market.resolved_yes.is_not(None)) )).all() if len(rows) < MIN_RESOLVED_FOR_CALIBRATION: return 1.0 - observed = sum((r.blended_probability - (1.0 if r.resolved_yes else 0.0)) ** 2 for r in rows) / len(rows) + outcome = [1.0 if r.resolved_yes else 0.0 for r in rows] + observed = sum((r.blended_probability - y) ** 2 for r, y in zip(rows, outcome)) / len(rows) expected = sum(r.blended_probability * (1 - r.blended_probability) for r in rows) / len(rows) - return max(0.75, min(2.0, math.sqrt(observed / max(expected, 1e-4)))) + factor = max(0.75, min(2.0, math.sqrt(observed / max(expected, 1e-4)))) + # Narrowing the uncertainty makes the bets bigger: only when the blend has shown it beats + # the market price on the same markets (paired Brier gain, two standard errors above zero). + # Being consistent with itself is not enough. + gains = [(r.market_probability - y) ** 2 - (r.blended_probability - y) ** 2 for r, y in zip(rows, outcome)] + mean = sum(gains) / len(gains) + se = math.sqrt(sum((g - mean) ** 2 for g in gains) / (len(gains) - 1) / len(gains)) if len(gains) > 1 else float("inf") + if mean - 2 * se <= 0: + factor = max(1.0, factor) + return factor async def update_market_category(db: AsyncSession, market: Market) -> Optional[str]: @@ -164,18 +174,33 @@ def days_to_end(market: Market) -> float: async def evaluate_prediction(db: AsyncSession, market: Market, prediction: MarketPrediction, quote: Optional[Quote] = None, preset: Optional[str] = None) -> Evaluation: + """Economic evaluation of a forecast at the market's current price. + + The blend is recomputed at the current price (Jev pooled with the price of now, with today's + calibration and weights): the stored blend was pooled with the price of the forecast, and + using it against a price that has moved since makes up an edge that is not there. Past + FORECAST_MAX_AGE_HOURS or FORECAST_MAX_PRICE_MOVE the forecast needs a new one before buying.""" + from backend.betting import plans + from backend.markets.forecast import disagreement_factor s = await get_settings(db) profile = get_profile(preset or s.preset) - side = "NO" if prediction.signal == "BUY_NO" else "YES" - weight = prediction.model_weight if prediction.model_weight is not None else \ - min(1.0, settings.MODEL_WEIGHT_MAX * prediction.evidence_strength) + price = market.yes_price + fc = plans.forecast_of(prediction, 0) + if price is None or plans.is_outcome(prediction): + # Outcomes keep the probability of their distribution (pooled over all the outcomes) + p_yes, signal = prediction.blended_probability, prediction.signal + weight = prediction.model_weight if prediction.model_weight is not None else fc.weight + else: + p_yes, signal = fc.p_yes(price), plans.signal_at(prediction, price) + weight = fc.weight * disagreement_factor(fc.model, price) + side = "NO" if signal == "BUY_NO" else "YES" sigma = model_sigma(prediction.model_probability, prediction.evidence_strength, weight, settings.MODEL_PSEUDO_COUNT, await calibration_factor(db)) await update_market_category(db, market) ledger_now = await ledger(db) return evaluate( - signal=prediction.signal, - p_yes=prediction.blended_probability, + signal=signal, + p_yes=p_yes, sigma=sigma, quote=quote or await build_quote(market, side), days=days_to_end(market), @@ -185,9 +210,43 @@ async def evaluate_prediction(db: AsyncSession, market: Market, prediction: Mark exposure=await exposure_for(db, market), liquidity=market.liquidity or 0.0, risk_free_rate=settings.RISK_FREE_RATE, + hours_to_end=hours_to_end(market), + extra_reasons=forecast_reasons(market, prediction), ) +def hours_to_end(market: Market) -> Optional[float]: + """Hours left before the market resolves (None without an end date).""" + if market.end_date is None: + return None + return (market.end_date - _now()).total_seconds() / 3600 + + +def forecast_reasons(market: Market, prediction) -> list: + """Blocking reasons about the forecast itself: too old, the price moved too much since, + or a market decided by an asset's price (see markets/kinds.py).""" + from backend.betting.economics import Reason + from backend.markets.kinds import is_price_market + reasons = [] + if settings.EXCLUDE_PRICE_MARKETS and is_price_market(market.question): + reasons.append(Reason("price_market", tr( + "Mercato sul prezzo di un asset: si decide sul prezzo del momento, che Jev non vede e il mercato sì.", + "Market on an asset's price: it is decided by the price of the moment, which Jev does not see and the market does."))) + created = getattr(prediction, "created_at", None) + age = (_now() - created).total_seconds() / 3600 if created else 0.0 + from backend.markets.forecast import logit + has_prices = market.yes_price is not None and prediction.market_probability is not None + move = abs(market.yes_price - prediction.market_probability) if has_prices else 0.0 + moved = has_prices and abs(logit(market.yes_price) - logit(prediction.market_probability)) > settings.FORECAST_MAX_PRICE_MOVE + if age > settings.FORECAST_MAX_AGE_HOURS or moved: + why = tr(f"ha {age:.0f} ore", f"is {age:.0f} hours old") if age > settings.FORECAST_MAX_AGE_HOURS else \ + tr(f"il prezzo del SÌ si è mosso di {move * 100:.0f} punti da allora", f"the YES price has moved {move * 100:.0f} points since") + reasons.append(Reason("stale_forecast", tr( + f"La previsione {why}: prima di comprare serve una previsione nuova.", + f"The forecast {why}: a new forecast is needed before buying."))) + return reasons + + PORTFOLIO_REASONS = ("exposure_cap", "no_cash") # reasons that depend on the portfolio, not on the market diff --git a/backend/betting/profiles.py b/backend/betting/profiles.py index e02afd4..5e15905 100644 --- a/backend/betting/profiles.py +++ b/backend/betting/profiles.py @@ -20,6 +20,8 @@ class RiskProfile: max_book_share: float # max share of the visible order-book depth we would take min_liquidity: float # skip markets with less liquidity (USD) max_days: int # skip markets resolving further away than this + min_hours_to_end: float = 24.0 # skip markets resolving sooner: the price already knows the outcome + min_roi: float = 0.06 # minimum prudent expected return on the outlay, per bet (not annualized) def as_dict(self) -> dict: out = asdict(self) @@ -42,21 +44,21 @@ def as_dict(self) -> dict: description="Poche scommesse, piccole, solo con margine ampio e mercati liquidi che si chiudono entro 4 mesi.", kelly_scale=0.15, z=1.64, min_net_edge=0.04, min_apr_premium=0.15, max_market_frac=0.02, max_event_frac=0.05, max_category_frac=0.15, max_total_frac=0.40, - max_book_share=0.10, min_liquidity=25_000, max_days=120, + max_book_share=0.10, min_liquidity=25_000, max_days=120, min_hours_to_end=72, min_roi=0.1, ), "bilanciato": RiskProfile( key="bilanciato", label="Bilanciato", description="Il compromesso di default: un quarto di Kelly, margine di 3 punti dopo i costi, massimo 4% del capitale per mercato.", kelly_scale=0.25, z=1.0, min_net_edge=0.03, min_apr_premium=0.08, max_market_frac=0.04, max_event_frac=0.08, max_category_frac=0.25, max_total_frac=0.60, - max_book_share=0.20, min_liquidity=10_000, max_days=365, + max_book_share=0.20, min_liquidity=10_000, max_days=365, min_hours_to_end=24, min_roi=0.06, ), "aggressivo": RiskProfile( key="aggressivo", label="Aggressivo", description="Più scommesse e più grandi: mezzo Kelly, margini ridotti, anche mercati lontani. Oscillazioni del capitale ampie.", kelly_scale=0.5, z=0.5, min_net_edge=0.02, min_apr_premium=0.03, max_market_frac=0.08, max_event_frac=0.15, max_category_frac=0.40, max_total_frac=0.85, - max_book_share=0.30, min_liquidity=5_000, max_days=730, + max_book_share=0.30, min_liquidity=5_000, max_days=730, min_hours_to_end=12, min_roi=0.03, ), } diff --git a/backend/betting/strategy.py b/backend/betting/strategy.py index 1a85e14..11e7aa6 100644 --- a/backend/betting/strategy.py +++ b/backend/betting/strategy.py @@ -22,7 +22,7 @@ GRID = [round(0.01 + i * 0.005, 3) for i in range(197)] # YES prices from 1¢ to 99¢ # Reasons that no price can fix: the market itself does not suit the preset or the portfolio -STRUCTURAL = {"illiquid", "too_far", "exposure_cap", "no_cash", "no_book"} +STRUCTURAL = {"illiquid", "too_far", "too_close", "price_market", "exposure_cap", "no_cash", "no_book"} def _cents(p: Optional[float]) -> str: @@ -88,6 +88,8 @@ def _buy_ok(fc: Forecast, q: float, side: str, profile: RiskProfile, fee_bps: fl p_cons = max(0.0, p_side - profile.z * fc.sigma) if p_cons - cost < profile.min_net_edge: # margin after costs and uncertainty return False + if p_cons / cost - 1 < profile.min_roi: # prudent return of the bet itself + return False return annualize(p_cons / cost - 1, days) >= hurdle # worth the time the money is locked @@ -270,6 +272,19 @@ def build_plan(*, ev: dict, fc: Forecast, signal: str, market_price: float, prof # ----- No position ----- side = "NO" if signal == "BUY_NO" else "YES" if signal == "BUY_YES" else None + # Reasons that no price can fix (a market on an asset's price, too close to the end): no + # buy levels, they would suggest a trade the assessment will never allow + unfixable = [r for r in blocking if r["code"] in ("price_market", "too_close")] + if unfixable: + return Plan("AVOID" if side else "NONE", side, tr("Evita questo mercato", "Avoid this market"), + " ".join(r["text"] for r in unfixable), levels={}, pros=[], cons=[r["text"] for r in unfixable], + confidence=confidence, confidence_why=why) + stale = next((r for r in blocking if r["code"] == "stale_forecast"), None) + if stale: + return Plan("WAIT", side, tr("Serve una nuova previsione", "A new forecast is needed"), + stale["text"] + tr(" I conti di sotto sono al prezzo attuale, ma la stima di Jev è di allora.", + " The numbers below are at the current price, but Jev's estimate is from then."), + levels=levels, pros=[], cons=[stale["text"]] + track_cons, confidence=confidence, confidence_why=why) if side is None: # No signal: say at which prices there would be one side_hint = "YES" if p_now >= market_price else "NO" diff --git a/backend/config.py b/backend/config.py index fe8daf0..5c5da8f 100644 --- a/backend/config.py +++ b/backend/config.py @@ -82,7 +82,8 @@ class Settings(BaseSettings): # Forecasting / betting signals PREDICTION_AUTO: bool = False # run Jev predictions automatically after each ingest PREDICTION_MAX_PER_RUN: int = 10 - MODEL_WEIGHT_MAX: float = 0.5 # max weight of Jev vs market price in the blended probability + MODEL_WEIGHT_MAX: float = 0.25 # max weight of Jev vs market price in the blended probability + MODEL_DISAGREEMENT_LOGIT: float = 2.0 # Jev's weight shrinks when it is further than this (log-odds) from the price; 0 = off BLEND_METHOD: str = "logodds" # "logodds" (pool in log-odds) or "linear" JEV_CALIB_A: float = 0.0 # Platt scaling of Jev: logit p' = A + B · logit p (fitted by the backtest) JEV_CALIB_B: float = 1.0 @@ -90,6 +91,13 @@ class Settings(BaseSettings): MIN_EDGE: float = 0.05 # minimum |edge| to emit a BUY signal MIN_EVIDENCE: float = 0.5 # minimum normalized evidence strength to emit a signal KELLY_FRACTION: float = 0.25 # fractional Kelly sizing + # A forecast is re-evaluated at the current price; past these limits it needs a new one before buying + FORECAST_MAX_AGE_HOURS: float = 6.0 + FORECAST_MAX_PRICE_MOVE: float = 0.5 # YES price moved more than this since the forecast, in log-odds + # (≈ 12 points around 50%, 4 points around 90%: moves near the extremes weigh more) + # Markets decided by an asset price at a date ("Bitcoin above $84,000 on September 24"): Jev reads + # news, not the live price, and the market price already knows it. No bets, no paid forecasts. + EXCLUDE_PRICE_MARKETS: bool = True # Betting economics (simulated portfolio) RISK_FREE_RATE: float = 0.04 # annual return of the risk-free alternative (e.g. T-bills) diff --git a/backend/markets/bulk.py b/backend/markets/bulk.py index eebbb5b..88c92e7 100644 --- a/backend/markets/bulk.py +++ b/backend/markets/bulk.py @@ -70,7 +70,8 @@ async def eligible_ids(session: AsyncSession, only_new: bool = False) -> list[st if only_new: # Never predicted, or with news linked after the last forecast return [m.id for m in await service.markets_needing_prediction(session, limit=10_000)] - return [m.id for m in (await session.execute(eligible_query())).scalars().all()] + from backend.markets.kinds import skip_paid_forecast + return [m.id for m in (await session.execute(eligible_query())).scalars().all() if not skip_paid_forecast(m.question)] async def count_eligible(session: AsyncSession) -> dict: diff --git a/backend/markets/forecast.py b/backend/markets/forecast.py index ed164a1..07b2c46 100644 --- a/backend/markets/forecast.py +++ b/backend/markets/forecast.py @@ -54,9 +54,23 @@ def calibrate(p: float, a: Optional[float] = None, b: Optional[float] = None) -> return sigmoid(a + b * logit(p)) +def disagreement_factor(model_p: float, market_p: float, scale: Optional[float] = None) -> float: + """How much of Jev's weight to keep when it is far from the price: 1 up to `scale` log-odds + apart, then scale / distance. A liquid market that disagrees by that much is more often + right than Jev (the biggest disagreements are the likeliest Jev errors), so a huge gap + must not turn into a huge edge. MODEL_DISAGREEMENT_LOGIT = 0 turns it off.""" + scale = settings.MODEL_DISAGREEMENT_LOGIT if scale is None else scale + if scale <= 0: + return 1.0 + distance = abs(logit(model_p) - logit(market_p)) + return 1.0 if distance <= scale else scale / distance + + def pool(model_p: float, market_p: float, w: float, method: Optional[str] = None) -> float: - """Combines two probabilities: log-odds (default) or linear average, weight w on the model.""" + """Combines two probabilities: log-odds (default) or linear average, weight w on the model, + reduced when the model is far from the market (see disagreement_factor).""" method = method or settings.BLEND_METHOD + w = w * disagreement_factor(model_p, market_p) if w <= 0: return market_p if method == "linear": @@ -103,6 +117,7 @@ def compute_signal( blended = blend_probability(model_p, market_p, evidence_strength) edge = blended - market_p + weight = model_weight(evidence_strength) * disagreement_factor(calibrate(model_p), market_p) signal, stake = "HOLD", 0.0 if evidence_strength >= min_evidence and abs(edge) >= min_edge: @@ -114,7 +129,7 @@ def compute_signal( return Signal( blended_probability=round(blended, 4), - model_weight=round(model_weight(evidence_strength), 4), + model_weight=round(weight, 4), # effective weight, after the disagreement reduction edge=round(edge, 4), signal=signal, kelly_fraction=round(stake * kelly_scale, 4), diff --git a/backend/markets/kinds.py b/backend/markets/kinds.py new file mode 100644 index 0000000..5f7a20a --- /dev/null +++ b/backend/markets/kinds.py @@ -0,0 +1,44 @@ +"""Kinds of markets that need special handling. + +Price markets ("Will Bitcoin be above $84,000 on September 24?", "Will ETH reach $2,800 this +week?", "Will gold hit $3,000?") are decided by the price of an asset at a moment. Jev reads the +news, not the live price and its volatility, while the market price already reflects both: its +estimates there are uninformed and a large "edge" is almost always Jev being wrong. +""" +import re + +from backend.config import settings + +ASSETS = ( + # crypto + "bitcoin", "btc", "ethereum", "ether", "eth", "solana", "sol", "xrp", "ripple", "dogecoin", "doge", + "cardano", "ada", "bnb", "litecoin", "ltc", "avalanche", "avax", "chainlink", "link", "polkadot", "dot", + "shiba", "shib", "pepe", "tron", "trx", "toncoin", "ton", "sui", "hyperliquid", "hype", + # indices and stocks + "s&p", "s&p 500", "spx", "spy", "nasdaq", "qqq", "dow", "dow jones", "russell", "vix", "nikkei", "ftse", + "dax", "stock", "stocks", "shares", "tesla", "tsla", "nvidia", "nvda", "apple", "aapl", "microsoft", "msft", + "amazon", "amzn", "meta", "google", "googl", "alphabet", "netflix", "coinbase", "microstrategy", "mstr", + # commodities, currencies, yields + "gold", "silver", "oil", "crude", "wti", "brent", "natural gas", "copper", "eur/usd", "usd/jpy", "dollar index", + "dxy", "treasury yield", "10-year yield", +) +_ASSET = re.compile(r"(? bool: + """True for a market decided by an asset's price crossing a level (see module docstring).""" + q = question or "" + if not _ASSET.search(q): + return False + return bool(re.search(r"\bup or down\b", q, re.I) or (_LEVEL.search(q) and _THRESHOLD.search(q))) + + +def skip_paid_forecast(question: str) -> bool: + """Automatic forecasts, «Assess all» and alerts skip price markets (EXCLUDE_PRICE_MARKETS): no + bet can come out of them, so the Jev call would be wasted. A forecast asked by hand still runs.""" + return settings.EXCLUDE_PRICE_MARKETS and is_price_market(question) diff --git a/backend/markets/service.py b/backend/markets/service.py index abdf15e..3344d6c 100644 --- a/backend/markets/service.py +++ b/backend/markets/service.py @@ -378,9 +378,10 @@ async def markets_needing_prediction(session: AsyncSession, limit: int) -> list[ select(Market) .where(and_(Market.closed == False, Market.yes_price.is_not(None), Market.id.in_(new_links))) # noqa: E712 .order_by(Market.volume.desc()) - .limit(limit) ) - return list((await session.execute(stmt)).scalars().all()) + from backend.markets.kinds import skip_paid_forecast + markets = [m for m in (await session.execute(stmt)).scalars().all() if not skip_paid_forecast(m.question)] + return markets[:limit] async def run_market_pipeline(session: AsyncSession) -> dict: diff --git a/docs/configuration.md b/docs/configuration.md index c849d1d..5ae83e4 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -62,7 +62,11 @@ All variables are set in `.env`. Placeholder values `your_...` count as "not con | `TARGETED_NEWS_LOCALE` | `hl=en-US&gl=US&ceid=US:en` | Language and country of the results | | `PREDICTION_AUTO` | `false` | Automatic forecasts after every collection (each forecast is a paid call) | | `PREDICTION_MAX_PER_RUN` | `10` | Cap on automatic forecasts per run | -| `MODEL_WEIGHT_MAX` | `0.5` | Jev's maximum weight against the market price | +| `MODEL_WEIGHT_MAX` | `0.25` | Jev's maximum weight against the market price | +| `MODEL_DISAGREEMENT_LOGIT` | `2.0` | Jev's weight shrinks when it is further than this from the price, in log-odds (`0` = never) | +| `FORECAST_MAX_AGE_HOURS` | `6` | A forecast older than this needs a new one before buying | +| `FORECAST_MAX_PRICE_MOVE` | `0.5` | Same if the YES price has moved more than this since the forecast, in log-odds (≈ 12 points at 50 %, 4 at 90 %) | +| `EXCLUDE_PRICE_MARKETS` | `true` | Markets decided by an asset's price get no bets and no automatic, bulk or alert forecasts | | `BLEND_METHOD` | `logodds` | How Jev and price are combined: `logodds` or `linear` | | `JEV_CALIB_A` / `JEV_CALIB_B` | `0` / `1` | Platt calibration of Jev's estimate (estimated by the backtest) | | `JEV_SAMPLES` | `1` | Jev calls per forecast, averaged (each one is paid) | diff --git a/docs/it/configurazione.md b/docs/it/configurazione.md index e2fdc93..e0345a1 100644 --- a/docs/it/configurazione.md +++ b/docs/it/configurazione.md @@ -64,7 +64,11 @@ Tutte le variabili si impostano in `.env`. I valori segnaposto `your_...` contan | `TARGETED_NEWS_LOCALE` | `hl=en-US&gl=US&ceid=US:en` | Lingua e paese dei risultati | | `PREDICTION_AUTO` | `false` | Previsioni automatiche dopo ogni raccolta (ogni previsione è una chiamata a pagamento) | | `PREDICTION_MAX_PER_RUN` | `10` | Tetto di previsioni automatiche per esecuzione | -| `MODEL_WEIGHT_MAX` | `0.5` | Peso massimo di Jev rispetto al prezzo di mercato | +| `MODEL_WEIGHT_MAX` | `0.25` | Peso massimo di Jev rispetto al prezzo di mercato | +| `MODEL_DISAGREEMENT_LOGIT` | `2.0` | Il peso di Jev si riduce quando è più lontano di così dal prezzo, in log-odds (`0` = mai) | +| `FORECAST_MAX_AGE_HOURS` | `6` | Una previsione più vecchia richiede una previsione nuova prima di comprare | +| `FORECAST_MAX_PRICE_MOVE` | `0.5` | Lo stesso se il prezzo del SÌ si è mosso più di così dalla previsione, in log-odds (≈ 12 punti al 50 %, 4 al 90 %) | +| `EXCLUDE_PRICE_MARKETS` | `true` | I mercati decisi dal prezzo di un asset non ricevono scommesse né previsioni automatiche, di massa o per le allerte | | `BLEND_METHOD` | `logodds` | Come si uniscono Jev e prezzo: `logodds` o `linear` | | `JEV_CALIB_A` / `JEV_CALIB_B` | `0` / `1` | Calibrazione di Platt della stima di Jev (si stimano col backtest) | | `JEV_SAMPLES` | `1` | Chiamate a Jev per previsione, mediate (ognuna si paga) | diff --git a/docs/it/metodo.md b/docs/it/metodo.md index 099d0b7..8f188d6 100644 --- a/docs/it/metodo.md +++ b/docs/it/metodo.md @@ -90,11 +90,19 @@ così com'è: 2. **Unione col prezzo in log-odds**, con un peso che cresce con la forza delle evidenze: ``` -w = MODEL_WEIGHT_MAX × evidence_strength +d = |logit(P_cal) − logit(prezzo)| +w = MODEL_WEIGHT_MAX × evidence_strength × min(1, MODEL_DISAGREEMENT_LOGIT / d) logit(blended) = w × logit(P_cal) + (1 − w) × logit(prezzo) edge = blended − prezzo ``` +`MODEL_WEIGHT_MAX` vale 0,25 e `MODEL_DISAGREEMENT_LOGIT` 2. Il peso si riduce quando Jev è molto +lontano dal prezzo: un mercato liquido che dissente da Jev di più di 2 in log-odds (per esempio +80 % contro 35 %, o 66 % contro 5,5 %) ha ragione più spesso di Jev. Senza la riduzione le +distanze più grandi, cioè i probabili errori di Jev, diventerebbero gli edge più grandi e le +prime scommesse. Il primo portafoglio reale ha perso proprio lì (vedi +[strategia](strategia.md#cosa-è-andato-storto-nel-primo-portafoglio)). + La media in log-odds è il modo standard di unire previsioni calibrate: la media semplice (`BLEND_METHOD=linear`, il metodo di prima) le rende sistematicamente troppo timide. Con `w = 0` il blended è il prezzo, con `w = 1` è Jev. @@ -116,13 +124,26 @@ per il NO la formula simmetrica sul prezzo del NO. | Prezzo di mercato (SÌ) | 0.35 | | Stima Jev | 0.80 | | Forza evidenze | 3/4 → 0.75 | -| Peso `w` | 0.5 × 0.75 = 0.375 | -| Probabilità blended | logit⁻¹(0.375 × logit 0.80 + 0.625 × logit 0.35) ≈ **0.533** | -| Edge | **+0.183** → `BUY_YES` | -| Puntata (Kelly semplice) | (0.533 − 0.35) / 0.65 × 0.25 ≈ **7.0 % del bankroll** | +| Distanza `d` | \|logit 0,80 − logit 0,35\| = 2,005 → riduzione 2 / 2,005 = 0,997 | +| Peso `w` | 0,25 × 0,75 × 0,997 ≈ 0,187 | +| Probabilità blended | logit⁻¹(0,187 × logit 0,80 + 0,813 × logit 0,35) ≈ **0,439** | +| Edge | **+0,089** → `BUY_YES` | +| Puntata (Kelly semplice) | (0,439 − 0,35) / 0,65 × 0,25 ≈ **3,4 % del bankroll** | La puntata effettiva la decide poi la [valutazione economica](strategia.md#valutazione-economica-e-portafoglio-simulato), che tiene conto di prezzo reale, costi, incertezza, tempo e limiti. +## Quali mercati restano fuori + +- **Mercati decisi dal prezzo di un asset** («Bitcoin sopra 84.000 $ il 24 settembre?», «ETH + arriva a 2.800 $ questa settimana?», «L'oro tocca 3.000 $?», «Bitcoin Up or Down»): Jev legge + notizie, non il prezzo del momento e la sua volatilità, mentre il prezzo del mercato li + contiene già. Si riconoscono dalla domanda (un asset, un livello di prezzo e una soglia come + sopra, sotto, tra, raggiunge, scende a: `backend/markets/kinds.py`). Con + `EXCLUDE_PRICE_MARKETS=true` (predefinito) non ricevono scommesse, e previsioni automatiche, + «Valuta tutti» e allerte li saltano, così non si spendono chiamate a Jev. Una previsione + chiesta a mano parte comunque. +- **Mercati vicini alla scadenza**: sotto le ore minime del preset (72 / 24 / 12) nessuna scommessa. + ## Mercati a più esiti Molti degli eventi più scambiati su Polymarket hanno più risposte possibili, una sola delle diff --git a/docs/it/strategia.md b/docs/it/strategia.md index 86a4d9a..2c33ccf 100644 --- a/docs/it/strategia.md +++ b/docs/it/strategia.md @@ -10,6 +10,14 @@ Un edge sulla carta non basta: la valutazione economica (`backend/betting/`) dec previsione conviene davvero, quanto puntare e a che prezzo massimo. Si calcola dopo ogni previsione e, dal vivo, nella scheda **Conviene?** del dettaglio mercato. +0. **La previsione, al prezzo di oggi.** La probabilità finale viene ricalcolata al prezzo + attuale (Jev calibrato unito al prezzo di adesso), non presa dalla previsione, che era unita + al prezzo di allora. Una previsione più vecchia di `FORECAST_MAX_AGE_HOURS` (6) ore, o il cui + prezzo si è mosso più di `FORECAST_MAX_PRICE_MOVE` (0,5 in log-odds: circa 12 punti intorno + al 50 %, 4 punti intorno al 90 %), richiede una previsione nuova prima di comprare. Niente + scommesse sui mercati decisi dal prezzo di un asset (vedi [il metodo](metodo.md#quali-mercati-restano-fuori)) + né su quelli che si chiudono prima delle ore minime del preset: a quel punto il prezzo + conosce già l'esito. 1. **Prezzo reale.** Legge il book del lato da comprare dal CLOB di Polymarket (API pubblica, sola lettura) e calcola il prezzo medio che pagheresti per quella cifra. Commissione (solo per chi compra dal book, come qui): `tasso × prezzo × (1 − prezzo)` per @@ -24,11 +32,14 @@ previsione e, dal vivo, nella scheda **Conviene?** del dettaglio mercato. `σ = w × √(p_jev (1 − p_jev) / (MODEL_PSEUDO_COUNT × evidenze + 1))`. Dopo 30 mercati risolti σ viene moltiplicata per `√(Brier osservato / Brier atteso)`, dove il Brier atteso è quello che avrebbero previsioni perfettamente calibrate (media di `p (1 − p)`): allargata - se le previsioni blended sono state troppo sicure, ristretta se sono state prudenti (fattore - tra 0,75 e 2). + se le previsioni blended sono state troppo sicure (fino a 2). Viene ristretta (fino a 0,75) + solo se la probabilità finale ha anche battuto il prezzo di mercato sugli stessi mercati, di + due errori standard: essere coerente con sé stessa non basta per puntare di più. 3. **Margine netto.** `p_prudente − (prezzo + commissione)` deve superare la soglia del preset. -4. **Tempo.** Il rendimento atteso prudente viene annualizzato sui giorni che mancano alla - scadenza e deve superare `RISK_FREE_RATE` + il premio del preset. +4. **Tempo e rendimento.** Il rendimento atteso prudente viene annualizzato sui giorni che + mancano alla scadenza e deve superare `RISK_FREE_RATE` + il premio del preset. Le scommesse + brevi superano sempre questo controllo (pochi punti in due giorni sono un tasso annuo enorme), + quindi anche il rendimento prudente della scommessa in sé deve raggiungere il minimo del preset. 5. **Quanto puntare.** Il capitale di Kelly è calcolato sul book, perché comprare di più peggiora il prezzo. Se ne prende una frazione (in base al preset) e poi si applicano i limiti per mercato, evento, categoria, totale investito, liquidità disponibile e quota del @@ -37,11 +48,11 @@ previsione e, dal vivo, nella scheda **Conviene?** del dettaglio mercato. 6. **Verdetto.** *Conviene*, *Conviene poco* (la puntata è stata ridotta a meno della metà dai limiti) oppure *Non conviene*, sempre con i motivi. -| Preset | Kelly | z | Margine netto | Premio annuo | Max per mercato | Max per evento | Max per categoria | Max investito | Liquidità min. | Scadenza max | -|---|---|---|---|---|---|---|---|---|---|---| -| Prudente | ×0,15 | 1,64 | 4 pt | 15% | 2% | 5% | 15% | 40% | 25.000 $ | 120 gg | -| Bilanciato | ×0,25 | 1,0 | 3 pt | 8% | 4% | 8% | 25% | 60% | 10.000 $ | 365 gg | -| Aggressivo | ×0,5 | 0,5 | 2 pt | 3% | 8% | 15% | 40% | 85% | 5.000 $ | 730 gg | +| Preset | Kelly | z | Margine netto | Rendimento min. | Premio annuo | Max per mercato | Max per evento | Max per categoria | Max investito | Liquidità min. | Scadenza | +|---|---|---|---|---|---|---|---|---|---|---|---| +| Prudente | ×0,15 | 1,64 | 4 pt | 10% | 15% | 2% | 5% | 15% | 40% | 25.000 $ | da 72 h a 120 gg | +| Bilanciato | ×0,25 | 1,0 | 3 pt | 6% | 8% | 4% | 8% | 25% | 60% | 10.000 $ | da 24 h a 365 gg | +| Aggressivo | ×0,5 | 0,5 | 2 pt | 3% | 3% | 8% | 15% | 40% | 85% | 5.000 $ | da 12 h a 730 gg | **Portafoglio simulato** (pagina *Portafoglio*): - **Scommesse automatiche:** ogni previsione con verdetto *Conviene* o *Conviene poco* diventa una scommessa virtuale al prezzo reale del momento, al massimo una aperta per mercato. @@ -78,6 +89,25 @@ previsione e, dal vivo, nella scheda **Conviene?** del dettaglio mercato. frazioni tra 0 e 1, gli importi in dollari, le ore in UTC. «CSV» scarica solo le scommesse (separatore virgola, decimali con il punto). +## Cosa è andato storto nel primo portafoglio + +Il primo portafoglio simulato reale ha perso il 24 % dei suoi 50 $ in un giorno e mezzo (13 +scommesse); i soli mercati sul prezzo delle crypto hanno perso 11 $ degli 11,86. Le cause, e +cosa è cambiato: + +| Causa | Cosa è cambiato | +|---|---| +| Jev stimava mercati decisi dal prezzo di Bitcoin o Ethereum (66 % contro un mercato al 5,5 %) senza vedere quel prezzo | I mercati sul prezzo restano fuori: niente scommesse, niente previsioni a pagamento | +| Due scommesse comprate 24 minuti prima della scadenza, contro un prezzo che conosceva già l'esito | Ore minime alla scadenza per preset (72 / 24 / 12) | +| Una scommessa manuale ha usato una previsione di 19 ore prima, con la probabilità finale unita al prezzo di allora: +22 $ attesi su 1,89 $ | La probabilità finale si ricalcola al prezzo attuale; previsioni vecchie o prezzi molto mossi richiedono una previsione nuova | +| Ogni scommessa nasceva da una distanza di circa 50 punti tra Jev e il prezzo: le distanze più grandi sono i probabili errori di Jev | Peso massimo di Jev da 0,5 a 0,25, e peso minore quanto più Jev è lontano dal prezzo | +| L'incertezza veniva ristretta (fattore 0,75) perché la probabilità finale era coerente con sé stessa, non perché battesse il prezzo | Si restringe solo se la probabilità finale batte il prezzo sui mercati risolti | +| Il rendimento annualizzato aveva un tetto del 1000 %, quindi il controllo sul rendimento non escludeva mai nulla | Rendimento prudente minimo per scommessa (10 / 6 / 3 %) | + +Con 13 scommesse il risultato in sé dimostra poco; le cause qui sopra sono strutturali. Lancia un +[backtest](verifica.md#backtest) e applica la calibrazione suggerita prima di fidarti di nuovo +dei segnali. + ## Strategia di acquisto e vendita La scheda **Cosa fare** del dettaglio di un mercato (e di ogni esito dei mercati a più esiti) diff --git a/docs/method.md b/docs/method.md index 57686bd..650524c 100644 --- a/docs/method.md +++ b/docs/method.md @@ -84,14 +84,22 @@ Liquid markets are usually already well calibrated, so Jev's estimate is not use with `logit p = ln(p / (1 − p))`. The two numbers are estimated by the backtest on resolved markets: `B < 1` softens an overconfident Jev, `B > 1` sharpens one that is too cautious. By default (0 and 1) the estimate stays as it is. -2. **Pooling with the price in log-odds**, with a weight that grows with the evidence strength: +2. **Pooling with the price in log-odds**, with a weight that grows with the evidence strength + and shrinks when Jev is very far from the price: ``` -w = MODEL_WEIGHT_MAX × evidence_strength +d = |logit(P_cal) − logit(price)| +w = MODEL_WEIGHT_MAX × evidence_strength × min(1, MODEL_DISAGREEMENT_LOGIT / d) logit(blended) = w × logit(P_cal) + (1 − w) × logit(price) edge = blended − price ``` +`MODEL_WEIGHT_MAX` is 0.25 and `MODEL_DISAGREEMENT_LOGIT` 2. A liquid market that disagrees +with Jev by more than 2 log-odds (for example 80 % against 35 %, or 66 % against 5.5 %) is more +often right than Jev: without the reduction the biggest disagreements, the likeliest Jev errors, +would become the biggest edges and the first bets. The first real portfolio lost money exactly +there (see [strategy](strategy.md#what-went-wrong-in-the-first-portfolio)). + Averaging in log-odds is the standard way to combine calibrated forecasts: a simple average (`BLEND_METHOD=linear`, the earlier method) makes them systematically too timid. With `w = 0` the blend is the price, with `w = 1` it is Jev. @@ -113,13 +121,25 @@ for NO the symmetric formula on the NO price. | Market price (YES) | 0.35 | | Jev estimate | 0.80 | | Evidence strength | 3/4 → 0.75 | -| Weight `w` | 0.5 × 0.75 = 0.375 | -| Blended probability | logit⁻¹(0.375 × logit 0.80 + 0.625 × logit 0.35) ≈ **0.533** | -| Edge | **+0.183** → `BUY_YES` | -| Stake (simple Kelly) | (0.533 − 0.35) / 0.65 × 0.25 ≈ **7.0 % of bankroll** | +| Distance `d` | \|logit 0.80 − logit 0.35\| = 2.005 → reduction 2 / 2.005 = 0.997 | +| Weight `w` | 0.25 × 0.75 × 0.997 ≈ 0.187 | +| Blended probability | logit⁻¹(0.187 × logit 0.80 + 0.813 × logit 0.35) ≈ **0.439** | +| Edge | **+0.089** → `BUY_YES` | +| Stake (simple Kelly) | (0.439 − 0.35) / 0.65 × 0.25 ≈ **3.4 % of bankroll** | The actual stake is then decided by the [economic assessment](strategy.md#economic-assessment-and-simulated-portfolio), which accounts for the real price, costs, uncertainty, time and limits. +## Which markets are left out + +- **Markets decided by an asset's price** («Will Bitcoin be above $84,000 on September 24?», + «Will ETH reach $2,800 this week?», «Will gold hit $3,000?», «Bitcoin Up or Down»): Jev reads + news, not the live price and its volatility, while the market price already reflects both. + They are recognised by the question (an asset, a price level and a threshold such as above, + below, between, reach, dip, hit: `backend/markets/kinds.py`). With `EXCLUDE_PRICE_MARKETS=true` + (the default) they get no bets, and automatic forecasts, «Assess all» and alerts skip them, so + no Jev call is spent there. A forecast asked by hand still runs. +- **Markets close to the end**: below the preset's minimum hours (72 / 24 / 12) no bet. + ## Multi-outcome markets Many of the most traded events on Polymarket have several possible answers, only one of which diff --git a/docs/strategy.md b/docs/strategy.md index 4fba06f..9a59cad 100644 --- a/docs/strategy.md +++ b/docs/strategy.md @@ -10,6 +10,14 @@ An edge on paper is not enough: the economic assessment (`backend/betting/`) dec forecast is really worth it, how much to stake and at what maximum price. It is computed after every forecast and, live, in the **Worth it?** card of the market detail page. +0. **The forecast, at today's price.** The blend is recomputed at the current price (calibrated + Jev pooled with the price of now), not taken from the forecast, which was pooled with the + price of its time. A forecast older than `FORECAST_MAX_AGE_HOURS` (6), or whose price has + moved more than `FORECAST_MAX_PRICE_MOVE` (0.5 in log-odds: about 12 points around 50 %, + 4 points around 90 %), needs a new one before buying. No bets on markets decided by an + asset's price (see [the method](method.md#which-markets-are-left-out)), nor on markets + ending in less than the preset's minimum hours: at that point the price already knows the + outcome. 1. **Real price.** It reads the book of the side to buy from Polymarket's CLOB (public API, read only) and computes the average price you would pay for that amount. Fee (only for those who take from the book, as here): `rate × price × (1 − price)` per @@ -23,11 +31,14 @@ every forecast and, live, in the **Worth it?** card of the market detail page. `σ = w × √(p_jev (1 − p_jev) / (MODEL_PSEUDO_COUNT × evidence + 1))`. After 30 resolved markets σ is multiplied by `√(observed Brier / expected Brier)`, where the expected Brier is the one perfectly calibrated forecasts would have (the average of `p (1 − p)`): widened if - the blended forecasts were overconfident, narrowed if they were cautious (a factor between - 0.75 and 2). + the blended forecasts were overconfident (up to 2). It is narrowed (down to 0.75) only when + the blend has also beaten the market price on the same markets, by two standard errors: a + blend that is merely consistent with itself does not earn bigger stakes. 3. **Net margin.** `p_prudent − (price + fee)` must exceed the preset's threshold. -4. **Time.** The prudent expected return is annualised over the days left before the end date - and must exceed `RISK_FREE_RATE` + the preset's premium. +4. **Time and return.** The prudent expected return is annualised over the days left before the + end date and must exceed `RISK_FREE_RATE` + the preset's premium. Short bets always pass that + test (a few points in two days are a huge annual rate), so the prudent return of the bet + itself must also reach the preset's minimum. 5. **How much to stake.** Kelly capital is computed on the book, because buying more worsens the price. A fraction of it is taken (depending on the preset) and then the limits per market, event, category, total invested, available cash and share of the book apply. The @@ -36,11 +47,11 @@ every forecast and, live, in the **Worth it?** card of the market detail page. 6. **Verdict.** *Worth it*, *Barely worth it* (the stake was cut to less than half by the limits) or *Not worth it*, always with the reasons. -| Preset | Kelly | z | Net margin | Annual premium | Max per market | Max per event | Max per category | Max invested | Min liquidity | Max end date | -|---|---|---|---|---|---|---|---|---|---|---| -| Prudent | ×0.15 | 1.64 | 4 pts | 15% | 2% | 5% | 15% | 40% | $25,000 | 120 d | -| Balanced | ×0.25 | 1.0 | 3 pts | 8% | 4% | 8% | 25% | 60% | $10,000 | 365 d | -| Aggressive | ×0.5 | 0.5 | 2 pts | 3% | 8% | 15% | 40% | 85% | $5,000 | 730 d | +| Preset | Kelly | z | Net margin | Min return | Annual premium | Max per market | Max per event | Max per category | Max invested | Min liquidity | End date | +|---|---|---|---|---|---|---|---|---|---|---|---| +| Prudent | ×0.15 | 1.64 | 4 pts | 10% | 15% | 2% | 5% | 15% | 40% | $25,000 | 72 h – 120 d | +| Balanced | ×0.25 | 1.0 | 3 pts | 6% | 8% | 4% | 8% | 25% | 60% | $10,000 | 24 h – 365 d | +| Aggressive | ×0.5 | 0.5 | 2 pts | 3% | 3% | 8% | 15% | 40% | 85% | $5,000 | 12 h – 730 d | **Simulated portfolio** (*Portfolio* page): - **Automatic bets:** every forecast with a *Worth it* or *Barely worth it* verdict becomes a virtual bet at the real price of the moment, at most one open per market. @@ -75,6 +86,24 @@ every forecast and, live, in the **Worth it?** card of the market detail page. 0–1, amounts in dollars, times in UTC. «CSV» downloads only the bets (comma separator, decimal point). +## What went wrong in the first portfolio + +The first real simulated portfolio lost 24 % of its $50 in a day and a half (13 bets); the +crypto price markets alone lost $11 of the $11.86. The causes, and what changed: + +| Cause | What changed | +|---|---| +| Jev estimated markets decided by the price of Bitcoin or Ethereum (66 % against a 5.5 % market) without seeing that price | Price markets are left out: no bets, no paid forecasts | +| Two bets were bought 24 minutes before the end, against a price that already knew the outcome | Minimum hours to the end per preset (72 / 24 / 12) | +| A manual bet used a 19-hour-old forecast, whose blend had been pooled with the price of then: +$22 expected on $1.89 | The blend is recomputed at the current price; old forecasts or big price moves need a new one | +| Every bet came from a disagreement of about 50 points between Jev and the price: the biggest disagreements are the likeliest Jev errors | Jev's maximum weight 0.5 → 0.25, and less weight the further Jev is from the price | +| The uncertainty was narrowed (factor 0.75) because the blend was consistent with itself, not because it beat the price | Narrowing only when the blend beats the price on resolved markets | +| Annualised returns were capped at 1000 %, so the return check never excluded anything | Minimum prudent return per bet (10 / 6 / 3 %) | + +With 13 bets the result itself proves little; the causes above are structural. Run a +[backtest](verification.md#backtest) and apply the calibration it suggests before trusting the +signals again. + ## Buy and sell strategy The **What to do** card of a market's detail page (and of each outcome of multi-outcome diff --git a/frontend/explain.js b/frontend/explain.js index 2f34694..1b94de2 100644 --- a/frontend/explain.js +++ b/frontend/explain.js @@ -92,7 +92,16 @@ export function explainCard(p, market, evidence, status) { step(2, t("Quanto contano le notizie"), GLOSSARY.weight, h("p", {}, t("Forza delle evidenze "), h("b", { class: "mono" }, pct(p.evidence_strength)), t(". Il peso di Jev è il peso massimo per la forza delle evidenze:")), - formula(t("w = {0} × {1} = ", pct(maxW), pct(p.evidence_strength)), h("b", {}, pct(w))), + (() => { + // Weight reduced because Jev was far from the price (MODEL_DISAGREEMENT_LOGIT), or computed + // with the parameters of the time: show the formula that gives the stored weight + const base = maxW * p.evidence_strength; + if (Math.abs(w - base) < 0.0005) return formula(t("w = {0} × {1} = ", pct(maxW), pct(p.evidence_strength)), h("b", {}, pct(w))); + if (w < base) return h("div", {}, + formula(t("w = {0} × {1} × {2} = ", pct(maxW), pct(p.evidence_strength), fmt.dec(w / base, 2)), h("b", {}, pct(w))), + h("p", { class: "muted small" }, t("Jev era molto lontano dal prezzo: il suo peso viene ridotto, perché su un mercato liquido le distanze più grandi sono più spesso errori di Jev che informazioni."))); + return h("div", {}, formula(t("w = "), h("b", {}, pct(w))), h("p", { class: "muted small" }, t("Calcolato con i parametri di allora."))); + })(), ), step(3, t("Probabilità finale (blended)"), GLOSSARY.blended, cal != null && Math.abs(cal - p.model_probability) >= 0.0005 diff --git a/frontend/i18n-en.js b/frontend/i18n-en.js index dc31d3b..7e2933f 100644 --- a/frontend/i18n-en.js +++ b/frontend/i18n-en.js @@ -1190,4 +1190,19 @@ export default { "CSV": "CSV", "Riepilogo, scommesse con tutti i dettagli, curva del capitale ed esclusioni, in un file Excel (si apre anche con Google Sheets e LibreOffice)": "Summary, bets with every detail, equity curve and exclusions, in an Excel file (it also opens in Google Sheets and LibreOffice)", "Solo le scommesse, in CSV (separatore virgola, decimali con il punto)": "Bets only, as CSV (comma separator, decimal point)", + "Kelly ×{0} · max {1}% per mercato · margine ≥ {2} pt · rendimento ≥ {3}% · da {4} ore a {5} giorni dalla scadenza": "Kelly ×{0} · max {1}% per market · margin ≥ {2} pts · return ≥ {3}% · from {4} hours to {5} days before the end", + "Peso ridotto se Jev dista dal prezzo più di": "Weight reduced when Jev is further from the price than", + "{0} in log-odds": "{0} in log-odds", + "mai": "never", + "Previsione valida per comprare": "Forecast valid for buying", + "{0} ore, prezzo mosso di meno di {1} punti": "{0} hours, price moved less than {1} points", + "Mercati sul prezzo di un asset": "Markets on an asset's price", + "esclusi": "excluded", + "ammessi": "allowed", + "{0} ore, prezzo mosso di meno di {1} in log-odds": "{0} hours, price moved less than {1} in log-odds", + "w = {0} × {1} × {2} = ": "w = {0} × {1} × {2} = ", + "Jev era molto lontano dal prezzo: il suo peso viene ridotto, perché su un mercato liquido le distanze più grandi sono più spesso errori di Jev che informazioni.": "Jev was very far from the price: its weight is reduced, because on a liquid market the biggest gaps are more often Jev's errors than information.", + "w = ": "w = ", + "Calcolato con i parametri di allora.": "Computed with the parameters of the time.", + "Se Jev è molto lontano dal prezzo (più di {0} in log-odds) il suo peso si riduce in proporzione: su un mercato liquido le distanze più grandi sono più spesso errori di Jev che informazioni, e altrimenti diventerebbero gli edge più grandi.": "If Jev is very far from the price (more than {0} in log-odds) its weight shrinks in proportion: on a liquid market the biggest gaps are more often Jev's errors than information, and they would otherwise become the biggest edges.", }; diff --git a/frontend/strategy.js b/frontend/strategy.js index a2b142e..c996e20 100644 --- a/frontend/strategy.js +++ b/frontend/strategy.js @@ -96,7 +96,7 @@ export function strategySection(plan, priceYes) { ); } -const STRUCTURAL = new Set(["illiquid", "too_far", "exposure_cap", "no_cash", "no_book"]); +const STRUCTURAL = new Set(["illiquid", "too_far", "too_close", "price_market", "exposure_cap", "no_cash", "no_book"]); /** One line for lists, from the evaluation stored with the forecast (prices of that moment). */ export function planLine(ev) { diff --git a/frontend/views/method.js b/frontend/views/method.js index cd4414d..0568d62 100644 --- a/frontend/views/method.js +++ b/frontend/views/method.js @@ -86,6 +86,7 @@ export async function viewMethod(ctx) { (st.jev_calib_a ?? 0) === 0 && (st.jev_calib_b ?? 1) === 1 ? t("Per ora nessuna correzione è attiva.") : t("Correzione attiva: a {0}, b {1}.", fmt.num3(st.jev_calib_a), fmt.num3(st.jev_calib_b))), h("p", { class: "formula" }, t("logit(Jev corretto) = a + b × logit(stima Jev)")), h("p", { class: "formula" }, t("w = {0} × forza delle evidenze", fmt.pct(maxW))), + st.model_disagreement_logit ? p(t("Se Jev è molto lontano dal prezzo (più di {0} in log-odds) il suo peso si riduce in proporzione: su un mercato liquido le distanze più grandi sono più spesso errori di Jev che informazioni, e altrimenti diventerebbero gli edge più grandi.", fmt.dec(st.model_disagreement_logit))) : null, st.blend_method === "linear" ? h("p", { class: "formula" }, t("blended = w × Jev corretto + (1 − w) × prezzo")) : h("p", { class: "formula" }, t("logit(blended) = w × logit(Jev corretto) + (1 − w) × logit(prezzo)")), diff --git a/frontend/views/portfolio.js b/frontend/views/portfolio.js index 280a8c7..3e7cc61 100644 --- a/frontend/views/portfolio.js +++ b/frontend/views/portfolio.js @@ -183,7 +183,7 @@ function settingsCard(ctx, data, refresh) { h("span", { class: "preset-name" }, p.label), h("span", { class: "preset-desc" }, p.description), h("span", { class: "preset-params mono" }, - t("Kelly ×{0} · max {1}% per mercato · margine ≥ {2} pt · ≤ {3} giorni", fmt.dec(p.kelly_scale), Math.round(p.max_market_frac * 100), Math.round(p.min_net_edge * 100), p.max_days)), + t("Kelly ×{0} · max {1}% per mercato · margine ≥ {2} pt · rendimento ≥ {3}% · da {4} ore a {5} giorni dalla scadenza", fmt.dec(p.kelly_scale), Math.round(p.max_market_frac * 100), Math.round(p.min_net_edge * 100), Math.round((p.min_roi ?? 0) * 100), p.min_hours_to_end ?? "–", p.max_days)), )); })); diff --git a/frontend/views/settings.js b/frontend/views/settings.js index e849ea7..be8be66 100644 --- a/frontend/views/settings.js +++ b/frontend/views/settings.js @@ -358,6 +358,9 @@ function parametersCard(ctx) { row(t("Pertinenza minima notizia–mercato"), fmt.pct(st.market_match_threshold), "MARKET_MATCH_THRESHOLD"), row(t("Finestra delle notizie"), t("{0} ore", st.market_news_window_hours ?? "–"), "MARKET_NEWS_WINDOW_HOURS"), row(t("Peso massimo di Jev"), fmt.pct(st.model_weight_max), "MODEL_WEIGHT_MAX"), + row(t("Peso ridotto se Jev dista dal prezzo più di"), st.model_disagreement_logit ? t("{0} in log-odds", fmt.dec(st.model_disagreement_logit)) : t("mai"), "MODEL_DISAGREEMENT_LOGIT"), + row(t("Previsione valida per comprare"), t("{0} ore, prezzo mosso di meno di {1} in log-odds", st.forecast_max_age_hours ?? "–", fmt.dec(st.forecast_max_price_move ?? 0)), "FORECAST_MAX_AGE_HOURS / FORECAST_MAX_PRICE_MOVE"), + row(t("Mercati sul prezzo di un asset"), st.exclude_price_markets ? t("esclusi") : t("ammessi"), "EXCLUDE_PRICE_MARKETS"), row(t("Come si uniscono Jev e prezzo"), st.blend_method === "linear" ? t("media lineare") : t("media in log-odds"), "BLEND_METHOD"), row(t("Calibrazione di Jev (a, b)"), `${fmt.num3(st.jev_calib_a ?? 0)} · ${fmt.num3(st.jev_calib_b ?? 1)}`, "JEV_CALIB_A / JEV_CALIB_B"), row(t("Chiamate a Jev per previsione"), String(st.jev_samples ?? 1), "JEV_SAMPLES"), diff --git a/tests/conftest.py b/tests/conftest.py index 370f144..baea787 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -19,6 +19,11 @@ os.environ["OLLAMA_BASE_URL"] = "http://127.0.0.1:9" # The existing tests check the Italian texts; tests/test_i18n.py covers English os.environ["APP_LANGUAGE"] = "it" +# The existing tests were written for the earlier forecast parameters; tests/test_trust.py covers +# the lower Jev weight, the disagreement reduction and the exclusion of price markets +os.environ["MODEL_WEIGHT_MAX"] = "0.5" +os.environ["MODEL_DISAGREEMENT_LOGIT"] = "0" +os.environ["EXCLUDE_PRICE_MARKETS"] = "false" # Fast client-side rate limits in tests (test_ratelimit.py sets its own) os.environ["JEV_RPM"] = "6000" os.environ["GROQ_RPM"] = "6000" diff --git a/tests/test_strategy.py b/tests/test_strategy.py index ad710d2..1107fbd 100644 --- a/tests/test_strategy.py +++ b/tests/test_strategy.py @@ -38,7 +38,8 @@ def test_buy_level_is_where_buying_stops_being_worth_it(): for price, ok in ((q, True), (q + 0.01, False)): p = f.p_yes(price) cost = price + fee_per_share(price, 400) - passes = p - price >= 0.05 and (p - PROFILE.z * f.sigma) - cost >= PROFILE.min_net_edge + p_cons = p - PROFILE.z * f.sigma + passes = p - price >= 0.05 and p_cons - cost >= PROFILE.min_net_edge and p_cons / cost - 1 >= PROFILE.min_roi assert passes == ok diff --git a/tests/test_trust.py b/tests/test_trust.py new file mode 100644 index 0000000..caebcf5 --- /dev/null +++ b/tests/test_trust.py @@ -0,0 +1,189 @@ +"""Guards added after the first real portfolio lost money (see docs/strategy.md, "What went wrong"). + +1. The forecast is re-evaluated at the current price; old forecasts, or a price that moved a lot, + need a new forecast before buying. +2. No bets close to the end of a market. +3. Markets decided by an asset's price are excluded. +4. Less trust in Jev: lower maximum weight, and less weight the further Jev is from the price. +5. The uncertainty is narrowed only when the blend has beaten the market. +6. A minimum absolute return per bet. +""" +from datetime import datetime, timedelta, timezone + +import pytest +from sqlalchemy import select + +from backend.betting import economics, portfolio +from backend.betting.plans import plan_for +from backend.betting.profiles import get_profile +from backend.config import settings +from backend.db.database import SessionLocal +from backend.db.models import Market, MarketPrediction, PaperBet +from backend.markets import forecast +from backend.markets.kinds import is_price_market +from tests.test_portfolio import NOW, book, make_market, make_prediction # noqa: F401 (fixture) + + +@pytest.fixture +def new_defaults(monkeypatch): + """The parameters the app ships with (the other tests run with the earlier ones).""" + monkeypatch.setattr(settings, "MODEL_WEIGHT_MAX", 0.25) + monkeypatch.setattr(settings, "MODEL_DISAGREEMENT_LOGIT", 2.0) + monkeypatch.setattr(settings, "EXCLUDE_PRICE_MARKETS", True) + + +# ---------- 4. Trust in Jev ---------- + +def test_the_real_losing_bets_give_no_signal_with_the_new_defaults(new_defaults): + """Jev far from liquid prices: 66% vs 5.5%, 67% vs 11.5%, 72% vs 7.5%. + With weight 0.5 and no reduction these became edges of 15–25 points; now they are noise. + (17% vs 68% on «Bitcoin above $84,000» still gives a signal: that one is stopped by the + minimum time to the end and by the exclusion of price markets.)""" + for jev, price, evidence in ((0.66, 0.055, 0.815), (0.67, 0.115, 0.7625), (0.72, 0.075, 0.6475)): + sig = forecast.compute_signal(jev, price, evidence) + assert sig.signal == "HOLD", (jev, price, sig) + assert abs(sig.edge) < settings.MIN_EDGE + + +def test_disagreement_reduces_the_weight_only_past_the_threshold(new_defaults): + assert forecast.disagreement_factor(0.6, 0.5) == 1.0 + far = forecast.disagreement_factor(0.66, 0.055) # ≈ 3.5 log-odds apart + assert far == pytest.approx(2.0 / abs(forecast.logit(0.66) - forecast.logit(0.055))) + # A moderate, well-supported disagreement still moves the blend + sig = forecast.compute_signal(0.8, 0.35, 1.0) + assert sig.model_weight == pytest.approx(0.25, abs=0.002) and sig.blended_probability > 0.4 + # Pooling at another price recomputes the reduction there + assert forecast.pool(0.8, 0.05, 0.25) < forecast.pool(0.8, 0.05, 0.25 * 1.0) + 1e-9 + assert forecast.pool(0.8, 0.05, 0.25) - 0.05 < 0.05 * 0.8 + + +# ---------- 1. Re-evaluation at the current price ---------- + +async def test_evaluation_uses_the_current_price_not_the_stored_blend(db, book): + async with SessionLocal() as s: + m = await make_market(s, yes=0.35) + p = await make_prediction(s, m, p_jev=0.8) # blended ≈ 0.53 at 0.35 + m.yes_price = 0.45 # moved 10 points: still within the limit + await s.commit() + ev = await portfolio.evaluate_prediction(s, m, p) + from backend.betting.plans import forecast_of + assert ev.p_side == pytest.approx(forecast_of(p, 0).p_yes(0.45), abs=1e-6) + assert ev.p_side != pytest.approx(p.blended_probability, abs=1e-3) + + +async def test_a_price_that_moved_or_an_old_forecast_needs_a_new_forecast(db, book): + """The manual bet of the real portfolio: forecast at 89.45%, price later 98.3%, stored blend + turned a 1.7¢ NO into +22 $ expected. Now it is refused until a new forecast.""" + async with SessionLocal() as s: + m = await make_market(s, yes=0.8945) + p = await make_prediction(s, m, p_jev=0.44, evidence=0.55) + m.yes_price = 0.983 # 8.85 points, but from 2.1 to 4.1 in log-odds + await s.commit() + ev = await portfolio.evaluate_prediction(s, m, p) + assert ev.verdict == "NO" and any(r["code"] == "stale_forecast" for r in ev.reasons) + plan = await plan_for(s, m, p, ev, get_profile("bilanciato")) + assert plan.action in ("WAIT", "NONE") + + m2 = await make_market(s, mid="m-old", yes=0.35, event="old") + p2 = await make_prediction(s, m2, p_jev=0.8) + p2.created_at = NOW - timedelta(hours=settings.FORECAST_MAX_AGE_HOURS + 1) + await s.commit() + ev2 = await portfolio.evaluate_prediction(s, m2, p2) + assert ev2.verdict == "NO" and any(r["code"] == "stale_forecast" for r in ev2.reasons) + plan2 = await plan_for(s, m2, p2, ev2, get_profile("bilanciato")) + assert plan2.action == "WAIT" and plan2.orders == [] + async with login_client_admin() as api: + r = await api.post(f"/markets/{m2.id}/paper-bet") + assert r.status_code == 409 + + +def login_client_admin(): + from tests.conftest import login_client + return login_client("admin") + + +# ---------- 2. Close to the end ---------- + +async def test_no_bet_close_to_the_end(db, book): + async with SessionLocal() as s: + m = await make_market(s, yes=0.35) + m.end_date = datetime.now(timezone.utc) + timedelta(minutes=25) + await s.commit() + p = await make_prediction(s, m, p_jev=0.8) + ev = await portfolio.apply_economics(s, m, p) + assert ev.verdict == "NO" and any(r["code"] == "too_close" and "25 minuti" in r["text"] for r in ev.reasons) + assert (await s.execute(select(PaperBet))).first() is None + plan = await plan_for(s, m, p, ev, get_profile("bilanciato")) + assert plan.action == "AVOID" and plan.levels == {} + + +# ---------- 3. Price markets ---------- + +def test_price_markets_are_recognised(): + assert is_price_market("Will the price of Bitcoin be above $84,000 on September 24?") + assert is_price_market("Will Bitcoin dip to $75,000 in September?") + assert is_price_market("Will Ethereum reach $2,800 September 21-27?") + assert is_price_market("Will gold hit $3,000 by December 31?") + assert is_price_market("Bitcoin Up or Down on September 25?") + assert not is_price_market("Will a Bitcoin ETF be approved in 2026?") + assert not is_price_market("Will 1 Fed rate hike happen in 2026?") + assert not is_price_market("Saudi Oil Pipeline (East-West) restarts by October 31?") + assert not is_price_market("Will inflation be above 3% in October?") + + +async def test_price_markets_get_no_bets_and_no_paid_forecasts(db, book, new_defaults): + from backend.markets.service import markets_needing_prediction + async with SessionLocal() as s: + m = await make_market(s, mid="m-btc", yes=0.055, event="btc") + m.question = "Will Bitcoin dip to $75,000 in September?" + await make_market(s, mid="m-fed", yes=0.35) + await s.commit() + ids = [x.id for x in await markets_needing_prediction(s, 10)] + assert "m-btc" not in ids and "m-fed" in ids + p = await make_prediction(s, m, p_jev=0.66) + ev = await portfolio.evaluate_prediction(s, m, p) + assert ev.verdict == "NO" and any(r["code"] == "price_market" for r in ev.reasons) + + +# ---------- 5. Uncertainty correction ---------- + +async def test_uncertainty_is_not_narrowed_unless_the_blend_beats_the_market(db): + """A blend that is consistent with itself (observed Brier below expected) but no better + than the price: the factor stays at 1 or above.""" + async with SessionLocal() as s: + for i in range(40): + yes = i % 2 == 0 + m = Market(id=f"r{i}", question=f"Resolved {i}?", yes_price=1.0 if yes else 0.0, closed=True, + resolved_yes=yes, volume=1e5, end_date=NOW - timedelta(days=1)) + s.add(m) + # price and blend equally good: 0.7 on the winner + p = 0.7 if yes else 0.3 + s.add(MarketPrediction(market_id=m.id, market_probability=p, model_probability=p, evidence_strength=0.5, + blended_probability=p, model_weight=0.1, edge=0.0, signal="HOLD")) + await s.commit() + assert await portfolio.calibration_factor(s) >= 1.0 + + # Now the blend is clearly better than the price on every market: narrowing is allowed + for pred in (await s.execute(select(MarketPrediction))).scalars().all(): + m = await s.get(Market, pred.market_id) + pred.market_probability = 0.5 + pred.blended_probability = 0.9 if m.resolved_yes else 0.1 + await s.commit() + assert await portfolio.calibration_factor(s) < 1.0 + + +# ---------- 6. Minimum absolute return ---------- + +def test_minimum_return_per_bet(): + profile = get_profile("bilanciato") + q = economics.Quote(asks=[(0.60, 1e5)], mid=0.595, fee_bps=0, source="book") + # 3.2 points of prudent margin on a 60¢ share: over 2 days it passes any annual threshold, + # but the bet returns 5.3%, under the 6% of the preset + ev = economics.evaluate(signal="BUY_YES", p_yes=0.632, sigma=0.0, quote=q, days=2, profile=profile, + equity=1000, available_cash=1000, exposure=economics.Exposure(), liquidity=1e6, + risk_free_rate=0.04) + assert ev.verdict == "NO" and [r["code"] for r in ev.reasons if r["blocking"]] == ["roi_too_low"] + ev = economics.evaluate(signal="BUY_YES", p_yes=0.70, sigma=0.0, quote=q, days=2, profile=profile, + equity=1000, available_cash=1000, exposure=economics.Exposure(), liquidity=1e6, + risk_free_rate=0.04) + assert ev.verdict in ("GO", "SMALL")