feat: simulate trading fees/slippage in backtest, compute real PnL for real trades

Backtest/walk-forward priced every fill at the exact candle close with zero
cost, making reported win rate/profit factor systematically more optimistic
than live trading. Added configurable taker-fee + slippage simulation
(defaults 0.1%/0.05% per fill) applied to every entry/exit, threaded through
walk-forward's grid search and both API endpoints.

sync_real_trades() hardcoded pnl=0 for every real trade needing it, silently
reporting break-even for real-money trades regardless of actual outcome.
Replaced with FIFO lot matching per (user, symbol, exchange), and fixed
orders.py to persist the exchange's actual average fill price instead of
the (always-None-for-market-orders) requested price, so there's real price
data to match against.

Also verified (and locked in with regression tests) that Divergence/SMC's
pivot-confirmation delay is already causally consistent between live and
backtest — no repaint, no look-ahead leak.

187 backend tests pass (+17).

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
Le
2026-07-04 14:50:36 +07:00
parent 1bcd15829e
commit 662586c6bc
12 changed files with 733 additions and 90 deletions
+104
View File
@@ -196,6 +196,110 @@ async def test_simulate_trades_active_from_index_skips_warmup_region(monkeypatch
assert all_signals == []
class TestFeeAndSlippage:
"""Regression tests for fee (o): backtest/walk-forward used to price
every fill at the exact candle close with zero cost, which made every
reported win rate/profit factor systematically more optimistic than
live trading could ever achieve. See DEFAULT_TAKER_FEE_PCT/
DEFAULT_SLIPPAGE_PCT and _fill_price/_open_position/_close_position in
backtest_engine.py.
"""
def test_fill_price_moves_against_the_trader(self):
mark = 100.0
slip = 0.001
# LONG entry buys -> fills higher than mark.
assert backtest_engine._fill_price(mark, "LONG", True, slip) == pytest.approx(100.1)
# LONG exit sells -> fills lower than mark.
assert backtest_engine._fill_price(mark, "LONG", False, slip) == pytest.approx(99.9)
# SHORT entry sells -> fills lower than mark.
assert backtest_engine._fill_price(mark, "SHORT", True, slip) == pytest.approx(99.9)
# SHORT exit buys -> fills higher than mark.
assert backtest_engine._fill_price(mark, "SHORT", False, slip) == pytest.approx(100.1)
async def test_reversal_exit_pnl_is_net_of_fees_and_slippage(self, monkeypatch, db_session):
_, symbol = await _seed_symbol(db_session)
base = datetime.now(timezone.utc) - timedelta(hours=40)
await _seed_candles(db_session, symbol.id, "1h", base, 40, timedelta(hours=1), lambda i: i)
candles = await backtest_engine._fetch_candles(db_session, symbol.id, "1h", since=base)
precomputed = backtest_engine._precompute_indicators(candles, "1h")
# STRONG_BUY at candle 30 (price 30), STRONG_SELL at candle 35
# (price 35) reverses and closes it — a straightforward winning
# LONG before costs. The STRONG_SELL that closes it also opens a
# new SHORT in the same step (existing reversal behavior), which
# then rides to END_OF_DATA — only the first (LONG) trade matters
# for this assertion.
fake, _ = _make_fake_score_fn(buy_at={30}, sell_at={35})
monkeypatch.setattr(backtest_engine, "_compute_adjusted_score", fake)
_, trades = backtest_engine._simulate_trades(candles, precomputed, Decimal("10"))
assert len(trades) == 2
trade = trades[0]
assert trade["direction"] == "LONG"
assert trade["exit_reason"] == "REVERSAL"
fee_pct = backtest_engine.DEFAULT_TAKER_FEE_PCT
slip = backtest_engine.DEFAULT_SLIPPAGE_PCT
expected_entry = 30.0 * (1 + slip)
expected_qty = 10.0 / expected_entry
expected_exit = 35.0 * (1 - slip)
expected_gross = (expected_exit - expected_entry) * expected_qty
expected_fees = (expected_entry + expected_exit) * expected_qty * fee_pct
expected_net = expected_gross - expected_fees
assert trade["entry_price"] == pytest.approx(expected_entry)
assert trade["exit_price"] == pytest.approx(expected_exit)
assert trade["gross_pnl"] == pytest.approx(expected_gross)
assert trade["fees"] == pytest.approx(expected_fees)
assert trade["pnl"] == pytest.approx(expected_net)
# The whole point of the fix: costs must actually eat into PnL.
assert trade["pnl"] < trade["gross_pnl"]
assert trade["fees"] > 0
async def test_zero_fee_and_slippage_matches_raw_price_pnl(self, monkeypatch, db_session):
"""fee_pct=0/slippage_pct=0 must reduce to the old frictionless
behavior — fills at the exact close, no cost — so existing callers
that don't care about costs (or want to see raw signal quality)
aren't forced into a changed baseline."""
_, symbol = await _seed_symbol(db_session)
base = datetime.now(timezone.utc) - timedelta(hours=40)
await _seed_candles(db_session, symbol.id, "1h", base, 40, timedelta(hours=1), lambda i: i)
candles = await backtest_engine._fetch_candles(db_session, symbol.id, "1h", since=base)
precomputed = backtest_engine._precompute_indicators(candles, "1h")
fake, _ = _make_fake_score_fn(buy_at={30}, sell_at={35})
monkeypatch.setattr(backtest_engine, "_compute_adjusted_score", fake)
_, trades = backtest_engine._simulate_trades(
candles, precomputed, Decimal("10"), fee_pct=0.0, slippage_pct=0.0,
)
trade = trades[0]
assert trade["direction"] == "LONG"
assert trade["entry_price"] == pytest.approx(30.0)
assert trade["exit_price"] == pytest.approx(35.0)
expected_qty = 10.0 / 30.0
assert trade["pnl"] == pytest.approx((35.0 - 30.0) * expected_qty)
assert trade["fees"] == pytest.approx(0.0)
async def test_compute_stats_reports_total_fees(self, monkeypatch, db_session):
_, symbol = await _seed_symbol(db_session)
base = datetime.now(timezone.utc) - timedelta(hours=40)
await _seed_candles(db_session, symbol.id, "1h", base, 40, timedelta(hours=1), lambda i: i)
candles = await backtest_engine._fetch_candles(db_session, symbol.id, "1h", since=base)
precomputed = backtest_engine._precompute_indicators(candles, "1h")
fake, _ = _make_fake_score_fn(buy_at={30}, sell_at={35})
monkeypatch.setattr(backtest_engine, "_compute_adjusted_score", fake)
all_signals, trades = backtest_engine._simulate_trades(candles, precomputed, Decimal("10"))
stats = backtest_engine._compute_stats(all_signals, trades)
assert stats["trades"]["total_fees"] == pytest.approx(
sum(t["fees"] for t in trades), abs=0.01,
)
assert stats["trades"]["total_fees"] > 0
def _build_candle_series(base, prices):
return [
Candle(
+64
View File
@@ -16,12 +16,15 @@ from app.services.indicator_service import (
detect_market_regime,
ema,
macd,
market_structure,
mfi,
obv,
obv_signal,
rsi,
sma,
vwap,
_find_pivot_highs,
_find_pivot_lows,
)
@@ -264,3 +267,64 @@ class TestDetectMarketRegime:
self._adx(22), bb, atr_pct=1.0, volume_data=None,
)
assert regime == "neutral"
class TestPivotCausalConsistency:
"""A pivot at index i is only knowable once `right` bars after it exist
(see `_find_pivot_highs`/`_find_pivot_lows`'s definition). Live trading
(`candle_service.get_indicators`) calls `detect_divergence`/
`market_structure` on data "as of now" with no future bars — these
tests lock in that this is self-consistent (never claims a pivot it
can't yet know about) and never repaints (a pivot's status, once
confirmable, doesn't change as more future data arrives). This is the
same causal delay backtest_engine.py's explicit confirmed-pointer
bookkeeping was built to match (see its _PIVOT_LOOKBACK handling and
test_scores_are_causal_future_prices_dont_change_earlier_scores) — if
either side ever stopped honoring it, backtest and live would silently
diverge on how early Divergence/SMC signals fire.
"""
def _wavy_prices(self, n: int, seed: int = 0) -> list[float]:
return [100 + math.sin((i + seed) / 3.0) * 10 + (i % 5) for i in range(n)]
def test_pivot_highs_never_flag_the_last_right_bars(self):
prices = self._wavy_prices(40)
right = 3
highs = _find_pivot_highs(prices, left=right, right=right)
assert all(v is None for v in highs[-right:])
def test_pivot_lows_never_flag_the_last_right_bars(self):
prices = self._wavy_prices(40, seed=2)
right = 3
lows = _find_pivot_lows(prices, left=right, right=right)
assert all(v is None for v in lows[-right:])
def test_pivot_status_never_repaints_once_confirmable(self):
prices = self._wavy_prices(30)
right = 3
highs_before = _find_pivot_highs(prices, right, right)
lows_before = _find_pivot_lows(prices, right, right)
# More candles arrive — bars that already had enough future
# confirmation must keep the exact same pivot/non-pivot verdict.
extended = prices + [95.0, 130.0, 80.0, 140.0, 70.0, 150.0]
highs_after = _find_pivot_highs(extended, right, right)
lows_after = _find_pivot_lows(extended, right, right)
confirmable = len(prices) - right
assert highs_after[:confirmable] == highs_before[:confirmable]
assert lows_after[:confirmable] == lows_before[:confirmable]
def test_market_structure_swings_never_repaint_as_more_candles_arrive(self):
prices = self._wavy_prices(30, seed=1)
candles = [candle(p + 2, p - 2, p) for p in prices]
ms_before = market_structure(candles, pivot_lookback=3)
more_candles = candles + [
candle(122, 98, 120), candle(92, 68, 70), candle(142, 118, 140),
]
ms_after = market_structure(more_candles, pivot_lookback=3)
confirmable = len(candles) - 3
assert ms_after["swing_highs"][:confirmable] == ms_before["swing_highs"][:confirmable]
assert ms_after["swing_lows"][:confirmable] == ms_before["swing_lows"][:confirmable]
@@ -35,6 +35,8 @@ class StubAdapter:
order_id="STUB-ORDER-1",
filled=req.amount,
status="closed",
average=None,
price=None,
)
+182
View File
@@ -22,12 +22,15 @@ from sqlalchemy import select
from app.models import Exchange, HypotheticalTrade, Signal, Symbol, User
from app.models.candle import Candle
from app.models.real_trade import RealTrade
from app.services.trade_executor import (
MAX_OPEN_TRADES,
STRONG_BUY,
STRONG_SELL,
_calculate_pnl,
_recompute_realized_pnl,
execute_signal_trade,
sync_real_trades,
)
@@ -296,3 +299,182 @@ async def test_hybrid_eviction_uses_each_trades_own_symbol_price_not_incoming_si
assert closed_trades[0].exit_price == Decimal("50"), "exit price must come from REAL/USDT's own candle, not the signal's 1000"
assert closed_trades[0].pnl == Decimal("-50")
assert any(t.symbol == "NEW/USDT" and t.status == "OPEN" for t in all_trades)
def make_real_trade(user_id, symbol="BTC/USDT", side="buy", amount="1", price="100",
filled_amount=None, status="filled", created_at=None) -> RealTrade:
return RealTrade(
user_id=user_id,
exchange="mexc",
symbol=symbol,
side=side,
order_type="market",
amount=Decimal(amount),
price=Decimal(price) if price is not None else None,
filled_amount=Decimal(filled_amount if filled_amount is not None else amount),
status=status,
created_at=created_at or datetime.now(timezone.utc),
)
class TestRecomputeRealizedPnl:
"""Regression tests for the sync_real_trades() fix: `RealTrade` rows are
individual order fills (one buy or one sell), not paired open/close
positions — PnL was previously hardcoded to 0 for every trade needing
it, silently reporting "break-even" for real trades that may have won
or lost real money. `_recompute_realized_pnl` matches opposing fills
FIFO per (user, symbol, exchange) instead.
"""
async def test_opening_fill_has_no_pnl_until_something_closes_it(self, db_session):
user = make_user()
db_session.add(user)
await db_session.flush()
buy = make_real_trade(user.id, side="buy", price="100")
db_session.add(buy)
await db_session.flush()
updated = await _recompute_realized_pnl(db_session)
assert updated == 0
assert buy.pnl is None
async def test_full_close_realizes_pnl_on_the_closing_fill(self, db_session):
user = make_user()
db_session.add(user)
await db_session.flush()
t0 = datetime.now(timezone.utc) - timedelta(minutes=10)
buy = make_real_trade(user.id, side="buy", amount="1", price="100", created_at=t0)
sell = make_real_trade(user.id, side="sell", amount="1", price="110", created_at=t0 + timedelta(minutes=5))
db_session.add_all([buy, sell])
await db_session.flush()
updated = await _recompute_realized_pnl(db_session)
assert updated == 1
assert buy.pnl is None, "opening fill never realizes PnL on itself"
assert sell.pnl == Decimal("10")
assert sell.pnl_percent == Decimal("10")
async def test_partial_close_realizes_pnl_only_on_matched_quantity(self, db_session):
user = make_user()
db_session.add(user)
await db_session.flush()
t0 = datetime.now(timezone.utc) - timedelta(minutes=10)
buy = make_real_trade(user.id, amount="2", price="100", side="buy", created_at=t0)
sell = make_real_trade(user.id, amount="1", price="110", side="sell", created_at=t0 + timedelta(minutes=5))
db_session.add_all([buy, sell])
await db_session.flush()
await _recompute_realized_pnl(db_session)
assert sell.pnl == Decimal("10"), "only the 1 matched unit realizes, not the full 2-unit lot"
assert sell.pnl_percent == Decimal("10")
async def test_short_side_profits_when_price_falls(self, db_session):
user = make_user()
db_session.add(user)
await db_session.flush()
t0 = datetime.now(timezone.utc) - timedelta(minutes=10)
# sell-first (open short) then buy-to-cover lower -> profit
open_short = make_real_trade(user.id, amount="1", price="100", side="sell", created_at=t0)
cover = make_real_trade(user.id, amount="1", price="90", side="buy", created_at=t0 + timedelta(minutes=5))
db_session.add_all([open_short, cover])
await db_session.flush()
await _recompute_realized_pnl(db_session)
assert open_short.pnl is None
assert cover.pnl == Decimal("10")
async def test_trades_without_price_are_skipped_not_matched(self, db_session):
"""A market-order fill persisted before the orders.py fix (no
order.average recorded) has price=None — it must not be treated as
a zero-cost lot that corrupts FIFO matching for real, priced fills."""
user = make_user()
db_session.add(user)
await db_session.flush()
t0 = datetime.now(timezone.utc) - timedelta(minutes=10)
unpriced_buy = make_real_trade(user.id, amount="1", price=None, side="buy", created_at=t0)
sell = make_real_trade(user.id, amount="1", price="110", side="sell", created_at=t0 + timedelta(minutes=5))
db_session.add_all([unpriced_buy, sell])
await db_session.flush()
await _recompute_realized_pnl(db_session)
assert unpriced_buy.pnl is None
assert sell.pnl is None, "sell opens a new SHORT lot since the unpriced buy couldn't be matched"
async def test_different_symbols_do_not_cross_match(self, db_session):
user = make_user()
db_session.add(user)
await db_session.flush()
t0 = datetime.now(timezone.utc) - timedelta(minutes=10)
buy_btc = make_real_trade(user.id, symbol="BTC/USDT", amount="1", price="100", side="buy", created_at=t0)
sell_eth = make_real_trade(user.id, symbol="ETH/USDT", amount="1", price="110", side="sell", created_at=t0 + timedelta(minutes=5))
db_session.add_all([buy_btc, sell_eth])
await db_session.flush()
await _recompute_realized_pnl(db_session)
assert buy_btc.pnl is None
assert sell_eth.pnl is None, "ETH sell must open its own SHORT lot, not close the unrelated BTC buy"
class TestSyncRealTrades:
async def test_never_filled_stale_order_is_zeroed_out(self, session_factory, monkeypatch):
import app.services.trade_executor as trade_executor_module
monkeypatch.setattr(trade_executor_module, "async_session_factory", session_factory)
async with session_factory() as db:
user = make_user()
db.add(user)
await db.flush()
stale = make_real_trade(
user.id, side="buy", amount="1", price=None, filled_amount="0",
status="open", created_at=datetime.now(timezone.utc) - timedelta(hours=25),
)
db.add(stale)
await db.commit()
stale_id = stale.id
await sync_real_trades()
async with session_factory() as db:
refreshed = await db.get(RealTrade, stale_id)
assert refreshed.status == "closed"
assert refreshed.pnl == Decimal("0")
async def test_stale_order_with_a_real_fill_gets_real_pnl_not_zero(self, session_factory, monkeypatch):
"""The bug this fixes: a stale 'open' order that DID partially fill
used to be force-closed with pnl=0 regardless of what actually
happened. If a later trade already closed that fill's position, the
FIFO recompute (run every sync) must report the real PnL instead."""
import app.services.trade_executor as trade_executor_module
monkeypatch.setattr(trade_executor_module, "async_session_factory", session_factory)
t0 = datetime.now(timezone.utc) - timedelta(hours=25)
async with session_factory() as db:
user = make_user()
db.add(user)
await db.flush()
stale_buy = make_real_trade(
user.id, side="buy", amount="1", price="100", filled_amount="1",
status="open", created_at=t0,
)
closing_sell = make_real_trade(
user.id, side="sell", amount="1", price="120", filled_amount="1",
status="filled", created_at=t0 + timedelta(minutes=1),
)
db.add_all([stale_buy, closing_sell])
await db.commit()
stale_id, sell_id = stale_buy.id, closing_sell.id
await sync_real_trades()
async with session_factory() as db:
stale_refreshed = await db.get(RealTrade, stale_id)
sell_refreshed = await db.get(RealTrade, sell_id)
assert stale_refreshed.status == "closed", "stuck-open order is still force-closed after 24h"
assert stale_refreshed.pnl is None, "opening fill itself never carries the realized PnL"
assert sell_refreshed.pnl == Decimal("20"), "the closing fill must show the real, non-zero PnL"
+2 -2
View File
@@ -89,7 +89,7 @@ async def test_grid_search_picks_the_best_scoring_combo(monkeypatch):
best one and reports its stats."""
good_params = {"strong_threshold": 4.5, "signal_threshold": 1.5, "max_hold_candles": 96}
def fake_run_combo(candles, scores_series, trade_size, params):
def fake_run_combo(candles, scores_series, trade_size, params, fee_pct=None, slippage_pct=None):
if params == good_params:
trades = [{"pnl": 10.0, "status": "CLOSED", "entry_price": 100.0, "exit_price": 110.0} for _ in range(10)]
else:
@@ -113,7 +113,7 @@ async def test_grid_search_picks_the_best_scoring_combo(monkeypatch):
@pytest.mark.asyncio
async def test_grid_search_falls_back_when_every_combo_too_sparse(monkeypatch):
def fake_run_combo(candles, scores_series, trade_size, params):
def fake_run_combo(candles, scores_series, trade_size, params, fee_pct=None, slippage_pct=None):
# 1 trade, below MIN_TRADES_PER_FOLD
return [], [{"pnl": 1.0, "status": "CLOSED", "entry_price": 100.0, "exit_price": 101.0}]