446 lines
16 KiB
Python
446 lines
16 KiB
Python
#!/usr/bin/env python
|
|
"""Round-trip trade analytics over the append-only ledger (trades.jsonl).
|
|
|
|
Reconstructs closed round trips by FIFO-matching sells against buys per
|
|
(strategy, symbol), then derives the performance statistics that actually drive
|
|
parameter decisions: expectancy, profit factor, win rate, drawdown, and
|
|
attribution by strategy / symbol / entry signal / exit reason / hour of day.
|
|
|
|
Pure and offline: reads only the ledger, never connects to IB. Used by
|
|
dashboard.py (HTML report) and usable directly:
|
|
|
|
.venv/bin/python analytics.py # whole ledger
|
|
.venv/bin/python analytics.py --since 2026-08-01
|
|
.venv/bin/python analytics.py --json # machine-readable dump
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import json
|
|
import math
|
|
import os
|
|
from collections import defaultdict, deque
|
|
from dataclasses import asdict, dataclass, field
|
|
from datetime import datetime
|
|
from pathlib import Path
|
|
from typing import Callable, Iterable, Optional
|
|
|
|
LEDGER = Path(os.environ.get("TRADES_LEDGER", Path(__file__).resolve().parent / "trades.jsonl"))
|
|
|
|
STRATEGY_SHORT = {
|
|
"MAStockStrategy": "MAStock",
|
|
"ShortTermMAVWAPStrategy": "ShortTerm",
|
|
"MeanReversionStrategy": "MeanRev",
|
|
"ForexMAStrategy": "Forex",
|
|
}
|
|
|
|
BUY_TYPES = ("seed", "buy")
|
|
SELL_TYPES = ("sell", "sell_external")
|
|
|
|
|
|
def short(name: str) -> str:
|
|
return STRATEGY_SHORT.get(name, name[:12])
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
# round-trip reconstruction
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
|
|
@dataclass
|
|
class RoundTrip:
|
|
"""One closed position slice: a sell matched against an earlier buy lot."""
|
|
|
|
strategy: str
|
|
symbol: str
|
|
qty: float
|
|
entry_ts: Optional[str]
|
|
exit_ts: str
|
|
entry_price: float
|
|
exit_price: float
|
|
entry_reason: str
|
|
exit_reason: str
|
|
gross_pnl: float
|
|
commission: float
|
|
estimated: bool # exit price was estimated (missed external fill)
|
|
seeded_entry: bool # entry lot predates the ledger (basis approximate)
|
|
|
|
@property
|
|
def pnl(self) -> float:
|
|
"""Net realized P&L after commissions."""
|
|
return self.gross_pnl - self.commission
|
|
|
|
@property
|
|
def pnl_pct(self) -> float:
|
|
"""Net return on the entry notional, in percent."""
|
|
cost = self.entry_price * self.qty
|
|
return (self.pnl / cost * 100) if cost else 0.0
|
|
|
|
@property
|
|
def hold_minutes(self) -> Optional[float]:
|
|
if not self.entry_ts:
|
|
return None
|
|
try:
|
|
a = datetime.fromisoformat(self.entry_ts)
|
|
b = datetime.fromisoformat(self.exit_ts)
|
|
except ValueError:
|
|
return None
|
|
return (b - a).total_seconds() / 60
|
|
|
|
@property
|
|
def won(self) -> bool:
|
|
return self.pnl > 0
|
|
|
|
def as_row(self) -> dict:
|
|
d = asdict(self)
|
|
d.update(
|
|
pnl=self.pnl,
|
|
pnl_pct=self.pnl_pct,
|
|
hold_minutes=self.hold_minutes,
|
|
won=self.won,
|
|
strategy_short=short(self.strategy),
|
|
)
|
|
return d
|
|
|
|
|
|
@dataclass
|
|
class OpenLot:
|
|
"""A buy lot still (partly) unmatched at the end of the ledger."""
|
|
|
|
strategy: str
|
|
symbol: str
|
|
qty: float
|
|
price: float
|
|
ts: Optional[str]
|
|
reason: str
|
|
seeded: bool
|
|
|
|
def as_row(self) -> dict:
|
|
d = asdict(self)
|
|
d["strategy_short"] = short(self.strategy)
|
|
d["cost"] = self.qty * self.price
|
|
return d
|
|
|
|
|
|
def _read_records(path: Path) -> list[dict]:
|
|
"""Parse the JSONL ledger, skipping blank and malformed lines."""
|
|
records = []
|
|
for lineno, line in enumerate(path.read_text().splitlines(), 1):
|
|
line = line.strip()
|
|
if not line:
|
|
continue
|
|
try:
|
|
records.append(json.loads(line))
|
|
except json.JSONDecodeError:
|
|
print(f"warning: {path.name}:{lineno} is not valid JSON, skipped")
|
|
# the ledger is append-only so it is already chronological, but a manual
|
|
# repair could have disturbed that; sorting keeps FIFO matching honest
|
|
records.sort(key=lambda r: r.get("ts", ""))
|
|
return records
|
|
|
|
|
|
def build_round_trips(records: Iterable[dict]) -> tuple[list[RoundTrip], list[OpenLot], float]:
|
|
"""FIFO-match sells against buys per (strategy, symbol).
|
|
|
|
Returns (closed round trips, still-open lots, quantity sold with no known
|
|
cost basis). A sell that exhausts the available lots is reported in that
|
|
last figure rather than being silently priced at zero.
|
|
"""
|
|
lots: dict[tuple[str, str], deque] = defaultdict(deque)
|
|
trips: list[RoundTrip] = []
|
|
unmatched_qty = 0.0
|
|
|
|
for r in records:
|
|
rtype = r.get("type")
|
|
key = (r.get("strategy", "?"), r.get("symbol", "?"))
|
|
qty = float(r.get("qty", 0) or 0)
|
|
price = float(r.get("price", 0) or 0)
|
|
if qty <= 0:
|
|
continue
|
|
|
|
if rtype in BUY_TYPES:
|
|
lots[key].append({
|
|
"qty": qty,
|
|
"price": price,
|
|
"ts": r.get("ts"),
|
|
"reason": r.get("reason", ""),
|
|
# commission is per-fill; carry it per share so partial
|
|
# matches take a proportional slice
|
|
"comm_per_share": (float(r.get("commission", 0) or 0) / qty),
|
|
"seeded": rtype == "seed",
|
|
})
|
|
|
|
elif rtype in SELL_TYPES:
|
|
remaining = qty
|
|
sell_comm_per_share = (float(r.get("commission", 0) or 0) / qty)
|
|
dq = lots[key]
|
|
while remaining > 1e-9 and dq:
|
|
lot = dq[0]
|
|
take = min(lot["qty"], remaining)
|
|
trips.append(RoundTrip(
|
|
strategy=key[0],
|
|
symbol=key[1],
|
|
qty=take,
|
|
entry_ts=lot["ts"],
|
|
exit_ts=r.get("ts", ""),
|
|
entry_price=lot["price"],
|
|
exit_price=price,
|
|
entry_reason=lot["reason"],
|
|
exit_reason=r.get("reason", "") or ("external" if rtype == "sell_external" else ""),
|
|
gross_pnl=(price - lot["price"]) * take,
|
|
commission=(lot["comm_per_share"] + sell_comm_per_share) * take,
|
|
estimated=bool(r.get("est")),
|
|
seeded_entry=lot["seeded"],
|
|
))
|
|
lot["qty"] -= take
|
|
remaining -= take
|
|
if lot["qty"] <= 1e-9:
|
|
dq.popleft()
|
|
if remaining > 1e-9:
|
|
unmatched_qty += remaining
|
|
|
|
open_lots = [
|
|
OpenLot(strategy=s, symbol=sym, qty=lot["qty"], price=lot["price"],
|
|
ts=lot["ts"], reason=lot["reason"], seeded=lot["seeded"])
|
|
for (s, sym), dq in lots.items()
|
|
for lot in dq
|
|
if lot["qty"] > 1e-9
|
|
]
|
|
return trips, open_lots, unmatched_qty
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
# statistics
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
|
|
def _mean(xs: list[float]) -> float:
|
|
return sum(xs) / len(xs) if xs else 0.0
|
|
|
|
|
|
def _median(xs: list[float]) -> float:
|
|
if not xs:
|
|
return 0.0
|
|
s = sorted(xs)
|
|
mid = len(s) // 2
|
|
return s[mid] if len(s) % 2 else (s[mid - 1] + s[mid]) / 2
|
|
|
|
|
|
def summarize(trips: list[RoundTrip]) -> dict:
|
|
"""Core performance statistics for a set of round trips."""
|
|
n = len(trips)
|
|
if n == 0:
|
|
return {"n": 0, "pnl": 0.0, "win_rate": 0.0, "expectancy": 0.0,
|
|
"profit_factor": None, "avg_win": 0.0, "avg_loss": 0.0,
|
|
"gross_profit": 0.0, "gross_loss": 0.0, "commission": 0.0,
|
|
"avg_pnl_pct": 0.0, "median_pnl_pct": 0.0,
|
|
"best": 0.0, "worst": 0.0, "avg_hold_minutes": None,
|
|
"breakeven_win_rate": None, "payoff_ratio": None}
|
|
|
|
pnls = [t.pnl for t in trips]
|
|
wins = [p for p in pnls if p > 0]
|
|
losses = [p for p in pnls if p <= 0]
|
|
gross_profit = sum(wins)
|
|
gross_loss = -sum(losses) # positive magnitude
|
|
holds = [h for h in (t.hold_minutes for t in trips) if h is not None]
|
|
|
|
avg_win = _mean(wins)
|
|
avg_loss = _mean([-p for p in losses]) # positive magnitude
|
|
payoff = (avg_win / avg_loss) if avg_loss else None
|
|
# win rate this strategy would need just to break even, given its own
|
|
# realized win/loss sizes - the number to compare the actual win rate against
|
|
breakeven_wr = (1 / (1 + payoff) * 100) if payoff else None
|
|
|
|
return {
|
|
"n": n,
|
|
"pnl": sum(pnls),
|
|
"win_rate": len(wins) / n * 100,
|
|
"expectancy": _mean(pnls),
|
|
"profit_factor": (gross_profit / gross_loss) if gross_loss else None,
|
|
"avg_win": avg_win,
|
|
"avg_loss": avg_loss,
|
|
"payoff_ratio": payoff,
|
|
"breakeven_win_rate": breakeven_wr,
|
|
"gross_profit": gross_profit,
|
|
"gross_loss": gross_loss,
|
|
"commission": sum(t.commission for t in trips),
|
|
"avg_pnl_pct": _mean([t.pnl_pct for t in trips]),
|
|
"median_pnl_pct": _median([t.pnl_pct for t in trips]),
|
|
"best": max(pnls),
|
|
"worst": min(pnls),
|
|
"avg_hold_minutes": _mean(holds) if holds else None,
|
|
"n_wins": len(wins),
|
|
"n_losses": len(losses),
|
|
}
|
|
|
|
|
|
def equity_curve(trips: list[RoundTrip]) -> list[dict]:
|
|
"""Cumulative realized P&L in exit order, with running peak and drawdown."""
|
|
ordered = sorted(trips, key=lambda t: t.exit_ts)
|
|
curve, cum, peak = [], 0.0, 0.0
|
|
for t in ordered:
|
|
cum += t.pnl
|
|
peak = max(peak, cum)
|
|
curve.append({
|
|
"ts": t.exit_ts,
|
|
"cum_pnl": cum,
|
|
"peak": peak,
|
|
"drawdown": cum - peak,
|
|
"pnl": t.pnl,
|
|
"symbol": t.symbol,
|
|
"strategy_short": short(t.strategy),
|
|
})
|
|
return curve
|
|
|
|
|
|
def max_drawdown(curve: list[dict]) -> float:
|
|
"""Largest peak-to-trough decline of the realized equity curve (<= 0)."""
|
|
return min((p["drawdown"] for p in curve), default=0.0)
|
|
|
|
|
|
def group_stats(trips: list[RoundTrip], keyfn: Callable[[RoundTrip], str]) -> dict[str, dict]:
|
|
"""Summarize round trips bucketed by an arbitrary key."""
|
|
buckets: dict[str, list[RoundTrip]] = defaultdict(list)
|
|
for t in trips:
|
|
buckets[keyfn(t) or "(none)"].append(t)
|
|
return {k: summarize(v) for k, v in buckets.items()}
|
|
|
|
|
|
def daily_pnl(trips: list[RoundTrip]) -> dict[str, float]:
|
|
"""Net realized P&L per calendar day, keyed by exit date."""
|
|
out: dict[str, float] = defaultdict(float)
|
|
for t in trips:
|
|
out[t.exit_ts[:10]] += t.pnl
|
|
return dict(sorted(out.items()))
|
|
|
|
|
|
def _exit_hour(t: RoundTrip) -> str:
|
|
try:
|
|
return f"{datetime.fromisoformat(t.exit_ts).hour:02d}:00"
|
|
except ValueError:
|
|
return "(none)"
|
|
|
|
|
|
def analyze(path: Path = LEDGER, since: Optional[str] = None,
|
|
until: Optional[str] = None) -> dict:
|
|
"""Full analysis bundle. `since`/`until` are inclusive YYYY-MM-DD exit-date bounds.
|
|
|
|
Round trips are always reconstructed from the *whole* ledger so cost basis
|
|
stays correct; the date filter is applied to the resulting closed trades.
|
|
"""
|
|
records = _read_records(path)
|
|
trips, open_lots, unmatched = build_round_trips(records)
|
|
|
|
if since:
|
|
trips = [t for t in trips if t.exit_ts[:10] >= since]
|
|
if until:
|
|
trips = [t for t in trips if t.exit_ts[:10] <= until]
|
|
|
|
curve = equity_curve(trips)
|
|
return {
|
|
"ledger": str(path),
|
|
"generated": datetime.now().isoformat(timespec="seconds"),
|
|
"since": since,
|
|
"until": until,
|
|
"n_records": len(records),
|
|
"unmatched_sell_qty": unmatched,
|
|
"overall": summarize(trips),
|
|
"max_drawdown": max_drawdown(curve),
|
|
"by_strategy": group_stats(trips, lambda t: short(t.strategy)),
|
|
"by_symbol": group_stats(trips, lambda t: t.symbol),
|
|
"by_entry_reason": group_stats(trips, lambda t: t.entry_reason),
|
|
"by_exit_reason": group_stats(trips, lambda t: t.exit_reason),
|
|
"by_exit_hour": group_stats(trips, _exit_hour),
|
|
"by_weekday": group_stats(
|
|
trips,
|
|
lambda t: (datetime.fromisoformat(t.exit_ts).strftime("%a")
|
|
if t.exit_ts else "(none)"),
|
|
),
|
|
"daily_pnl": daily_pnl(trips),
|
|
"equity_curve": curve,
|
|
"trips": [t.as_row() for t in sorted(trips, key=lambda x: x.exit_ts, reverse=True)],
|
|
"open_lots": [l.as_row() for l in sorted(open_lots, key=lambda x: (x.symbol, x.strategy))],
|
|
}
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
# CLI
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
|
|
def _fmt(v, spec="+.2f", dash="-"):
|
|
return dash if v is None else format(v, spec)
|
|
|
|
|
|
def _print_table(title: str, stats: dict[str, dict]):
|
|
if not stats:
|
|
return
|
|
print(f"\n {title}")
|
|
print(f" {'bucket':<14}{'n':>4}{'net P&L':>11}{'win%':>7}{'exp':>9}{'PF':>7}{'be.win%':>9}")
|
|
for k, s in sorted(stats.items(), key=lambda kv: -kv[1]["pnl"]):
|
|
print(f" {k[:14]:<14}{s['n']:>4}{s['pnl']:>+11.2f}{s['win_rate']:>7.1f}"
|
|
f"{s['expectancy']:>+9.2f}{_fmt(s['profit_factor'], '.2f', ' n/a'):>7}"
|
|
f"{_fmt(s['breakeven_win_rate'], '.1f', ' n/a'):>9}")
|
|
|
|
|
|
def main():
|
|
ap = argparse.ArgumentParser(description="Round-trip analytics over trades.jsonl")
|
|
ap.add_argument("--ledger", default=str(LEDGER), help="path to trades.jsonl")
|
|
ap.add_argument("--since", help="only count trades closed on/after YYYY-MM-DD")
|
|
ap.add_argument("--until", help="only count trades closed on/before YYYY-MM-DD")
|
|
ap.add_argument("--json", action="store_true", help="dump the full bundle as JSON")
|
|
args = ap.parse_args()
|
|
|
|
path = Path(args.ledger)
|
|
if not path.exists():
|
|
raise SystemExit(f"ledger not found: {path}")
|
|
|
|
data = analyze(path, args.since, args.until)
|
|
|
|
if args.json:
|
|
print(json.dumps(data, indent=2, default=str))
|
|
return
|
|
|
|
o = data["overall"]
|
|
span = " ".join(filter(None, [
|
|
f"since {args.since}" if args.since else "",
|
|
f"until {args.until}" if args.until else "",
|
|
])) or "full history"
|
|
print(f"========== round-trip performance ({span}) ==========")
|
|
if o["n"] == 0:
|
|
print(" no closed round trips in range")
|
|
else:
|
|
print(f" closed trips {o['n']} ({o['n_wins']}W / {o['n_losses']}L)")
|
|
print(f" net realized {o['pnl']:+.2f} (commissions {o['commission']:.2f})")
|
|
print(f" win rate {o['win_rate']:.1f}% breakeven needs "
|
|
f"{_fmt(o['breakeven_win_rate'], '.1f', 'n/a')}%")
|
|
print(f" expectancy {o['expectancy']:+.2f} per trade")
|
|
print(f" profit factor {_fmt(o['profit_factor'], '.2f', 'n/a')}")
|
|
print(f" avg win / loss {o['avg_win']:+.2f} / -{o['avg_loss']:.2f}"
|
|
f" (payoff {_fmt(o['payoff_ratio'], '.2f', 'n/a')})")
|
|
print(f" best / worst {o['best']:+.2f} / {o['worst']:+.2f}")
|
|
print(f" max drawdown {data['max_drawdown']:+.2f}")
|
|
if o["avg_hold_minutes"] is not None:
|
|
print(f" avg hold {o['avg_hold_minutes']:.0f} min")
|
|
|
|
_print_table("by strategy", data["by_strategy"])
|
|
_print_table("by symbol", data["by_symbol"])
|
|
if any(k != "(none)" for k in data["by_entry_reason"]):
|
|
_print_table("by entry signal", data["by_entry_reason"])
|
|
if any(k != "(none)" for k in data["by_exit_reason"]):
|
|
_print_table("by exit reason", data["by_exit_reason"])
|
|
|
|
if data["unmatched_sell_qty"]:
|
|
print(f"\n note: {data['unmatched_sell_qty']:g} unit(s) sold with no known "
|
|
f"cost basis (lots predate the ledger) - excluded above")
|
|
|
|
if data["open_lots"]:
|
|
print(f"\n open lots ({len(data['open_lots'])}):")
|
|
for l in data["open_lots"]:
|
|
print(f" {l['strategy_short']:<10}{l['symbol']:<6} x{l['qty']:<6g} "
|
|
f"@ {l['price']:>9.2f} cost {l['cost']:>10.2f}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|