From 46c3cb7c4ab974b6a6e6b574e37395856e23948d Mon Sep 17 00:00:00 2001 From: brianmmaina Date: Sun, 16 Aug 2026 08:34:40 -0700 Subject: [PATCH] research: multi-seed comparison with error bars MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 30 seeds, and the result is sharper and more negative than one run suggested. Nothing is reproduced. The model's own variance is small enough that every real value sits far outside it. Corrects the previous writeup, which called the structural properties 'broadly reproduced' from a single seed on the strength of an 88% cancel ratio against a real 96%. With error bars that gap is 17 standard deviations: the model lands on 87.8% +/- 0.5% every time, consistently in the wrong place. Precise is not the same as right, and one draw could not tell the difference. Reports relative error alongside z because either alone misleads in opposite directions. A precise model can be many sigma out while a few percent wrong; a noisy one can be an order of magnitude wrong and one sigma away. Trade-level kurtosis is the second case — 175 against 11.8, which is 1389% wrong and |z| = 1.4, and a z-only verdict would have called it reproduced. The sign-autocorrelation result stands and sharpens: 35% low at lag 1, 96% low by lag 5, nothing beyond. The lag-1 value is mechanical, not memory. compare.py imports stylized_facts.analyse rather than reimplementing it, so a difference in the analysis cannot masquerade as a difference in the market. --- analysis/compare.py | 198 ++++++++++++++++++++++++++++++++++++++++++ docs/ZI_COMPARISON.md | 151 +++++++++++++++++++------------- 2 files changed, 290 insertions(+), 59 deletions(-) create mode 100644 analysis/compare.py diff --git a/analysis/compare.py b/analysis/compare.py new file mode 100644 index 0000000..e807112 --- /dev/null +++ b/analysis/compare.py @@ -0,0 +1,198 @@ +#!/usr/bin/env python3 +"""Compare synthetic order flow against the real thing, across many seeds. + +Runs tools/zi_sim for N seeds, measures each output with the SAME analysis code +that measures the real LOBSTER data — `stylized_facts.analyse`, imported rather +than reimplemented — and reports where the model does and does not reproduce the +market. + +WHAT THE ERROR BARS DO AND DO NOT COVER + +The synthetic side gets a real distribution: N independent seeds, reported as +mean ± standard deviation across them. That says how much of a difference is +just sampling noise in the model. + +The real side is ONE observation. One session, one symbol, one day. It has no +error bar here and cannot have one from this data, so the comparison is +necessarily one-sided: it asks whether the real value falls inside the spread of +synthetic outcomes, not whether the two distributions differ. A real value far +outside the synthetic range is meaningful. A real value inside it means the +model is *not excluded*, which is weaker than the model being right. + +Reported per statistic: + real the measured value from LOBSTER + zi mean ± sd across seeds + rel relative error, |real − zi_mean| / |real| + z (real − zi_mean) / zi_sd, when sd > 0 + +Both are needed. A precise model can be many sigma from the truth while being +only a few percent wrong (large z, small rel) — it captures the phenomenon and +misses the value. A noisy model can be an order of magnitude wrong and only one +sigma away (small z, large rel) — it does not capture anything and merely +cannot be excluded. Reading either column alone gets one of those backwards. + +|z| under about 2 means the real value sits within the model's ordinary +variation. Large |z| means the model does not produce that behaviour. + +Usage: + python3 analysis/compare.py --symbol AMZN --seeds 30 +""" + +import argparse +import math +import os +import subprocess +import sys +import tempfile + +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +import stylized_facts as sf # noqa: E402 + +ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) + + +def mean(xs): + return sum(xs) / len(xs) if xs else float("nan") + + +def stdev(xs): + if len(xs) < 2: + return 0.0 + m = mean(xs) + return math.sqrt(sum((x - m) ** 2 for x in xs) / (len(xs) - 1)) + + +def run_seed(config, seed, duration, workdir): + """One simulated session. Returns the analysis dict for its message log.""" + path = os.path.join(workdir, f"zi_{seed}_message_10.csv") + r = subprocess.run( + [os.path.join(ROOT, "build", "zi_sim"), + "--config", config, "--seed", str(seed), + "--duration", str(duration), "--messages", path], + capture_output=True, text=True) + if r.returncode != 0: + print(f" seed {seed} failed: {r.stderr.strip()}", file=sys.stderr) + return None + return sf.analyse(f"ZI{seed}", path) + + +# (label, how to pull the number out of an analysis dict) +STATS = [ + ("cancels / new", lambda r: r["cancel_ratio"] * 100), + ("executions / new", lambda r: r["exec_ratio"] * 100), + ("effective spread (ticks)", lambda r: r["eff_spread"][0]), + ("trade-level kurtosis", lambda r: r["ret_kurtosis"]), + ("sign ACF lag 1", lambda r: dict(r["sign_acf"])[1]), + ("sign ACF lag 2", lambda r: dict(r["sign_acf"])[2]), + ("sign ACF lag 5", lambda r: dict(r["sign_acf"])[5]), + ("sign ACF lag 10", lambda r: dict(r["sign_acf"])[10]), + ("sign ACF lag 50", lambda r: dict(r["sign_acf"])[50]), + ("|ret| ACF 10s lag 1", lambda r: dict(r["absret_acf_fine"])[1]), + ("|ret| ACF 10s lag 2", lambda r: dict(r["absret_acf_fine"])[2]), + ("|ret| ACF 10s lag 5", lambda r: dict(r["absret_acf_fine"])[5]), +] + + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("--symbol", default="AMZN", help="real symbol to compare against") + ap.add_argument("--config", default="config/zi_amzn.conf") + ap.add_argument("--seeds", type=int, default=30) + ap.add_argument("--duration", type=float, default=23400.0) + ap.add_argument("--markdown", action="store_true") + args = ap.parse_args() + + msg, _ = None, None + for d in sorted(os.listdir(sf.DATA_ROOT)): + if args.symbol in d: + import glob + hits = glob.glob(os.path.join(sf.DATA_ROOT, d, "*_message_*.csv")) + if hits: + msg = hits[0] + break + if not msg: + print(f"no LOBSTER data for {args.symbol}", file=sys.stderr) + return 1 + + print(f"measuring real {args.symbol} ...", file=sys.stderr) + real = sf.analyse(args.symbol, msg) + + print(f"running {args.seeds} synthetic sessions of {args.duration:.0f}s ...", file=sys.stderr) + runs = [] + with tempfile.TemporaryDirectory() as wd: + for s in range(1, args.seeds + 1): + r = run_seed(args.config, s, args.duration, wd) + if r: + runs.append(r) + if s % 10 == 0: + print(f" {s}/{args.seeds}", file=sys.stderr) + if len(runs) < 2: + print("not enough successful runs", file=sys.stderr) + return 1 + + hdr = (f"{'statistic':<26} {'real':>10} {'zi mean':>10} {'zi sd':>8} " + f"{'rel':>7} {'z':>7} verdict") + if args.markdown: + print(f"| Statistic | Real {args.symbol} | ZI mean | ZI sd | rel | z | Verdict |") + print("|---|---|---|---|---|---|---|") + else: + print() + print(hdr) + print("-" * len(hdr)) + + for label, get in STATS: + try: + rv = get(real) + vals = [get(r) for r in runs] + except (KeyError, IndexError, TypeError): + continue + vals = [v for v in vals if isinstance(v, (int, float)) and not math.isnan(v)] + if not vals: + continue + m, sd = mean(vals), stdev(vals) + z = (rv - m) / sd if sd > 0 else float("inf") + + # z alone is not enough, in two directions. + # + # A model with huge seed variance produces a small z for a value that is + # wildly wrong — trade-level kurtosis came out at 175 against a real 11.8 + # with an sd of 117, which is |z| = 1.4 and would read as "reproduced". + # That is the model being too noisy to exclude, not the model being right. + # + # And two numbers that are both approximately zero agree trivially. An + # autocorrelation of 0.01 matching one of 0.03 says nothing about whether + # the mechanism is there. + rel = abs(rv - m) / max(abs(rv), 1e-9) + both_tiny = abs(rv) < 0.05 and abs(m) < 0.05 + + if both_tiny: + verdict = "both ~0" + elif abs(z) >= 6: + verdict = "NOT reproduced" + elif rel > 0.5: + verdict = "inconclusive (model too variable)" + elif abs(z) >= 2: + verdict = "off" + else: + verdict = "reproduced" + if args.markdown: + print(f"| {label} | {rv:.3f} | {m:.3f} | {sd:.3f} | {rel*100:.0f}% " + f"| {z:+.1f} | {verdict} |") + else: + print(f"{label:<26} {rv:>10.3f} {m:>10.3f} {sd:>8.3f} " + f"{rel*100:>6.0f}% {z:>+7.1f} {verdict}") + + if not args.markdown: + print() + print(f"{len(runs)} seeds. The real column is a SINGLE session and has no error bar:") + print("a real value inside the synthetic spread means the model is not excluded,") + print("which is weaker than the model being right.") + print() + print("'inconclusive' means the seed-to-seed variance is large enough to swallow") + print("a big relative error — the model is too noisy to exclude, not correct.") + print("'both ~0' means the two values agree only because neither is far from zero.") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/docs/ZI_COMPARISON.md b/docs/ZI_COMPARISON.md index e2423e8..a8e0c4d 100644 --- a/docs/ZI_COMPARISON.md +++ b/docs/ZI_COMPARISON.md @@ -2,64 +2,93 @@ Synthetic order flow, calibrated to AMZN (`docs/CALIBRATION.md`) and matched by the same `MatchingEngine` the gateway uses, compared against the real LOBSTER -stream. Both are analysed by `analysis/stylized_facts.py` — the same code, -unchanged, because the simulator writes its message log in LOBSTER's schema. - -Single seed, one 6.5-hour session. Error bars across seeds are still to come. - -## Structural properties: broadly reproduced - -| | Real AMZN | ZI | -|---|---|---| -| Cancels / new orders | 95.8% | **88.2%** | -| Executions / new orders | 6.8% | 14.4% | -| Median effective spread | 300 ticks | 691 | -| Messages | 269,748 | 269,893 | - -Order flow that is cancelled rather than executed, at roughly nine in ten, falls -straight out of the calibrated arrival and cancellation rates. Nothing in the -model intends it. - -## Dynamic properties: reproduced at lag 1, absent thereafter - -**Trade-sign autocorrelation** - -| Lag | 1 | 2 | 5 | 10 | 20 | 50 | 100 | -|---|---|---|---|---|---|---|---| -| Real | 0.720 | 0.575 | 0.340 | 0.205 | 0.132 | 0.085 | 0.009 | -| ZI | 0.474 | 0.183 | 0.016 | 0.013 | −0.003 | −0.007 | −0.005 | - -**Volatility clustering, |10-second returns|** - -| Lag | 1 | 2 | 3 | 5 | 10 | +stream. + +Both sides are measured by the same code — `analysis/compare.py` imports +`stylized_facts.analyse` rather than reimplementing it — because a difference in +the analysis would be indistinguishable from a difference in the market. + +**30 seeds**, one 6.5-hour session each. + +```bash +python3 analysis/compare.py --symbol AMZN --seeds 30 +``` + +## Results + +| Statistic | Real AMZN | ZI mean | ZI sd | rel | z | Verdict | +|---|---|---|---|---|---|---| +| cancels / new | 95.772 | 87.847 | 0.457 | 8% | +17.3 | NOT reproduced | +| executions / new | 6.801 | 14.321 | 0.149 | 111% | -50.3 | NOT reproduced | +| effective spread (ticks) | 300.000 | 693.433 | 8.928 | 131% | -44.1 | NOT reproduced | +| trade-level kurtosis | 11.756 | 174.993 | 117.047 | 1389% | -1.4 | inconclusive (model too variable) | +| sign ACF lag 1 | 0.720 | 0.468 | 0.006 | 35% | +42.5 | NOT reproduced | +| sign ACF lag 2 | 0.575 | 0.174 | 0.009 | 70% | +46.6 | NOT reproduced | +| sign ACF lag 5 | 0.340 | 0.015 | 0.008 | 96% | +40.7 | NOT reproduced | +| sign ACF lag 10 | 0.205 | 0.007 | 0.009 | 97% | +22.9 | NOT reproduced | +| sign ACF lag 50 | 0.085 | -0.001 | 0.009 | 101% | +9.4 | NOT reproduced | +| |ret| ACF 10s lag 1 | 0.155 | 0.376 | 0.043 | 143% | -5.1 | inconclusive (model too variable) | +| |ret| ACF 10s lag 2 | 0.087 | 0.015 | 0.029 | 82% | +2.5 | inconclusive (model too variable) | +| |ret| ACF 10s lag 5 | 0.036 | 0.010 | 0.031 | 74% | +0.9 | both ~0 | + +`rel` is relative error; `z` is `(real − zi_mean) / zi_sd`. + +**Both columns are necessary, and either alone misleads.** A precise model can be +many sigma from the truth while only a few percent wrong — it captures the +phenomenon and misses the value. A noisy model can be an order of magnitude +wrong and one sigma away — it captures nothing and merely cannot be excluded. +Trade-level kurtosis is the second case: 175 against a real 11.8, which is 1,389% +wrong and |z| = 1.4. + +## What this actually says + +**Nothing here is reproduced.** With 30 seeds the model's own variance is small +enough that every real value sits far outside it. + +That is a sharper and more negative result than a single seed suggested. The +first version of this document, written from one run, called the structural +properties "broadly reproduced" on the strength of an 88% cancel ratio against a +real 96%. With error bars that gap is **17 standard deviations**. The model is +internally consistent — seed to seed it lands on 87.8% ± 0.5% — and consistently +in the wrong place. Being precise is not the same as being right, and one draw +could not tell the difference. + +**The cancel ratio is the model's best showing anyway.** 8% relative error, and +qualitatively it gets the fact that matters: a book made overwhelmingly of orders +that will be withdrawn. That falls out of calibrated arrival and cancellation +rates with nothing intending it. + +**Trade-sign autocorrelation is where the model fails completely, and it fails in +a specific shape.** + +| Lag | 1 | 2 | 5 | 10 | 50 | |---|---|---|---|---|---| -| Real | 0.155 | 0.087 | 0.098 | 0.036 | 0.078 | -| ZI | 0.378 | 0.014 | 0.010 | −0.010 | 0.002 | - -This is the result. **ZI produces correlation at lag 1 and none beyond it.** +| Real | 0.720 | 0.575 | 0.340 | 0.205 | 0.085 | +| ZI mean | 0.468 | 0.174 | 0.015 | 0.007 | −0.001 | +| relative error | 35% | 70% | **96%** | **97%** | **101%** | -The lag-1 value is not evidence of memory — it is mechanical. One market order -walks several price levels and prints several executions, all with the same -sign, within the same instant. Any model with multi-level fills produces that. -It is gone by lag 5. +At lag 1 the model is only 35% low. By lag 5 it is 96% low, and past that it has +nothing at all. The lag-1 value is not memory — it is mechanical, one market +order walking several price levels and printing several same-signed executions +in the same instant. Any model with multi-level fills produces it. -Real order flow decays slowly over a hundred trades. That persistence is what -strategic behaviour looks like from the outside — order splitting, participants -reacting to each other — and deleting the strategy deletes it entirely. The -distinction is not "does correlation exist" but "how far does it reach", and -only the second one separates the two. +Real order flow decays slowly over a hundred trades. **The question is not +whether correlation exists but how far it reaches**, and only that separates the +two. Deleting strategic behaviour deletes the reach entirely while leaving the +mechanical artifact intact — which is exactly the distinction the experiment was +built to draw. -## A failure worth recording +## The failure that shaped the model -The first version placed limit orders relative to the **instantaneous** +The first generator placed limit orders relative to the **instantaneous** half-spread, which destroyed the market: | | Real | Adaptive scale | Fixed scale | |---|---|---|---| -| Cancels / new | 95.8% | 8.5% | 88.2% | -| Executions / new | 6.8% | 96.7% | 14.4% | -| Effective spread | 300 | **1 tick** | 691 | -| Trade-level kurtosis | 11.8 | **9,833** | 155 | +| Cancels / new | 95.8% | 8.5% | 87.8% | +| Executions / new | 6.8% | 96.7% | 14.3% | +| Effective spread | 300 ticks | **1 tick** | 693 | +| Trade-level kurtosis | 11.8 | **9,833** | 175 | Orders placed inside the spread narrow it; a narrower spread makes the next order's offset smaller in ticks; within seconds the book collapses to one tick @@ -70,16 +99,20 @@ Using the calibrated median spread as the placement scale breaks the loop: it is exogenous and does not move with the book. The instantaneous spread still sets the reference price; it no longer sets the scale. -The broken behaviour is preserved behind `--adaptive-scale` so it stays -reproducible rather than becoming a story about something that used to happen. +Preserved behind `--adaptive-scale` so the failure stays reproducible. -## Known gaps +## Limits of this comparison -- **Kurtosis is 155 against a real 11.8.** Returns are far too heavy-tailed. - Likely the same mechanism as the lag-1 sign correlation: market orders walk - too deep because the synthetic book is thinner than the real one. -- **Executions per order are 2× too high** (14.4% vs 6.8%), for the same reason. -- **Single seed.** Every number here is one draw. Multi-seed runs with error - bars are the next step; nothing above should be treated as settled. +- **The real side is a single session** and has no error bar. A real value + outside the synthetic spread is meaningful; one inside would mean the model is + *not excluded*, which is weaker than the model being right. +- **One symbol, one day, 2012.** Nothing here generalises to markets. +- **The synthetic book is thinner than the real one**, so market orders walk too + deep. That is the likely common cause of the excess kurtosis, the excess + execution rate, and the inflated lag-1 correlation. - **Hidden liquidity is not modelled** — 21% of real executions. -- **The cancel rate is an overestimate by construction** (`docs/CALIBRATION.md`). +- **The calibrated cancel rate is an overestimate by construction** + (`docs/CALIBRATION.md`). +- **Not a faithful reimplementation of any published model.** It is a + zero-intelligence model in the Farmer/Patelli/Zovko spirit, calibrated to this + data, not a reproduction of their specification.