#!/usr/bin/env python3
"""Offline reproduction of frozen Litecoin.watch LTC/EUR order-book observations.
Python 3.10+, standard library only. Run:
python analyze.py.txt --input . --output reproduced
Input directory needs history.json, capture-record.json, and either raw/ or
raw-books.zip. No network calls, order placement, or persistent collectors.
Own analysis/code: MIT. Third-party exchange data remains subject to its source
terms. Quantiles are descriptive sample quantiles, not confidence intervals.
"""
import argparse
from collections import Counter, defaultdict
import csv
from datetime import datetime, timezone
from decimal import Decimal, localcontext
import hashlib
import html
import json
import math
from pathlib import Path
import statistics
import zipfile
VENUES = ('kraken', 'coinbase', 'okx')
NAMES = {'kraken': 'Kraken', 'coinbase': 'Coinbase', 'okx': 'OKX'}
COLORS = {'kraken': '#305ea8', 'coinbase': '#a54723', 'okx': '#317958'}
BUDGETS = (100, 1000, 10000)
PROTOCOL = 'lw-ltceur-depth-2026-09-30-v1'
EXPECTED_HISTORY_SHA = '9a4988a5ee3977e844227392917c17a367db96b3d6782bbdef0df433e2691d32'
FLOAT_BPS_ABS_TOL = Decimal('1e-8')
DECIMAL_REL_TOL = Decimal('1e-22')
def utc(epoch):
return datetime.fromtimestamp(epoch, timezone.utc).strftime('%Y-%m-%dT%H:%M:%SZ')
def dec(value):
value = Decimal(str(value))
if not value.is_finite():
raise ValueError('Non-finite numeric input')
return value
def quantile(values, probability):
"""Linear interpolation at index (n - 1) p; no inferred population CI."""
values = sorted(values)
if not values:
return None
location = (len(values) - 1) * probability
lower = int(math.floor(location))
upper = int(math.ceil(location))
return values[lower] + (values[upper] - values[lower]) * (location - lower)
def descriptives(values):
return dict(n=len(values), minimum=min(values) if values else None,
p05=quantile(values, .05), median=quantile(values, .5),
p95=quantile(values, .95), maximum=max(values) if values else None,
mean=statistics.fmean(values) if values else None)
def normalize(raw, venue):
if venue == 'kraken':
if raw.get('error') or len(raw.get('result', {})) != 1:
raise ValueError('Kraken error or ambiguous pair')
book = next(iter(raw['result'].values()))
elif venue == 'okx':
if str(raw.get('code')) != '0' or len(raw.get('data', [])) != 1:
raise ValueError('OKX error or ambiguous instrument')
book = raw['data'][0]
else:
book = raw
sides = {}
for side in ('bids', 'asks'):
rows = book.get(side)
if not isinstance(rows, list) or not 0 < len(rows) <= 8000:
raise ValueError('Invalid side/depth')
merged = defaultdict(Decimal)
for row in rows:
if not isinstance(row, list) or len(row) < 2:
raise ValueError('Invalid level')
price, size = dec(row[0]), dec(row[1])
if price <= 0 or size <= 0:
raise ValueError('Non-positive price/size')
# Level size is already aggregate size; do not multiply by order count.
merged[price] += size
sides[side] = sorted(merged.items(), reverse=side == 'bids')
if sides['bids'][0][0] >= sides['asks'][0][0]:
raise ValueError('Locked/crossed book')
return sides, book
def replay(levels, target, buy):
remaining, quantity, cash = target, Decimal(0), Decimal(0)
for price, size in levels:
take = min(size, remaining / price if buy else remaining)
quantity += take
cash += take * price
remaining -= take * price if buy else take
if remaining <= Decimal('1e-20'):
break
full = remaining <= max(Decimal('1e-20'), target * Decimal('1e-18'))
return dict(full=full, filled_fraction=(target-remaining)/target,
ltc=quantity, eur=cash, vwap=cash/quantity if quantity else None)
def assert_close(actual, recorded, kind, context, errors):
actual, recorded = dec(actual), dec(recorded)
delta = abs(actual-recorded)
if kind == 'bps':
tolerance = FLOAT_BPS_ABS_TOL
elif kind == 'fraction':
tolerance = Decimal('1e-12')
else:
tolerance = max(Decimal('1e-22'), abs(actual)*DECIMAL_REL_TOL)
if delta > tolerance:
errors.append(f'{context}: delta {delta} exceeds {tolerance}')
return float(delta)
def csv_write(path, rows, fieldnames=None):
fieldnames = fieldnames or list(rows[0])
with path.open('w', encoding='utf-8', newline='') as stream:
writer = csv.DictWriter(stream, fieldnames=fieldnames)
writer.writeheader()
writer.writerows(rows)
def svg_open(title, subtitle, width=1100, height=620):
esc = html.escape
return [f'');path.write_text('\n'.join(parts)+'\n',encoding='utf-8')
def chart_history(path, times, rows):
title='€10,000 buy costs moved within the same short observation period'
subtitle='Visible-book replay cost versus own midpoint; full fills only, exchange fees excluded. UTC.'
parts=svg_open(title,subtitle,height=790)
left,right=100,1045
start=(times[0]//900)*900; end=((times[-1]//900)+1)*900
selected={v:{r['captured_at']:r for r in rows if r['venue']==v and r['side']=='buy' and r['budget_eur']==10000} for v in VENUES}
maximum=max(float(r['mid_cost_bps']) for v in VENUES for r in selected[v].values() if r['included'])
limit=math.ceil(maximum/10)*10
def xx(t): return left+(t-start)/(end-start)*(right-left)
missing=sorted(set(range(times[0]//900,times[-1]//900+1))-set(t//900 for t in times))
for j,venue in enumerate(VENUES):
top=120+j*195; bottom=top+145
for bucket in missing:
parts.append(f'')
for fraction in (0,.5,1):
y=bottom-145*fraction
parts.append(f'')
parts.append(text_svg(left-12,y+4,f'{limit*fraction:g}','tick','end'))
parts.append(text_svg(left,top-13,NAMES[venue]+' — bps','label',fill=COLORS[venue]))
segments=[];current=[];previous=None
for t in times:
row=selected[venue].get(t)
if not row or not row['included'] or (previous is not None and t//900-previous//900>1):
if current:segments.append(current)
current=[]
if row and row['included']:
current.append(f'{xx(t):.3f},{bottom-float(row["mid_cost_bps"])/limit*145:.3f}')
previous=t
if current:segments.append(current)
for segment in segments:
if len(segment)>1:parts.append(f'')
else:
x,y=segment[0].split(',');parts.append(f'')
# Daily UTC boundaries plus clear exact capture endpoints in the caption.
first_midnight=((start//86400)+1)*86400
for t in range(first_midnight,end,86400):
parts.append(text_svg(xx(t),715,datetime.fromtimestamp(t,timezone.utc).strftime('%d %b'),'tick','middle'))
parts.append(text_svg(52,750,'Zero baseline and common scale. Grey marks the unoccupied 15-minute bin; the line breaks across that gap.','small'))
parts.append(text_svg(52,773,f'First capture {utc(times[0])}; last {utc(times[-1])}. No observations between captures.','small'))
parts.append('');path.write_text('\n'.join(parts)+'\n',encoding='utf-8')
def chart_tails(path, stats, n):
title='A typical cost and a costly snapshot are different questions'
subtitle=f'€10,000 replays, {n} matched captures. Median and 95th sample percentile; fees excluded.'
parts=svg_open(title,subtitle,height=650)
left,right,top,bottom=115,940,118,475
maximum=max(stats[v][s]['10000']['p95'] for v in VENUES for s in ('buy','sell'))
limit=math.ceil(maximum/5)*5
for fraction in (0,.2,.4,.6,.8,1):
x=left+(right-left)*fraction
parts.append(f'')
parts.append(text_svg(x,505,f'{limit*fraction:g}','tick','middle'))
for j,venue in enumerate(VENUES):
for k,side in enumerate(('buy','sell')):
y=148+j*117+k*39
median=stats[venue][side]['10000']['median'];p95=stats[venue][side]['10000']['p95']
parts.append(text_svg(left-15,y+5,NAMES[venue]+' '+side,'label','end'))
parts.append(f'')
parts.append(f'')
parts.append(f'')
parts.append(text_svg(left+(right-left)*p95/limit+14,y+5,f'{median:.2f} / {p95:.2f}','small'))
parts.append(text_svg(565,534,'Cost versus midpoint, basis points','label','middle'))
parts.append('')
parts.append(text_svg(132,569,'Median','label'));parts.append(text_svg(310,569,'95th sample percentile','label'))
parts.append(text_svg(52,607,'Percentiles describe retained snapshots, not uncertainty or a future cost ceiling.','small'))
parts.append(text_svg(52,630,'Sell quantity is €10,000 divided by each venue’s best bid; quantities differ across venues.','small'))
parts.append('');path.write_text('\n'.join(parts)+'\n',encoding='utf-8')
def main():
parser=argparse.ArgumentParser(description=__doc__)
parser.add_argument('--input', type=Path, default=Path(__file__).resolve().parent)
parser.add_argument('--output', type=Path, default=Path('reproduced'))
args=parser.parse_args();base=args.input.resolve();out=args.output.resolve();out.mkdir(parents=True,exist_ok=True)
raw_history=(base/'history.json').read_bytes()
digest=hashlib.sha256(raw_history).hexdigest()
if digest != EXPECTED_HISTORY_SHA:
raise ValueError('This release reproduces one frozen input; history hash mismatch')
capture=json.loads((base/'capture-record.json').read_text(encoding='utf-8'))
if capture['sha256']!=digest or capture['bytes']!=len(raw_history):raise ValueError('Capture manifest mismatch')
history=json.loads(raw_history)
if history.get('protocol') != PROTOCOL or history.get('schema') != 1:raise ValueError('Unexpected protocol/schema')
entries=history['entries'];times=[e['captured_at'] for e in entries]
if times != sorted(set(times)):raise ValueError('Capture times not unique and strictly increasing')
archive=zipfile.ZipFile(base/'raw-books.zip') if not (base/'raw').is_dir() else None
archive_names={Path(n).name:n for n in archive.namelist()} if archive else {}
errors=[];rows=[];snapshots=[];raw_records=[];auction=Counter();status=Counter();max_deltas=defaultdict(float);negative_residues=[]
with localcontext() as ctx:
ctx.prec=50
for entry in entries:
t=entry['captured_at'];venue_list=entry['venues']
if sorted(v['venue'] for v in venue_list)!=sorted(VENUES):raise ValueError('Missing/duplicate venue record')
for record in venue_list:
venue=record['venue'];status[(venue,record['status'])]+=1
if record['status']!='ok':
for b in BUDGETS:
for s in ('buy','sell'):
rows.append(dict(captured_at=t,captured_utc=utc(t),venue=venue,side=s,budget_eur=b,status=record['status'],auction_mode='',full='',included=False,target_ltc='',ltc='',eur='',vwap='',mid_cost_bps='',impact_bps='',spread_bps='',mid_eur='',best_bid_eur='',best_ask_eur=''))
continue
filename=record['raw_file']
if Path(filename).name!=filename or filename!=f'{t}-{venue}.json':raise ValueError('Unexpected raw filename')
raw=(base/'raw'/filename).read_bytes() if archive is None else archive.read(archive_names[filename])
if hashlib.sha256(raw).hexdigest()!=record['sha256']:raise ValueError('Raw hash mismatch: '+filename)
data=json.loads(raw);sides,book=normalize(data,venue)
auction_flag=data.get('auction_mode') if venue=='coinbase' else None
if venue=='coinbase':auction[str(auction_flag)]+=1
valid_continuous=(venue!='coinbase' or auction_flag is False)
# A missing or true auction flag is not silently treated as continuous trading.
bid,ask=sides['bids'][0][0],sides['asks'][0][0];mid=(bid+ask)/2
spread=(ask-bid)/mid*10000;metrics=record['metrics']
for key,value in [('bid',bid),('ask',ask),('mid',mid),('spread_bps',spread)]:
kind='bps' if key=='spread_bps' else 'decimal'
delta=assert_close(value,metrics[key],kind,f'{filename} {key}',errors)
max_deltas[kind]=max(max_deltas[kind],delta)
if metrics['ask_levels']!=len(sides['asks']) or metrics['bid_levels']!=len(sides['bids']):errors.append(filename+' level counts mismatch')
for side in ('buy','sell'):
depth=sum((p*q for p,q in sides['asks'] if p<=ask*Decimal('1.01')),Decimal(0)) if side=='buy' else sum((p*q for p,q in sides['bids'] if p>=bid*Decimal('.99')),Decimal(0))
delta=assert_close(depth,metrics['depth_1pct_eur'][side],'decimal',f'{filename} depth {side}',errors)
max_deltas['decimal']=max(max_deltas['decimal'],delta)
orders=metrics['orders']
if len(orders)!=6 or {(o['side'],o['budget_eur']) for o in orders}!={(s,b) for s in ('buy','sell') for b in BUDGETS}:raise ValueError('Unexpected order replay set')
snap=dict(captured_at=t,captured_utc=utc(t),venue=venue,mid_eur=str(mid),best_bid_eur=str(bid),best_ask_eur=str(ask),spread_bps=float(spread),auction_mode=auction_flag,continuous=valid_continuous,request_at=record['request_at'],response_at=record['response_at'],elapsed_seconds=record['response_at']-record['request_at'],levels_buy=len(sides['asks']),levels_sell=len(sides['bids']))
snapshots.append(snap)
for order in orders:
side,budget=order['side'],order['budget_eur'];buy=side=='buy'
target=Decimal(budget) if buy else Decimal(budget)/bid
result=replay(sides['asks' if buy else 'bids'],target,buy)
if result['full']!=order['full']:errors.append(filename+' full fill mismatch')
for key in ('ltc','eur','vwap','filled_fraction'):
if result[key] is not None:
kind='fraction' if key=='filled_fraction' else 'decimal'
delta=assert_close(result[key],order[key],kind,f'{filename} {side}/{budget} {key}',errors)
max_deltas[kind]=max(max_deltas[kind],delta)
if not buy:assert_close(target,order['target_ltc'],'decimal',filename+' target_ltc',errors)
if buy and order['target_ltc'] is not None:errors.append(filename+' unexpected buy LTC target')
if result['full']:
vwap=result['vwap']
impact=max(Decimal(0),(vwap/ask-1 if buy else 1-vwap/bid)*10000)
cost=max(Decimal(0),(vwap/mid-1 if buy else 1-vwap/mid)*10000)
for key,value in [('impact_bps',impact),('mid_cost_bps',cost)]:
if dec(order[key]) < 0:
if dec(order[key]) < Decimal('-1e-12'):errors.append(filename+' materially negative '+key)
negative_residues.append(dict(filename=filename,side=side,budget_eur=budget,field=key,recorded_value=order[key],canonical_value=0))
delta=assert_close(value,order[key],'bps',f'{filename} {side}/{budget} {key}',errors)
max_deltas['bps']=max(max_deltas['bps'],delta)
else:
impact=cost=None
if order['impact_bps'] is not None or order['mid_cost_bps'] is not None:errors.append(filename+' partial replay has published cost')
rows.append(dict(captured_at=t,captured_utc=utc(t),venue=venue,side=side,budget_eur=budget,status='ok',auction_mode=auction_flag if venue=='coinbase' else '',full=result['full'],included=result['full'] and valid_continuous,target_ltc='' if buy else str(target),ltc=str(result['ltc']),eur=str(result['eur']),vwap=str(result['vwap']),mid_cost_bps='' if cost is None else float(cost),impact_bps='' if impact is None else float(impact),spread_bps=float(spread),mid_eur=str(mid),best_bid_eur=str(bid),best_ask_eur=str(ask)))
raw_records.append(dict(filename=filename,sha256=record['sha256'],bytes=len(raw),venue=venue,url=record['url'],capture_utc=utc(t),request_at=record['request_at'],response_at=record['response_at'],exchange_at=record['exchange_at'],http_headers=record['headers']))
if archive:archive.close()
if errors:raise ValueError('Reproduction mismatch:\n'+'\n'.join(errors[:30]))
# Matched comparisons use identical capture-round labels, not atomic market timestamps.
matched={}
indexed={(r['captured_at'],r['venue'],r['side'],r['budget_eur']):r for r in rows}
for side in ('buy','sell'):
for b in BUDGETS:
matched[(side,b)]=[t for t in times if all(indexed[(t,v,side,b)]['included'] for v in VENUES)]
stats={v:{s:{str(b):descriptives([indexed[(t,v,s,b)]['mid_cost_bps'] for t in matched[(s,b)]]) for b in BUDGETS} for s in ('buy','sell')} for v in VENUES}
coverage=[]
bins=set(t//900 for t in times);expected=times[-1]//900-times[0]//900+1
missing=sorted(set(range(times[0]//900,times[-1]//900+1))-bins)
for v in VENUES:
these=[r for r in rows if r['venue']==v]
coverage.append(dict(venue=v,capture_rounds=len(times),validated_books=status[(v,'ok')],unavailable_books=sum(n for (vv,ss),n in status.items() if vv==v and ss!='ok'),replays=len(these),full_replays=sum(r['full'] is True for r in these),partial_replays=sum(r['full'] is False for r in these),continuous_full_replays=sum(r['included'] for r in these)))
# Lowest own-midpoint friction, not absolute price, pool quality, or venue recommendation.
lowest={}
for s in ('buy','sell'):
for b in BUDGETS:
counts=Counter();ties=0
for t in matched[(s,b)]:
values={v:indexed[(t,v,s,b)]['mid_cost_bps'] for v in VENUES};m=min(values.values())
winners=[v for v,value in values.items() if abs(value-m)<=1e-10]
if len(winners)>1:ties+=1
else:counts[winners[0]]+=1
lowest[f'{s}_{b}']=dict(matched_n=len(matched[(s,b)]),strict_lowest_counts={v:counts[v] for v in VENUES},tied_rounds=ties)
round_skews=[]
for t in times:
group=[s for s in snapshots if s['captured_at']==t]
if len(group)==3:round_skews.append(max(s['response_at'] for s in group)-min(s['response_at'] for s in group))
spread={v:descriptives([s['spread_bps'] for s in snapshots if s['venue']==v and s['continuous']]) for v in VENUES}
extremes={}
for v in VENUES:
candidates=[r for r in rows if r['venue']==v and r['side']=='buy' and r['budget_eur']==10000 and r['included']]
extremes[v]=dict(minimum=min(candidates,key=lambda r:r['mid_cost_bps']),maximum=max(candidates,key=lambda r:r['mid_cost_bps']))
results=dict(release='liquidity-history-20261005-v1',protocol=PROTOCOL,history_sha256=digest,capture_record=capture,
study=dict(first_utc=utc(times[0]),last_utc=utc(times[-1]),elapsed_seconds=times[-1]-times[0],capture_rounds=len(times),planned_interval_seconds=900,quarter_hour_bins_intersecting_capture_period=expected,occupied_bins=len(bins),occupied_bin_fraction=len(bins)/expected,missing_bin_starts_utc=[utc(b*900) for b in missing],bin_definition='Floor capture-start Unix time by 900 seconds; include both endpoint bins, which may be partial. A bin is occupied if any retained capture starts within it. Sampling coverage, not venue or collector uptime.',off_quarter_hour_rounds=[dict(utc=utc(t),seconds_after_boundary=t%900) for t in times if t%900>60]),
verification=dict(raw_books_checked=len(raw_records),raw_bytes_checked=sum(r['bytes'] for r in raw_records),all_raw_hashes_match=True,all_recorded_replays_reproduced=True,decimal_precision=50,decimal_relative_tolerance=str(DECIMAL_REL_TOL),bps_absolute_tolerance=str(FLOAT_BPS_ABS_TOL),maximum_observed_absolute_deltas=dict(max_deltas),historical_negative_rounding_residues=negative_residues,negative_residue_policy='Frozen input unchanged. Reject negatives below -1e-12 bps; bounded residue is zero in the independently recomputed nonnegative cost. No replay is excluded solely for this residue.',coinbase_auction_mode_counts=dict(auction),auction_policy='Exclude true or missing Coinbase auction_mode from continuous-trading comparisons; this cohort has false in every response.'),
coverage=coverage,matched_round_counts={f'{s}_{b}':len(ts) for (s,b),ts in matched.items()},cost_stats_bps=stats,spread_stats_bps=spread,lowest_own_midpoint_friction=lowest,response_completion_skew_seconds=descriptives(round_skews),response_elapsed_seconds={v:descriptives([s['elapsed_seconds'] for s in snapshots if s['venue']==v]) for v in VENUES},buy_10000_extremes=extremes,last_capture=[r for r in rows if r['captured_at']==times[-1]],
method=dict(quantile='Sort sample, linearly interpolate at (n - 1) * p; descriptive percentiles, not confidence intervals.',buy='Walk asks for an exact EUR budget, with a fractional final price level. VWAP = EUR spent / LTC quantity.',sell='Target LTC = reference EUR budget / that venue’s best bid. Walk bids for that quantity, with a fractional final price level. Cross-venue sell targets differ.',mid='(best bid + best ask) / 2',cost_bps='Buy: max(0,(VWAP/mid - 1)*10000). Sell: max(0,(1 - VWAP/mid)*10000). Publish only for full replays.',impact_bps='Buy: max(0,(VWAP/best ask - 1)*10000). Sell: max(0,(1 - VWAP/best bid)*10000).',full_tolerance='remaining <= max(1e-20, target * 1e-18), matching the collector protocol.',limitations=['Retained visible REST books, not executed orders.','Requested depth: Kraken 100 price levels per side; OKX 100; Coinbase aggregated level 2.','Size already aggregates each price level; the order-count field is not a multiplier.','No exchange fees, quantity rounding, minimum orders, funding/withdrawal/FX costs, latency, queue competition, hidden liquidity, or replenishment.','Capture-round labels match observations but requests and venue clocks are not atomic.','Lower relative midpoint cost need not mean a lower absolute purchase price.','No causal claim about volatility, exchange reliability or future execution; this period is short and sampled.']))
(out/'results.json').write_text(json.dumps(results,indent=2,ensure_ascii=False)+'\n',encoding='utf-8')
(out/'raw-manifest.json').write_text(json.dumps(raw_records,indent=2)+'\n',encoding='utf-8')
csv_write(out/'replays.csv',rows)
csv_write(out/'snapshots.csv',snapshots)
csv_write(out/'coverage.csv',coverage)
summary=[]
for v in VENUES:
for s in ('buy','sell'):
for b in BUDGETS:summary.append(dict(venue=v,side=s,budget_eur=b,**stats[v][s][str(b)]))
csv_write(out/'summary.csv',summary)
chart_medians(out/'buy-cost-by-size.svg',stats,len(matched[('buy',10000)]))
chart_history(out/'buy-cost-history.svg',times,rows)
chart_tails(out/'buy-sell-percentiles.svg',stats,len(matched[('buy',10000)]))
print(json.dumps(dict(verified_books=len(raw_records),replays=len(rows),capture_rounds=len(times),output=str(out),coverage=coverage,median_buy_10000={v:stats[v]['buy']['10000']['median'] for v in VENUES},p95_buy_10000={v:stats[v]['buy']['10000']['p95'] for v in VENUES})))
if __name__ == '__main__':
main()