#!/usr/bin/env python3 """Offline reproduction of frozen Litecoin.watch LTC/EUR order-book observations. Python 3.10+, standard library only. Run: python analyze.py.txt --input . --output reproduced Input directory needs history.json, capture-record.json, and either raw/ or raw-books.zip. No network calls, order placement, or persistent collectors. Own analysis/code: MIT. Third-party exchange data remains subject to its source terms. Quantiles are descriptive sample quantiles, not confidence intervals. """ import argparse from collections import Counter, defaultdict import csv from datetime import datetime, timezone from decimal import Decimal, localcontext import hashlib import html import json import math from pathlib import Path import statistics import zipfile VENUES = ('kraken', 'coinbase', 'okx') NAMES = {'kraken': 'Kraken', 'coinbase': 'Coinbase', 'okx': 'OKX'} COLORS = {'kraken': '#305ea8', 'coinbase': '#a54723', 'okx': '#317958'} BUDGETS = (100, 1000, 10000) PROTOCOL = 'lw-ltceur-depth-2026-09-30-v1' EXPECTED_HISTORY_SHA = '9a4988a5ee3977e844227392917c17a367db96b3d6782bbdef0df433e2691d32' FLOAT_BPS_ABS_TOL = Decimal('1e-8') DECIMAL_REL_TOL = Decimal('1e-22') def utc(epoch): return datetime.fromtimestamp(epoch, timezone.utc).strftime('%Y-%m-%dT%H:%M:%SZ') def dec(value): value = Decimal(str(value)) if not value.is_finite(): raise ValueError('Non-finite numeric input') return value def quantile(values, probability): """Linear interpolation at index (n - 1) p; no inferred population CI.""" values = sorted(values) if not values: return None location = (len(values) - 1) * probability lower = int(math.floor(location)) upper = int(math.ceil(location)) return values[lower] + (values[upper] - values[lower]) * (location - lower) def descriptives(values): return dict(n=len(values), minimum=min(values) if values else None, p05=quantile(values, .05), median=quantile(values, .5), p95=quantile(values, .95), maximum=max(values) if values else None, mean=statistics.fmean(values) if values else None) def normalize(raw, venue): if venue == 'kraken': if raw.get('error') or len(raw.get('result', {})) != 1: raise ValueError('Kraken error or ambiguous pair') book = next(iter(raw['result'].values())) elif venue == 'okx': if str(raw.get('code')) != '0' or len(raw.get('data', [])) != 1: raise ValueError('OKX error or ambiguous instrument') book = raw['data'][0] else: book = raw sides = {} for side in ('bids', 'asks'): rows = book.get(side) if not isinstance(rows, list) or not 0 < len(rows) <= 8000: raise ValueError('Invalid side/depth') merged = defaultdict(Decimal) for row in rows: if not isinstance(row, list) or len(row) < 2: raise ValueError('Invalid level') price, size = dec(row[0]), dec(row[1]) if price <= 0 or size <= 0: raise ValueError('Non-positive price/size') # Level size is already aggregate size; do not multiply by order count. merged[price] += size sides[side] = sorted(merged.items(), reverse=side == 'bids') if sides['bids'][0][0] >= sides['asks'][0][0]: raise ValueError('Locked/crossed book') return sides, book def replay(levels, target, buy): remaining, quantity, cash = target, Decimal(0), Decimal(0) for price, size in levels: take = min(size, remaining / price if buy else remaining) quantity += take cash += take * price remaining -= take * price if buy else take if remaining <= Decimal('1e-20'): break full = remaining <= max(Decimal('1e-20'), target * Decimal('1e-18')) return dict(full=full, filled_fraction=(target-remaining)/target, ltc=quantity, eur=cash, vwap=cash/quantity if quantity else None) def assert_close(actual, recorded, kind, context, errors): actual, recorded = dec(actual), dec(recorded) delta = abs(actual-recorded) if kind == 'bps': tolerance = FLOAT_BPS_ABS_TOL elif kind == 'fraction': tolerance = Decimal('1e-12') else: tolerance = max(Decimal('1e-22'), abs(actual)*DECIMAL_REL_TOL) if delta > tolerance: errors.append(f'{context}: delta {delta} exceeds {tolerance}') return float(delta) def csv_write(path, rows, fieldnames=None): fieldnames = fieldnames or list(rows[0]) with path.open('w', encoding='utf-8', newline='') as stream: writer = csv.DictWriter(stream, fieldnames=fieldnames) writer.writeheader() writer.writerows(rows) def svg_open(title, subtitle, width=1100, height=620): esc = html.escape return [f'', f'{esc(title)}{esc(subtitle)}', f'', '', f'{esc(title)}', f'{esc(subtitle)}'] def text_svg(x,y,text,cls='label',anchor='start',fill=None): extra = f' style="fill:{fill}"' if fill else '' return f'{html.escape(str(text))}' def chart_medians(path, stats, n): title = 'Buying more LTC changes the visible-book cost' subtitle = f'Median cost versus each venue’s midpoint; {n} matched captures per size. Fees excluded.' parts=svg_open(title,subtitle,height=620) left,right,top,bottom=95,1035,118,475 maximum=max(stats[v]['buy'][str(b)]['median'] for v in VENUES for b in BUDGETS) limit=math.ceil(maximum/5)*5 or 5 for i in range(6): value=limit*i/5; y=bottom-(bottom-top)*i/5 parts.append(f'') parts.append(text_svg(left-12,y+4,f'{value:g}','tick','end')) parts.append(text_svg(52,105,'Basis points (bps)','small')) groups=(250,565,880) for budget,x in zip(BUDGETS,groups): for j,venue in enumerate(VENUES): value=stats[venue]['buy'][str(budget)]['median']; h=(bottom-top)*value/limit bx=x-100+j*70 parts.append(f'') parts.append(text_svg(bx+28,bottom-h-9,f'{value:.2f}','label','middle')) parts.append(text_svg(x-2,505,f'€{budget:,} buy','label','middle')) for j,venue in enumerate(VENUES): x=290+j*210 parts.append(f'') parts.append(text_svg(x+27,554,NAMES[venue])) parts.append(text_svg(52,592,'1 bp = 0.01%. Different venue midpoints mean lower friction does not guarantee a lower purchase price.','small')) parts.append('');path.write_text('\n'.join(parts)+'\n',encoding='utf-8') def chart_history(path, times, rows): title='€10,000 buy costs moved within the same short observation period' subtitle='Visible-book replay cost versus own midpoint; full fills only, exchange fees excluded. UTC.' parts=svg_open(title,subtitle,height=790) left,right=100,1045 start=(times[0]//900)*900; end=((times[-1]//900)+1)*900 selected={v:{r['captured_at']:r for r in rows if r['venue']==v and r['side']=='buy' and r['budget_eur']==10000} for v in VENUES} maximum=max(float(r['mid_cost_bps']) for v in VENUES for r in selected[v].values() if r['included']) limit=math.ceil(maximum/10)*10 def xx(t): return left+(t-start)/(end-start)*(right-left) missing=sorted(set(range(times[0]//900,times[-1]//900+1))-set(t//900 for t in times)) for j,venue in enumerate(VENUES): top=120+j*195; bottom=top+145 for bucket in missing: parts.append(f'') for fraction in (0,.5,1): y=bottom-145*fraction parts.append(f'') parts.append(text_svg(left-12,y+4,f'{limit*fraction:g}','tick','end')) parts.append(text_svg(left,top-13,NAMES[venue]+' — bps','label',fill=COLORS[venue])) segments=[];current=[];previous=None for t in times: row=selected[venue].get(t) if not row or not row['included'] or (previous is not None and t//900-previous//900>1): if current:segments.append(current) current=[] if row and row['included']: current.append(f'{xx(t):.3f},{bottom-float(row["mid_cost_bps"])/limit*145:.3f}') previous=t if current:segments.append(current) for segment in segments: if len(segment)>1:parts.append(f'') else: x,y=segment[0].split(',');parts.append(f'') # Daily UTC boundaries plus clear exact capture endpoints in the caption. first_midnight=((start//86400)+1)*86400 for t in range(first_midnight,end,86400): parts.append(text_svg(xx(t),715,datetime.fromtimestamp(t,timezone.utc).strftime('%d %b'),'tick','middle')) parts.append(text_svg(52,750,'Zero baseline and common scale. Grey marks the unoccupied 15-minute bin; the line breaks across that gap.','small')) parts.append(text_svg(52,773,f'First capture {utc(times[0])}; last {utc(times[-1])}. No observations between captures.','small')) parts.append('');path.write_text('\n'.join(parts)+'\n',encoding='utf-8') def chart_tails(path, stats, n): title='A typical cost and a costly snapshot are different questions' subtitle=f'€10,000 replays, {n} matched captures. Median and 95th sample percentile; fees excluded.' parts=svg_open(title,subtitle,height=650) left,right,top,bottom=115,940,118,475 maximum=max(stats[v][s]['10000']['p95'] for v in VENUES for s in ('buy','sell')) limit=math.ceil(maximum/5)*5 for fraction in (0,.2,.4,.6,.8,1): x=left+(right-left)*fraction parts.append(f'') parts.append(text_svg(x,505,f'{limit*fraction:g}','tick','middle')) for j,venue in enumerate(VENUES): for k,side in enumerate(('buy','sell')): y=148+j*117+k*39 median=stats[venue][side]['10000']['median'];p95=stats[venue][side]['10000']['p95'] parts.append(text_svg(left-15,y+5,NAMES[venue]+' '+side,'label','end')) parts.append(f'') parts.append(f'') parts.append(f'') parts.append(text_svg(left+(right-left)*p95/limit+14,y+5,f'{median:.2f} / {p95:.2f}','small')) parts.append(text_svg(565,534,'Cost versus midpoint, basis points','label','middle')) parts.append('') parts.append(text_svg(132,569,'Median','label'));parts.append(text_svg(310,569,'95th sample percentile','label')) parts.append(text_svg(52,607,'Percentiles describe retained snapshots, not uncertainty or a future cost ceiling.','small')) parts.append(text_svg(52,630,'Sell quantity is €10,000 divided by each venue’s best bid; quantities differ across venues.','small')) parts.append('');path.write_text('\n'.join(parts)+'\n',encoding='utf-8') def main(): parser=argparse.ArgumentParser(description=__doc__) parser.add_argument('--input', type=Path, default=Path(__file__).resolve().parent) parser.add_argument('--output', type=Path, default=Path('reproduced')) args=parser.parse_args();base=args.input.resolve();out=args.output.resolve();out.mkdir(parents=True,exist_ok=True) raw_history=(base/'history.json').read_bytes() digest=hashlib.sha256(raw_history).hexdigest() if digest != EXPECTED_HISTORY_SHA: raise ValueError('This release reproduces one frozen input; history hash mismatch') capture=json.loads((base/'capture-record.json').read_text(encoding='utf-8')) if capture['sha256']!=digest or capture['bytes']!=len(raw_history):raise ValueError('Capture manifest mismatch') history=json.loads(raw_history) if history.get('protocol') != PROTOCOL or history.get('schema') != 1:raise ValueError('Unexpected protocol/schema') entries=history['entries'];times=[e['captured_at'] for e in entries] if times != sorted(set(times)):raise ValueError('Capture times not unique and strictly increasing') archive=zipfile.ZipFile(base/'raw-books.zip') if not (base/'raw').is_dir() else None archive_names={Path(n).name:n for n in archive.namelist()} if archive else {} errors=[];rows=[];snapshots=[];raw_records=[];auction=Counter();status=Counter();max_deltas=defaultdict(float);negative_residues=[] with localcontext() as ctx: ctx.prec=50 for entry in entries: t=entry['captured_at'];venue_list=entry['venues'] if sorted(v['venue'] for v in venue_list)!=sorted(VENUES):raise ValueError('Missing/duplicate venue record') for record in venue_list: venue=record['venue'];status[(venue,record['status'])]+=1 if record['status']!='ok': for b in BUDGETS: for s in ('buy','sell'): rows.append(dict(captured_at=t,captured_utc=utc(t),venue=venue,side=s,budget_eur=b,status=record['status'],auction_mode='',full='',included=False,target_ltc='',ltc='',eur='',vwap='',mid_cost_bps='',impact_bps='',spread_bps='',mid_eur='',best_bid_eur='',best_ask_eur='')) continue filename=record['raw_file'] if Path(filename).name!=filename or filename!=f'{t}-{venue}.json':raise ValueError('Unexpected raw filename') raw=(base/'raw'/filename).read_bytes() if archive is None else archive.read(archive_names[filename]) if hashlib.sha256(raw).hexdigest()!=record['sha256']:raise ValueError('Raw hash mismatch: '+filename) data=json.loads(raw);sides,book=normalize(data,venue) auction_flag=data.get('auction_mode') if venue=='coinbase' else None if venue=='coinbase':auction[str(auction_flag)]+=1 valid_continuous=(venue!='coinbase' or auction_flag is False) # A missing or true auction flag is not silently treated as continuous trading. bid,ask=sides['bids'][0][0],sides['asks'][0][0];mid=(bid+ask)/2 spread=(ask-bid)/mid*10000;metrics=record['metrics'] for key,value in [('bid',bid),('ask',ask),('mid',mid),('spread_bps',spread)]: kind='bps' if key=='spread_bps' else 'decimal' delta=assert_close(value,metrics[key],kind,f'{filename} {key}',errors) max_deltas[kind]=max(max_deltas[kind],delta) if metrics['ask_levels']!=len(sides['asks']) or metrics['bid_levels']!=len(sides['bids']):errors.append(filename+' level counts mismatch') for side in ('buy','sell'): depth=sum((p*q for p,q in sides['asks'] if p<=ask*Decimal('1.01')),Decimal(0)) if side=='buy' else sum((p*q for p,q in sides['bids'] if p>=bid*Decimal('.99')),Decimal(0)) delta=assert_close(depth,metrics['depth_1pct_eur'][side],'decimal',f'{filename} depth {side}',errors) max_deltas['decimal']=max(max_deltas['decimal'],delta) orders=metrics['orders'] if len(orders)!=6 or {(o['side'],o['budget_eur']) for o in orders}!={(s,b) for s in ('buy','sell') for b in BUDGETS}:raise ValueError('Unexpected order replay set') snap=dict(captured_at=t,captured_utc=utc(t),venue=venue,mid_eur=str(mid),best_bid_eur=str(bid),best_ask_eur=str(ask),spread_bps=float(spread),auction_mode=auction_flag,continuous=valid_continuous,request_at=record['request_at'],response_at=record['response_at'],elapsed_seconds=record['response_at']-record['request_at'],levels_buy=len(sides['asks']),levels_sell=len(sides['bids'])) snapshots.append(snap) for order in orders: side,budget=order['side'],order['budget_eur'];buy=side=='buy' target=Decimal(budget) if buy else Decimal(budget)/bid result=replay(sides['asks' if buy else 'bids'],target,buy) if result['full']!=order['full']:errors.append(filename+' full fill mismatch') for key in ('ltc','eur','vwap','filled_fraction'): if result[key] is not None: kind='fraction' if key=='filled_fraction' else 'decimal' delta=assert_close(result[key],order[key],kind,f'{filename} {side}/{budget} {key}',errors) max_deltas[kind]=max(max_deltas[kind],delta) if not buy:assert_close(target,order['target_ltc'],'decimal',filename+' target_ltc',errors) if buy and order['target_ltc'] is not None:errors.append(filename+' unexpected buy LTC target') if result['full']: vwap=result['vwap'] impact=max(Decimal(0),(vwap/ask-1 if buy else 1-vwap/bid)*10000) cost=max(Decimal(0),(vwap/mid-1 if buy else 1-vwap/mid)*10000) for key,value in [('impact_bps',impact),('mid_cost_bps',cost)]: if dec(order[key]) < 0: if dec(order[key]) < Decimal('-1e-12'):errors.append(filename+' materially negative '+key) negative_residues.append(dict(filename=filename,side=side,budget_eur=budget,field=key,recorded_value=order[key],canonical_value=0)) delta=assert_close(value,order[key],'bps',f'{filename} {side}/{budget} {key}',errors) max_deltas['bps']=max(max_deltas['bps'],delta) else: impact=cost=None if order['impact_bps'] is not None or order['mid_cost_bps'] is not None:errors.append(filename+' partial replay has published cost') rows.append(dict(captured_at=t,captured_utc=utc(t),venue=venue,side=side,budget_eur=budget,status='ok',auction_mode=auction_flag if venue=='coinbase' else '',full=result['full'],included=result['full'] and valid_continuous,target_ltc='' if buy else str(target),ltc=str(result['ltc']),eur=str(result['eur']),vwap=str(result['vwap']),mid_cost_bps='' if cost is None else float(cost),impact_bps='' if impact is None else float(impact),spread_bps=float(spread),mid_eur=str(mid),best_bid_eur=str(bid),best_ask_eur=str(ask))) raw_records.append(dict(filename=filename,sha256=record['sha256'],bytes=len(raw),venue=venue,url=record['url'],capture_utc=utc(t),request_at=record['request_at'],response_at=record['response_at'],exchange_at=record['exchange_at'],http_headers=record['headers'])) if archive:archive.close() if errors:raise ValueError('Reproduction mismatch:\n'+'\n'.join(errors[:30])) # Matched comparisons use identical capture-round labels, not atomic market timestamps. matched={} indexed={(r['captured_at'],r['venue'],r['side'],r['budget_eur']):r for r in rows} for side in ('buy','sell'): for b in BUDGETS: matched[(side,b)]=[t for t in times if all(indexed[(t,v,side,b)]['included'] for v in VENUES)] stats={v:{s:{str(b):descriptives([indexed[(t,v,s,b)]['mid_cost_bps'] for t in matched[(s,b)]]) for b in BUDGETS} for s in ('buy','sell')} for v in VENUES} coverage=[] bins=set(t//900 for t in times);expected=times[-1]//900-times[0]//900+1 missing=sorted(set(range(times[0]//900,times[-1]//900+1))-bins) for v in VENUES: these=[r for r in rows if r['venue']==v] coverage.append(dict(venue=v,capture_rounds=len(times),validated_books=status[(v,'ok')],unavailable_books=sum(n for (vv,ss),n in status.items() if vv==v and ss!='ok'),replays=len(these),full_replays=sum(r['full'] is True for r in these),partial_replays=sum(r['full'] is False for r in these),continuous_full_replays=sum(r['included'] for r in these))) # Lowest own-midpoint friction, not absolute price, pool quality, or venue recommendation. lowest={} for s in ('buy','sell'): for b in BUDGETS: counts=Counter();ties=0 for t in matched[(s,b)]: values={v:indexed[(t,v,s,b)]['mid_cost_bps'] for v in VENUES};m=min(values.values()) winners=[v for v,value in values.items() if abs(value-m)<=1e-10] if len(winners)>1:ties+=1 else:counts[winners[0]]+=1 lowest[f'{s}_{b}']=dict(matched_n=len(matched[(s,b)]),strict_lowest_counts={v:counts[v] for v in VENUES},tied_rounds=ties) round_skews=[] for t in times: group=[s for s in snapshots if s['captured_at']==t] if len(group)==3:round_skews.append(max(s['response_at'] for s in group)-min(s['response_at'] for s in group)) spread={v:descriptives([s['spread_bps'] for s in snapshots if s['venue']==v and s['continuous']]) for v in VENUES} extremes={} for v in VENUES: candidates=[r for r in rows if r['venue']==v and r['side']=='buy' and r['budget_eur']==10000 and r['included']] extremes[v]=dict(minimum=min(candidates,key=lambda r:r['mid_cost_bps']),maximum=max(candidates,key=lambda r:r['mid_cost_bps'])) results=dict(release='liquidity-history-20261005-v1',protocol=PROTOCOL,history_sha256=digest,capture_record=capture, study=dict(first_utc=utc(times[0]),last_utc=utc(times[-1]),elapsed_seconds=times[-1]-times[0],capture_rounds=len(times),planned_interval_seconds=900,quarter_hour_bins_intersecting_capture_period=expected,occupied_bins=len(bins),occupied_bin_fraction=len(bins)/expected,missing_bin_starts_utc=[utc(b*900) for b in missing],bin_definition='Floor capture-start Unix time by 900 seconds; include both endpoint bins, which may be partial. A bin is occupied if any retained capture starts within it. Sampling coverage, not venue or collector uptime.',off_quarter_hour_rounds=[dict(utc=utc(t),seconds_after_boundary=t%900) for t in times if t%900>60]), verification=dict(raw_books_checked=len(raw_records),raw_bytes_checked=sum(r['bytes'] for r in raw_records),all_raw_hashes_match=True,all_recorded_replays_reproduced=True,decimal_precision=50,decimal_relative_tolerance=str(DECIMAL_REL_TOL),bps_absolute_tolerance=str(FLOAT_BPS_ABS_TOL),maximum_observed_absolute_deltas=dict(max_deltas),historical_negative_rounding_residues=negative_residues,negative_residue_policy='Frozen input unchanged. Reject negatives below -1e-12 bps; bounded residue is zero in the independently recomputed nonnegative cost. No replay is excluded solely for this residue.',coinbase_auction_mode_counts=dict(auction),auction_policy='Exclude true or missing Coinbase auction_mode from continuous-trading comparisons; this cohort has false in every response.'), coverage=coverage,matched_round_counts={f'{s}_{b}':len(ts) for (s,b),ts in matched.items()},cost_stats_bps=stats,spread_stats_bps=spread,lowest_own_midpoint_friction=lowest,response_completion_skew_seconds=descriptives(round_skews),response_elapsed_seconds={v:descriptives([s['elapsed_seconds'] for s in snapshots if s['venue']==v]) for v in VENUES},buy_10000_extremes=extremes,last_capture=[r for r in rows if r['captured_at']==times[-1]], method=dict(quantile='Sort sample, linearly interpolate at (n - 1) * p; descriptive percentiles, not confidence intervals.',buy='Walk asks for an exact EUR budget, with a fractional final price level. VWAP = EUR spent / LTC quantity.',sell='Target LTC = reference EUR budget / that venue’s best bid. Walk bids for that quantity, with a fractional final price level. Cross-venue sell targets differ.',mid='(best bid + best ask) / 2',cost_bps='Buy: max(0,(VWAP/mid - 1)*10000). Sell: max(0,(1 - VWAP/mid)*10000). Publish only for full replays.',impact_bps='Buy: max(0,(VWAP/best ask - 1)*10000). Sell: max(0,(1 - VWAP/best bid)*10000).',full_tolerance='remaining <= max(1e-20, target * 1e-18), matching the collector protocol.',limitations=['Retained visible REST books, not executed orders.','Requested depth: Kraken 100 price levels per side; OKX 100; Coinbase aggregated level 2.','Size already aggregates each price level; the order-count field is not a multiplier.','No exchange fees, quantity rounding, minimum orders, funding/withdrawal/FX costs, latency, queue competition, hidden liquidity, or replenishment.','Capture-round labels match observations but requests and venue clocks are not atomic.','Lower relative midpoint cost need not mean a lower absolute purchase price.','No causal claim about volatility, exchange reliability or future execution; this period is short and sampled.'])) (out/'results.json').write_text(json.dumps(results,indent=2,ensure_ascii=False)+'\n',encoding='utf-8') (out/'raw-manifest.json').write_text(json.dumps(raw_records,indent=2)+'\n',encoding='utf-8') csv_write(out/'replays.csv',rows) csv_write(out/'snapshots.csv',snapshots) csv_write(out/'coverage.csv',coverage) summary=[] for v in VENUES: for s in ('buy','sell'): for b in BUDGETS:summary.append(dict(venue=v,side=s,budget_eur=b,**stats[v][s][str(b)])) csv_write(out/'summary.csv',summary) chart_medians(out/'buy-cost-by-size.svg',stats,len(matched[('buy',10000)])) chart_history(out/'buy-cost-history.svg',times,rows) chart_tails(out/'buy-sell-percentiles.svg',stats,len(matched[('buy',10000)])) print(json.dumps(dict(verified_books=len(raw_records),replays=len(rows),capture_rounds=len(times),output=str(out),coverage=coverage,median_buy_10000={v:stats[v]['buy']['10000']['median'] for v in VENUES},p95_buy_10000={v:stats[v]['buy']['10000']['p95'] for v in VENUES}))) if __name__ == '__main__': main()