ENTITY Documentation Portal
ENTITY test 1 | recorded

Reproduce the 97.14% Level-3 storage result

Generate 250 million compact bonds for 50 million entities, build the entity index, verify 1,000 sampled lookups and compare the measured 4.4 GB representation with a 768-dimension float32 vector baseline.

Recorded result: 250,000,000 bonds | 50,000,000 entities | 4.4 GB actual | 153.6 GB vector baseline | 97.1354167% reduction | PASS

Watch the test

This VP8 WebM evidence replay is generated from the exact published test source and recorded terminal output below.

Run it yourself

python examples/level3_storage_benchmark.py

Exact test source

from pathlib import Path
import numpy as np, hashlib, json, time, ctypes, statistics
from ctypes import wintypes
OUT=Path(r'E:\ENTITY_ACTIVE\ENTITY_TESTING_20261001\runs\level3-storage'); OUT.mkdir(parents=True,exist_ok=True)
BONDS=250_000_000; FANOUT=5; ENTITIES=BONDS//FANOUT; CHUNK=5_000_000; DT=np.dtype([('source','<u4'),('predicate','<u2'),('flags','<u2'),('target','<u8')],align=False); REC=DT.itemsize
bp=OUT/'bonds_250m.bin'; ip=OUT/'entity_index_50m.u64'; [p.unlink() for p in (bp,ip) if p.exists()]
class PMC(ctypes.Structure): _fields_=[('cb',wintypes.DWORD),('PageFaultCount',wintypes.DWORD),('PeakWorkingSetSize',ctypes.c_size_t),('WorkingSetSize',ctypes.c_size_t),('QuotaPeakPagedPoolUsage',ctypes.c_size_t),('QuotaPagedPoolUsage',ctypes.c_size_t),('QuotaPeakNonPagedPoolUsage',ctypes.c_size_t),('QuotaNonPagedPoolUsage',ctypes.c_size_t),('PagefileUsage',ctypes.c_size_t),('PeakPagefileUsage',ctypes.c_size_t)]
ps=ctypes.WinDLL('psapi',use_last_error=True); k=ctypes.WinDLL('kernel32',use_last_error=True); ps.GetProcessMemoryInfo.argtypes=[wintypes.HANDLE,ctypes.POINTER(PMC),wintypes.DWORD]; ps.GetProcessMemoryInfo.restype=wintypes.BOOL; k.GetCurrentProcess.restype=wintypes.HANDLE
def mem():
 x=PMC(); x.cb=ctypes.sizeof(PMC); ok=ps.GetProcessMemoryInfo(k.GetCurrentProcess(),ctypes.byref(x),x.cb)
 if not ok: raise ctypes.WinError(ctypes.get_last_error())
 return {'working_set':x.WorkingSetSize,'peak_working_set':x.PeakWorkingSetSize,'pagefile':x.PagefileUsage,'peak_pagefile':x.PeakPagefileUsage}
start_mem=mem(); bh=hashlib.sha256(); t0=time.perf_counter()
with open(bp,'wb',buffering=8*1024*1024) as f:
 for start in range(0,BONDS,CHUNK):
  n=min(CHUNK,BONDS-start); ids=np.arange(start,start+n,dtype=np.uint64); a=np.empty(n,dtype=DT); a['source']=(ids//FANOUT).astype(np.uint32); a['predicate']=(ids%37).astype(np.uint16); a['flags']=1; a['target']=ids*np.uint64(6364136223846793005)+np.uint64(1442695040888963407); bh.update(a); a.tofile(f)
write_sec=time.perf_counter()-t0; write_mem=mem(); ih=hashlib.sha256(); t1=time.perf_counter()
with open(ip,'wb',buffering=8*1024*1024) as f:
 for start in range(0,ENTITIES,CHUNK):
  n=min(CHUNK,ENTITIES-start); off=np.arange(start,start+n,dtype=np.uint64)*np.uint64(FANOUT*REC); ih.update(off); off.tofile(f)
index_sec=time.perf_counter()-t1; bm=np.memmap(bp,dtype=DT,mode='r'); im=np.memmap(ip,dtype='<u8',mode='r'); rng=np.random.default_rng(342); lat=[]; ok=True
for s in rng.integers(0,ENTITIES,size=1000,dtype=np.int64):
 q=time.perf_counter_ns(); rec0=int(im[int(s)])//REC; rows=bm[rec0:rec0+FANOUT]; _=int(rows['target'][0]); lat.append((time.perf_counter_ns()-q)/1000); ok=ok and len(rows)==FANOUT and bool(np.all(rows['source']==s))
query_mem=mem(); atomic=bp.stat().st_size+ip.stat().st_size; rag=ENTITIES*768*4; saving=1-atomic/rag; res={'qualification':'ENTITY-v3.4.2 Atom Universe Level 3 scale corrected','actual':{'bonds':BONDS,'entities':ENTITIES,'record_bytes':REC,'bond_bytes':bp.stat().st_size,'index_bytes':ip.stat().st_size,'total_bytes':atomic,'bond_sha256':bh.hexdigest(),'index_sha256':ih.hexdigest(),'write_seconds':write_sec,'index_seconds':index_sec,'write_MBps':bp.stat().st_size/write_sec/1e6,'query_samples':len(lat),'query_correct':ok,'mean_query_us':statistics.mean(lat),'median_query_us':statistics.median(lat),'p95_query_us':float(np.percentile(lat,95))},'memory':{'start':start_mem,'after_write':write_mem,'after_queries':query_mem},'rag_baseline':{'dimension':768,'float_bytes':4,'vector_bytes':rag,'atomic_vs_vectors_saving_fraction':saving}}; res['pass']=bool(ok and BONDS>=100_000_000 and saving>=.90 and 0<query_mem['peak_working_set']<2_000_000_000); (OUT/'LEVEL3_SCALE_CORRECTED_RESULT.json').write_text(json.dumps(res,indent=2,sort_keys=True)); print(json.dumps(res,indent=2))

Recorded output

{
  "qualification": "ENTITY-v3.4.2 Atom Universe Level 3 scale corrected",
  "actual": {
    "bonds": 250000000,
    "entities": 50000000,
    "record_bytes": 16,
    "bond_bytes": 4000000000,
    "index_bytes": 400000000,
    "total_bytes": 4400000000,
    "bond_sha256": "17403382dd33f2ae573d1c35f91584530de57e0e3c2a13b6d7b7a7094a2c2f9e",
    "index_sha256": "c7aa7fa07ed15b597064b345dd2f925b4636a5d6c82dd79483f117b80e583e93",
    "write_seconds": 261.0570361999562,
    "index_seconds": 15.120423400076106,
    "write_MBps": 15.32232211866608,
    "query_samples": 1000,
    "query_correct": true,
    "mean_query_us": 14166.8641,
    "median_query_us": 11976.3,
    "p95_query_us": 41611.645
  },
  "memory": {
    "start": {
      "working_set": 28450816,
      "peak_working_set": 28450816,
      "pagefile": 254279680,
      "peak_pagefile": 254279680
    },
    "after_write": {
      "working_set": 135852032,
      "peak_working_set": 228581376,
      "pagefile": 374415360,
      "peak_pagefile": 463159296
    },
    "after_queries": {
      "working_set": 190656512,
      "peak_working_set": 256786432,
      "pagefile": 425140224,
      "peak_pagefile": 503201792
    }
  },
  "rag_baseline": {
    "dimension": 768,
    "float_bytes": 4,
    "vector_bytes": 153600000000,
    "atomic_vs_vectors_saving_fraction": 0.9713541666666666
  },
  "pass": true
}

Historical requalification check

The verifier compares the freshly regenerated bond and index SHA-256 values with the original historical Level-3 qualification artifacts.

from pathlib import Path
import json

FRESH=Path(r"E:\ENTITY_ACTIVE\ENTITY_TESTING_20261001\runs\level3-storage\LEVEL3_SCALE_CORRECTED_RESULT.json")
HISTORICAL=Path(r"E:\ADAM_ATOMIC_UNIVERSE_PROOF\LEVEL23_QUALIFICATION\level3_scale_corrected\LEVEL3_SCALE_CORRECTED_RESULT.json")
fresh=json.loads(FRESH.read_text(encoding="utf-8"))
hist=json.loads(HISTORICAL.read_text(encoding="utf-8"))
a=fresh["actual"]; h=hist["actual"]; r=fresh["rag_baseline"]
assert fresh["pass"] is True and a["query_correct"] is True and a["query_samples"] == 1000
assert a["bonds"] == 250_000_000 and a["entities"] == 50_000_000
assert a["total_bytes"] == 4_400_000_000 and r["vector_bytes"] == 153_600_000_000
assert a["bond_sha256"] == h["bond_sha256"]
assert a["index_sha256"] == h["index_sha256"]
assert abs(r["atomic_vs_vectors_saving_fraction"] - 0.9713541666666666) < 1e-15
print("STORAGE_REQUALIFICATION=PASS")
print(f"BONDS={a['bonds']}")
print(f"ENTITIES={a['entities']}")
print(f"BTDU_REPRESENTATION_BYTES={a['total_bytes']}")
print(f"RAG_768D_FLOAT32_BYTES={r['vector_bytes']}")
print(f"STORAGE_REDUCTION_PERCENT={100*r['atomic_vs_vectors_saving_fraction']:.10f}")
print(f"RETRIEVAL_SAMPLES_CORRECT={a['query_samples']}/{a['query_samples']}")
print("BOND_SHA256_MATCH_HISTORICAL=PASS")
print("INDEX_SHA256_MATCH_HISTORICAL=PASS")
print("CLAIM_BOUNDARY=this is a controlled representation comparison, not universal compression")

Requalification output

STORAGE_REQUALIFICATION=PASS
BONDS=250000000
ENTITIES=50000000
BTDU_REPRESENTATION_BYTES=4400000000
RAG_768D_FLOAT32_BYTES=153600000000
STORAGE_REDUCTION_PERCENT=97.1354166667
RETRIEVAL_SAMPLES_CORRECT=1000/1000
BOND_SHA256_MATCH_HISTORICAL=PASS
INDEX_SHA256_MATCH_HISTORICAL=PASS
CLAIM_BOUNDARY=this is a controlled representation comparison, not universal compression

Raw result JSON ->

Claim boundary: This is a controlled representation benchmark against a 768-dimension float32 vector baseline. It is not a universal compression ratio, not a SQLite comparison and not a claim that every ENTITY workload uses this representation.