{
  "model": "Julia 1",
  "checkpoint_sha256": "df853bf7fe424420011f3d0c47a05d7341aa9eefa7fb9f203ea4aada4ad95b72",
  "measurements": [
    {
      "date": "2026-09-26",
      "machine": "Apple M4",
      "runtime": "Local julia.inference.Engine, PyTorch CPU, four threads",
      "tokenizer": "Public jhu-clsp/mmBERT-small tokenizer",
      "workload": "128 synthetic requests, 100 context words and four options each",
      "source": "Local benchmark_mac.json and benchmark_mac.py, /Users/joaodavisn/Downloads/Julia-1",
      "runs": [
        {
          "batch_size": 1,
          "requests": 128,
          "median_batch_ms": 33.15,
          "p95_batch_ms": 44.23,
          "requests_per_second": 28.01,
          "rss_end_mib": 278.9
        },
        {
          "batch_size": 16,
          "requests": 128,
          "median_batch_ms": 312.11,
          "p95_batch_ms": 313.44,
          "requests_per_second": 51.20,
          "rss_end_mib": 370.6
        }
      ]
    },
    {
      "reported_on": "2026-09-26",
      "machine": "Samsung SM-X510",
      "soc": "Samsung s5e8835 (gts9fewifi)",
      "board": "erd8835",
      "cpu": "8 ARMv8-A 64-bit cores, maximum clock 2.00 GHz",
      "android": "16 (API 36)",
      "abi": "arm64-v8a",
      "build": "samsung/gts9fewifixx/gts9fewifi",
      "ram_total_gb": 5.6,
      "ram_available_start_gb": 1.7,
      "ram_available_after_gb": 1.1,
      "runtime": "Python 3.13.13; ONNX Runtime 1.27.0",
      "providers": ["NnapiExecutionProvider", "CPUExecutionProvider"],
      "gpu_path": "NNAPI available; driver falls back to CPU per operator in this run",
      "xnnpack": "Not requested; unusable on this graph because a Reshape is miscompiled",
      "tokenization": "Native aarch64 Rust tokenizers binary",
      "encoder": "/data/data/com.termux/files/home/Julia-1-ONNX/rust/target/release/julia-encode",
      "request_contract": "2–20 options, strict encoding, maximum 1,024 tokens",
      "weight_file": "model.onnx.data, 550.1 MB, memory-mapped",
      "source": "Device run summary provided by the user; raw per-decision trace was not attached",
      "runs": [
        {
          "task": "typed decisions",
          "requests": 40,
          "elapsed_seconds": 8,
          "requests_per_second": 5.0,
          "median_request_ms": 203,
          "p95_request_ms": 205,
          "min_request_ms": 193,
          "max_request_ms": 205,
          "tokens_processed": 3693,
          "tokens_per_request_rounded": 92,
          "effective_tokens_per_second": 462,
          "peak_process_rss_mb": 393.1,
          "resident_runtime_and_activations_mb_approx": 335.8,
          "weight_file_page_cache": "Weight file remains in page cache; process RSS is reported separately"
        }
      ]
    },
    {
      "date": "2026-09-25",
      "machine": "Intel Core i5-1235U",
      "machine_source": "Processor model supplied with the evaluation; environment.json records CPU without a processor identifier",
      "runtime": "PyTorch 2.14.0+cpu; runtime file hashes and strict encoding in config.json",
      "source": "/Users/joaodavisn/Downloads/julia1-local-20260925/predictions.jsonl",
      "calculation": "Group latency_ms by suite; median is the median of recorded values; p95 is sorted[ceil(0.95 * n) - 1]. All statuses remain in each task's denominator.",
      "runs": [
        {
          "task": "typed-decisions",
          "requests": 2000,
          "median_request_ms": 294.81,
          "p95_request_ms": 428.49
        },
        {
          "task": "agnews",
          "requests": 100,
          "median_request_ms": 107.83,
          "p95_request_ms": 142.89
        },
        {
          "task": "emotiondair",
          "requests": 100,
          "median_request_ms": 89.83,
          "p95_request_ms": 118.95
        },
        {
          "task": "banking77",
          "requests": 100,
          "median_request_ms": 3713.54,
          "p95_request_ms": 5125.10,
          "notes": "72-label shortlist wrapper; 3 abstentions included"
        }
      ]
    }
  ]
}
