{
  "date": "2026-08-30",
  "provider": "Vast.ai Serverless",
  "region": "NL",
  "endpoint_id": 35649,
  "workergroup_id": 44674,
  "worker_instance_id": 49315855,
  "stable_machine_id": 119701,
  "hardware": {
    "gpu": "RTX PRO 6000 S",
    "gpu_vram_gib": 96,
    "system_ram_gib": 2267,
    "effective_cpu_cores": 64,
    "pcie_generation": 5,
    "pcie_bandwidth_gbps": 55.0
  },
  "checkpoint": "LibertAIDAI/GLM-5.3-Flash-NVFP4",
  "runtime": "FreeToken",
  "runtime_commit": "3d5354c891a36a3a6e19e1062de194c29a5b3fcb",
  "template_id": 648749,
  "template_hash": "802f241a9200084f97085a5d194c4398",
  "docker_port": 3000,
  "vast_model_load_estimate_gib": 200,
  "cold_load_seconds": 262.69,
  "cached_resume": {
    "provision_stage": "fast_resume",
    "ready_seconds": 109,
    "within_vast_starting_deadline": true,
    "vast_starting_deadline_seconds": 300
  },
  "platform_benchmark": {
    "concurrency": 4,
    "measured_workload_per_second": 20.572114238848666
  },
  "semantic_smoke": {
    "http_status": 200,
    "latency_seconds": 13.761,
    "finish_reason": "stop",
    "expected": "GLM53_SERVERLESS_FAST_RESUME_OK",
    "actual": "GLM53_SERVERLESS_FAST_RESUME_OK",
    "prompt_tokens": 28,
    "completion_tokens": 165,
    "total_tokens": 193
  },
  "hourly_running_cost_usd": 1.6638888888888888,
  "lifecycle": {
    "endpoint_retained": true,
    "worker_retained": true,
    "inactivity_timeout_seconds": 300,
    "destructive_actions": false
  },
  "notes": "The first capped probe returned HTTP 200 but exhausted a 31-token completion allowance on reasoning. The qualification probe used a 256-token allowance and produced the exact visible sentinel. Cached resume skipped package, git, checkpoint download, dependency, and bandwidth setup stages."
}
