{"object":"list","data":[{"datacenters":[{"country_code":"JP","region":"Japan"}],"schema_version":"2.4","capacity":{"text":[{"type":"output","unit":"token","window":"minute","value":500}],"max_concurrency":2},"name":"Qwen3.8 27B (murakumo-main)","owned_by":"murakumo","created":0,"input_modalities":["text","image"],"murakumo":{"capacity-superseded":{"measured-at":"2026-09-05","measured-aggregate-tok-s":1.04,"method":"N concurrent completions through api.murakumo.cloud; wall clock; usage.completion_tokens summed. 2026-09-05 re-measure: conc1=1.04, conc2=0.80 (3/4 timeout). Degraded from 2026-08-31's 35.56@conc4; b70 head queueing server-side. Practical cap conc2 while degraded.","why-superseded":"Not withdrawn as wrong. The reading was real, and so was the slowdown it noticed. It is superseded because the metric conflated prefill with decode, which made it prompt-length-dependent, which made the 35.56 -> 1.04 comparison meaningless as a time series even though both endpoints of it were measured honestly."},"datacenter-source":"operator declaration 2026-08-31: the owned fleet head decoding murakumo-main is in Japan","capacity-decode-tokens-per-second-favourable":28.23,"capacity-end-to-end-aggregate-tok-s":1.04,"capacity-decode-tokens-per-second":16.83,"capacity-aggregate-prompt-length-dependent":true,"capacity-end-to-end-aggregate-at":"2026-09-05","capacity-method":{"output-tokens-per-minute":"Derived, not measured: conservative decode rate (16.83 tok/s) x 60 x 0.5 for headroom, for ONE stream on the head that currently leads murakumo-main. Not multiplied by max_concurrency.","decode-tokens-per-second":"llama-server timings.predicted_per_second, read server-side from the non-streaming response body. b70, 2026-09-07, n=114 output tokens, unpredictable content (speculative acceptance 66/144). Comparable across prompts because prefill is a separate field.","decode-tokens-per-second-favourable":"Same boundary and head, n=200, highly predictable content (speculative acceptance 148/150). Published to bracket the rate, not as the rate.","prefill-tokens-per-second":"llama-server timings.prompt_per_second, server-side, b70, 2026-09-07, n=684 prompt tokens. Rises with prompt length (636 tok/s at n=3,943) because fixed per-request overhead amortizes; the smaller figure is published. Divide your prompt token count by this for a time-to-first-token estimate, then add the gateway hop.","max-concurrency":"Real slot count, read from each head's /slots on 2026-09-07, not a policy constant. gad runs --parallel 2 with -c 524288, i.e. two slots of 262,144 tokens each -- the only slots in the pool that can hold the advertised context window. b70 (1 slot, 32,768) and xavier (1 slot, 8,192) add two more for requests that fit them, so a small request can find four. 2 is the number that holds for every request this contract admits.","end-to-end-aggregate-tok-s":"N concurrent completions through https://api.murakumo.cloud; wall clock from first dispatch to last response; usage.completion_tokens summed. Includes the Worker hop, admission probe, tunnel and any queueing, and includes prefill inside the wall clock -- so it is prompt-length-dependent and two such readings are comparable only if the prompt was identical. Retained because it is what a caller's stopwatch shows.","members":"Per-head slot count and context read from that head's own /slots and /v1/models; per-head rates from that head's own llama-server timings, server-side, 2026-09-07. A nil is unmeasured and is never inferred from a sibling or from a config file. gad's 4.95 tok/s was taken with both of its slots already busy -- a loaded rate, stated as such, because a rate without the load it was taken under is not reproducible."},"capacity-members":[{"head":"b70","slots":2,"context":16384,"decode-tokens-per-second":16.83,"prefill-tokens-per-second":507},{"head":"gad","slots":2,"context":262144,"decode-tokens-per-second":4.95,"prefill-tokens-per-second":null},{"head":"xavier","slots":1,"context":8192,"decode-tokens-per-second":2.45,"prefill-tokens-per-second":33.8}],"quant-gguf":"Q4_K_M","capacity-prefill-tokens-per-second":507,"capacity-measured-at":"2026-09-07","quantization-source":"kotoba-lang/murakumo infer.edn :model/gguf Qwen3.8-27B-Q4_K_M.gguf (alias target qwen3.8-27b)"},"output_modalities":["text"],"is_ready":true,"id":"murakumo-main","quantization":"int4","max_output_tokens":32768,"pricing":{"text":[{"type":"prompt","unit":"token","cost_usd":"0.0000001"},{"type":"completion","unit":"token","cost_usd":"0.0000004"}]},"object":"model","compliance":{"zdr":true,"hipaa":false},"context_window":262144},{"id":"murakumo-edge","object":"model","created":0,"owned_by":"murakumo","context_window":65536,"is_ready":false,"murakumo":{"withheld":["datacenter-undeclared","no-published-price","capacity-unmeasured","quantization-undeclared"]}},{"id":"qwen3.8-27b-fastmtp-aggressive","object":"model","created":0,"owned_by":"murakumo","context_window":32768,"is_ready":false,"murakumo":{"withheld":["datacenter-undeclared","no-published-price","capacity-unmeasured","quantization-undeclared"]}},{"id":"qwen3.8-27b-throughput","object":"model","created":0,"owned_by":"murakumo","context_window":262144,"is_ready":false,"murakumo":{"withheld":["datacenter-undeclared","no-published-price","capacity-unmeasured","quantization-undeclared"]}},{"id":"qwen3.8-27b-throughput-5090","object":"model","created":0,"owned_by":"murakumo","context_window":65536,"is_ready":false,"murakumo":{"withheld":["datacenter-undeclared","no-published-price","capacity-unmeasured","quantization-undeclared"]}},{"id":"qwen3.8-27b-throughput-b70","object":"model","created":0,"owned_by":"murakumo","context_window":16384,"is_ready":false,"murakumo":{"withheld":["datacenter-undeclared","no-published-price","capacity-unmeasured","quantization-undeclared"]}},{"id":"awai-network/basho","object":"model","created":0,"owned_by":"awai.network","context_window":16384},{"id":"z-ai/glm-5.3-flash","object":"model","created":0,"owned_by":"z-ai"},{"id":"qwen/qwen3.8-flash","object":"model","created":0,"owned_by":"qwen"},{"id":"anthropic/claude-haiku-4.5","object":"model","created":0,"owned_by":"anthropic"},{"id":"anthropic/claude-sonnet-5","object":"model","created":0,"owned_by":"anthropic"},{"id":"anthropic/claude-opus-5","object":"model","created":0,"owned_by":"anthropic"},{"id":"awai-network/hokusai","object":"model","owned_by":"awai-network","endpoint":"/v1/images/generations","kind":"image","free":true,"sizes":["1024x1024","512x512","512x768","768x512","768x768"]},{"id":"glm-5.3-flash-cybersecurity-w4a16","object":"model","created":0,"owned_by":"murakumo","context_window":131072,"max_output_tokens":16384,"endpoint":"/v1/chat/completions","hugging_face_id":"dealignai/GLM-5.3-Flash-CYBERSECURITY-W4A16","quantization":"W4A16 (int4 experts, bf16 attention)","origin":"modal 2x H200, scale-to-zero","scale_to_zero":true,"cold_start_budget_seconds":2400,"reasoning_effort_default":"low"},{"id":"qwen3.8-flash-next-cybersecurity-nvfp4","object":"model","created":0,"owned_by":"murakumo","context_window":131072,"max_output_tokens":16384,"endpoint":"/v1/chat/completions","hugging_face_id":"dealignai/Qwen3.8-Flash-Next-CYBERSECURITY-NVFP4","quantization":"NVFP4 (modelopt; experts fp4, attention/embeddings bf16)","origin":"modal 1x B200, scale-to-zero, GPU memory snapshot","scale_to_zero":true,"cold_start_budget_seconds":2000}]}