// Package metrics provides the Prometheus text exposition for gpu-turnstile // without any external dependencies. package metrics import ( "fmt" "io" "math" "sync/atomic" ) var buckets = []float64{0.05, 0.1, 0.25, 0.5, 1, 2.5, 5, 10, 30, 60, 120, 300, 600} // histogram is a fixed-bucket cumulative histogram safe for concurrent use. type histogram struct { counts []atomic.Uint64 // len(buckets)+1; last slot is +Inf sum atomic.Uint64 // math.Float64bits count atomic.Uint64 } func newHistogram() *histogram { return &histogram{counts: make([]atomic.Uint64, len(buckets)+1)} } func (h *histogram) observe(v float64) { for i, b := range buckets { if v <= b { h.counts[i].Add(1) } } h.counts[len(buckets)].Add(1) h.sum.Add(math.Float64bits(v)) h.count.Add(1) } func (h *histogram) write(w io.Writer, name, label string) { for i, b := range buckets { fmt.Fprintf(w, "%s_bucket{%sle=\"%g\"} %d\n", name, label, b, h.counts[i].Load()) } fmt.Fprintf(w, "%s_bucket{%sle=\"+Inf\"} %d\n", name, label, h.counts[len(buckets)].Load()) fmt.Fprintf(w, "%s_sum{%s} %g\n", name, trimLabel(label), math.Float64frombits(h.sum.Load())) fmt.Fprintf(w, "%s_count{%s} %d\n", name, trimLabel(label), h.count.Load()) } // trimLabel converts a `kind="llm",`-style prefix into the label set used for // _sum/_count lines (same labels, no trailing comma). func trimLabel(label string) string { if len(label) > 0 && label[len(label)-1] == ',' { return label[:len(label)-1] } return label } // Metrics holds all gpu-turnstile metric values. type Metrics struct { llmWait *histogram imageWait *histogram unload *histogram imageJobs atomic.Uint64 } // New returns a zeroed Metrics. func New() *Metrics { return &Metrics{ llmWait: newHistogram(), imageWait: newHistogram(), unload: newHistogram(), } } // ObserveLockWait records how long a lock acquisition of the given kind // ("llm" or "image") waited. func (m *Metrics) ObserveLockWait(kind string, seconds float64) { if kind == "image" { m.imageWait.observe(seconds) } else { m.llmWait.observe(seconds) } } // ObserveUnload records how long an Ollama unload took. func (m *Metrics) ObserveUnload(seconds float64) { m.unload.observe(seconds) } // IncImageJobs counts one accepted image job. func (m *Metrics) IncImageJobs() { m.imageJobs.Add(1) } // Render writes the Prometheus text exposition for the given lock snapshot. func (m *Metrics) Render(w io.Writer, state string, llmInflight int, imagePending bool) { fmt.Fprint(w, `# HELP gpu_turnstile_state Current GPU state (1 for the active state). # TYPE gpu_turnstile_state gauge `) for _, s := range []string{"idle", "llm", "image", "external"} { v := 0 if s == state { v = 1 } fmt.Fprintf(w, "gpu_turnstile_state{state=%q} %d\n", s, v) } pending := 0 if imagePending { pending = 1 } fmt.Fprintf(w, `# HELP gpu_turnstile_llm_inflight LLM requests currently in flight. # TYPE gpu_turnstile_llm_inflight gauge gpu_turnstile_llm_inflight %d # HELP gpu_turnstile_image_pending Whether an image job is active or waiting. # TYPE gpu_turnstile_image_pending gauge gpu_turnstile_image_pending %d # HELP gpu_turnstile_image_jobs_total Image jobs accepted by ComfyUI. # TYPE gpu_turnstile_image_jobs_total counter gpu_turnstile_image_jobs_total %d # HELP gpu_turnstile_lock_wait_seconds Time spent waiting to acquire the GPU lock. # TYPE gpu_turnstile_lock_wait_seconds histogram `, llmInflight, pending, m.imageJobs.Load()) m.llmWait.write(w, "gpu_turnstile_lock_wait_seconds", `kind="llm",`) m.imageWait.write(w, "gpu_turnstile_lock_wait_seconds", `kind="image",`) fmt.Fprint(w, `# HELP gpu_turnstile_unload_seconds Time spent unloading Ollama models before an image job. # TYPE gpu_turnstile_unload_seconds histogram `) m.unload.write(w, "gpu_turnstile_unload_seconds", "") }