diff --git a/src/runtime/metrics.go b/src/runtime/metrics.go index 1b049e0baa1b1a..0d37bbb036391c 100644 --- a/src/runtime/metrics.go +++ b/src/runtime/metrics.go @@ -26,6 +26,10 @@ var ( sizeClassBuckets []float64 timeHistBuckets []float64 + + // tailscaleStackSizeBuckets is a Tailscale addition; see + // tailscaleStackHistBuckets. + tailscaleStackSizeBuckets []float64 ) type metricData struct { @@ -86,6 +90,7 @@ func initMetrics() { sizeClassBuckets = append(sizeClassBuckets, float64Inf()) timeHistBuckets = timeHistogramMetricsBuckets() + tailscaleStackSizeBuckets = tailscaleStackHistBuckets() metrics = map[string]metricData{ "/cgo/go-to-c-calls:calls": { compute: func(_ *statAggregate, out *metricValue) { @@ -545,6 +550,30 @@ func initMetrics() { out.scalar = float64bits(nsToSec(totalMutexWaitTimeNanos())) }, }, + "/tailscale/sched/goroutines-by-stack-size:bytes": { + compute: func(_ *statAggregate, out *metricValue) { + hist := out.float64HistOrInit(tailscaleStackSizeBuckets) + tailscaleStackHistRead(hist.counts) + }, + }, + "/tailscale/sched/stacks/copied:bytes": { + compute: func(_ *statAggregate, out *metricValue) { + out.kind = metricKindUint64 + out.scalar = tailscaleStackBytesCopied.Load() + }, + }, + "/tailscale/sched/stacks/growths:events": { + compute: func(_ *statAggregate, out *metricValue) { + out.kind = metricKindUint64 + out.scalar = tailscaleStackGrowths.Load() + }, + }, + "/tailscale/sched/stacks/shrinks:events": { + compute: func(_ *statAggregate, out *metricValue) { + out.kind = metricKindUint64 + out.scalar = tailscaleStackShrinks.Load() + }, + }, } for _, info := range godebugs.All { diff --git a/src/runtime/metrics/description.go b/src/runtime/metrics/description.go index b9a6ab5fea6083..cfbb78e1f0dd26 100644 --- a/src/runtime/metrics/description.go +++ b/src/runtime/metrics/description.go @@ -509,6 +509,29 @@ var allDesc = []Description{ Kind: KindFloat64, Cumulative: true, }, + { + Name: "/tailscale/sched/goroutines-by-stack-size:bytes", + Description: "Distribution of live goroutines by the current size of their stacks, including runtime system goroutines. Stack sizes are always powers of two, so each bucket but the last counts exactly the goroutines whose stack size is the bucket's lower bound, and the last bucket counts every goroutine with a larger stack. The runtime updates these counts as goroutines are created, exit, and have their stacks grown or shrunk, so reading this metric does not visit every goroutine. The bucket counts sum to approximately /sched/goroutines:goroutines. This metric is specific to the Tailscale fork of Go.", + Kind: KindFloat64Histogram, + }, + { + Name: "/tailscale/sched/stacks/copied:bytes", + Description: "Cumulative bytes of goroutine stack copied while growing or shrinking stacks since program start. Every stack growth or shrink copies the in-use part of the goroutine's stack to a new allocation, so this measures the runtime's stack copying work. This metric is specific to the Tailscale fork of Go.", + Kind: KindUint64, + Cumulative: true, + }, + { + Name: "/tailscale/sched/stacks/growths:events", + Description: "Count of goroutine stack growths since program start. A stack doubles in size when a function call would overflow it. Compared with /tailscale/sched/stacks/shrinks:events this shows whether stacks are repeatedly growing and then being shrunk back by the garbage collector. This metric is specific to the Tailscale fork of Go.", + Kind: KindUint64, + Cumulative: true, + }, + { + Name: "/tailscale/sched/stacks/shrinks:events", + Description: "Count of goroutine stack shrinks since program start. The garbage collector halves a goroutine's stack when it finds the goroutine using less than a quarter of it. This metric is specific to the Tailscale fork of Go.", + Kind: KindUint64, + Cumulative: true, + }, } func init() { diff --git a/src/runtime/metrics/doc.go b/src/runtime/metrics/doc.go index 2ee0235fb566df..dca5c119f3df4d 100644 --- a/src/runtime/metrics/doc.go +++ b/src/runtime/metrics/doc.go @@ -587,5 +587,38 @@ Below is the full list of supported metrics, ordered lexicographically. is useful for identifying global changes in lock contention. Collect a mutex or block profile using the runtime/pprof package for more detailed contention data. + + /tailscale/sched/goroutines-by-stack-size:bytes + Distribution of live goroutines by the current size of their + stacks, including runtime system goroutines. Stack sizes are + always powers of two, so each bucket but the last counts exactly + the goroutines whose stack size is the bucket's lower bound, + and the last bucket counts every goroutine with a larger stack. + The runtime updates these counts as goroutines are created, + exit, and have their stacks grown or shrunk, so reading this + metric does not visit every goroutine. The bucket counts sum + to approximately /sched/goroutines:goroutines. This metric is + specific to the Tailscale fork of Go. + + /tailscale/sched/stacks/copied:bytes + Cumulative bytes of goroutine stack copied while growing or + shrinking stacks since program start. Every stack growth or + shrink copies the in-use part of the goroutine's stack to a new + allocation, so this measures the runtime's stack copying work. + This metric is specific to the Tailscale fork of Go. + + /tailscale/sched/stacks/growths:events + Count of goroutine stack growths since program start. A stack + doubles in size when a function call would overflow it. Compared + with /tailscale/sched/stacks/shrinks:events this shows whether + stacks are repeatedly growing and then being shrunk back by the + garbage collector. This metric is specific to the Tailscale fork + of Go. + + /tailscale/sched/stacks/shrinks:events + Count of goroutine stack shrinks since program start. + The garbage collector halves a goroutine's stack when it finds + the goroutine using less than a quarter of it. This metric is + specific to the Tailscale fork of Go. */ package metrics diff --git a/src/runtime/proc.go b/src/runtime/proc.go index 4c655405cbb117..12130b35055ace 100644 --- a/src/runtime/proc.go +++ b/src/runtime/proc.go @@ -2561,6 +2561,13 @@ func oneNewExtraM() { // has the same effect. sched.ngsys.Add(1) + // The goroutine's stack can grow while it runs cgo callbacks, and + // copystack records that in the stack size histogram, so record + // its initial size too or the histogram would go negative. Extra + // Ms are never destroyed, so there is no matching decrement. They + // are also rare, so the global counter is fine here. + tailscaleStackHistAdd(nil, gp.stack.hi-gp.stack.lo, 1) + // Add m to the extra list. addExtraM(mp) } @@ -4521,6 +4528,7 @@ func gdestroy(gp *g) { casgstatus(gp, _Grunning, _Gdead) gcController.addScannableStack(pp, -int64(gp.stack.hi-gp.stack.lo)) + tailscaleStackHistAdd(pp, gp.stack.hi-gp.stack.lo, -1) if isSystemGoroutine(gp, false) { sched.ngsys.Add(-1) } @@ -5420,6 +5428,7 @@ func newproc1(fn *funcval, callergp *g, callerpc uintptr, parked bool, waitreaso newg.tracking = true } gcController.addScannableStack(pp, int64(newg.stack.hi-newg.stack.lo)) + tailscaleStackHistAdd(pp, newg.stack.hi-newg.stack.lo, 1) // Get a goid and switch to runnable. This needs to happen under traceAcquire // since it's a goroutine transition. See tracer invariants in trace.go. @@ -6069,6 +6078,7 @@ func (pp *p) destroy() { pp.cleanupsQueued = 0 sched.goroutinesCreated.Add(int64(pp.goroutinesCreated)) pp.goroutinesCreated = 0 + tailscaleStackHistFlush(pp) pp.xRegs.free() pp.status = _Pdead } diff --git a/src/runtime/runtime2.go b/src/runtime/runtime2.go index 6f75df405b9924..37d0dadbcfaada 100644 --- a/src/runtime/runtime2.go +++ b/src/runtime/runtime2.go @@ -920,6 +920,20 @@ type p struct { // scheduler ASAP (regardless of what G is running on it). preempt bool + // tsStackHist counts live goroutines by stack size class. It is a + // Tailscale addition backing the + // /tailscale/sched/goroutines-by-stack-size:bytes metric; see the + // comment above tailscaleStackHistBaseOrder for the indexing and + // how the counts are used. It sits here, in the padding after + // preempt, so that it shares a cache line with goroutinesCreated, + // which newproc1 also writes, rather than dirtying one of its own. + // Only the P's owner writes it, and tailscaleStackHistRead reads + // it from other Ps without synchronization, like goroutinesCreated. + // The entries are int16 so that the whole array fits in that cache + // line; they are flushed to tailscaleStackHist once they drift + // tailscaleStackHistSlack from zero, so they cannot overflow. + tsStackHist [tailscaleStackHistLen]int16 + // gcStopTime is the nanotime timestamp that this P last entered _Pgcstop. gcStopTime int64 diff --git a/src/runtime/stack.go b/src/runtime/stack.go index fcaed68206a678..15e7faaa5bc9be 100644 --- a/src/runtime/stack.go +++ b/src/runtime/stack.go @@ -940,6 +940,7 @@ func copystack(gp *g, newsize uintptr) { // It's also fine if we have no P, addScannableStack can deal with // that case. gcController.addScannableStack(getg().m.p.ptr(), int64(newsize)-int64(old.hi-old.lo)) + tailscaleNoteStackCopy(getg().m.p.ptr(), old.hi-old.lo, newsize, used) // allocate new stack new := stackalloc(uint32(newsize)) diff --git a/src/runtime/tailscale_export_test.go b/src/runtime/tailscale_export_test.go new file mode 100644 index 00000000000000..7519522df8acb2 --- /dev/null +++ b/src/runtime/tailscale_export_test.go @@ -0,0 +1,40 @@ +// Copyright 2026 Tailscale. All rights reserved. +// Use of this source code is governed by a BSD-style +// license that can be found in the LICENSE file. + +package runtime + +import "unsafe" + +// TailscaleStackHistSlow stops the world and returns the goroutine +// stack size histogram computed two ways: want, by walking every +// goroutine, and got, from the incrementally maintained counts via the +// same code path the /tailscale/sched/goroutines-by-stack-size:bytes +// metric uses. Both are indexed like the metric's buckets. Nothing can +// change while the world is stopped, so the two must match exactly. +func TailscaleStackHistSlow() (want, got []uint64) { + stw := stopTheWorld(stwForTestReadMetricsSlow) + first := tailscaleStackHistFirst() + want = make([]uint64, tailscaleStackHistLen-first) + got = make([]uint64, tailscaleStackHistLen-first) + forEachG(func(gp *g) { + // Dead goroutines waiting on a free list still own a stack, + // but they are not live and are not counted. Goroutines that + // belong to extra Ms are _Gdeadextra rather than _Gdead + // between cgo callbacks, and those are counted. + if readgstatus(gp)&^_Gscan == _Gdead { + return + } + want[tailscaleStackHistIndex(gp.stack.hi-gp.stack.lo)-first]++ + }) + tailscaleStackHistRead(got) + startTheWorld(stw) + return want, got +} + +// TailscaleStackHistLayout reports where p.tsStackHist and +// p.goroutinesCreated live within p, so that a test can check that +// they share a cache line. +func TailscaleStackHistLayout() (histOff, histSize, goroutinesCreatedOff uintptr) { + return unsafe.Offsetof(p{}.tsStackHist), unsafe.Sizeof(p{}.tsStackHist), unsafe.Offsetof(p{}.goroutinesCreated) +} diff --git a/src/runtime/tailscale_runtime.go b/src/runtime/tailscale_runtime.go index 4798b54fbcf577..b176d66d2dc311 100644 --- a/src/runtime/tailscale_runtime.go +++ b/src/runtime/tailscale_runtime.go @@ -5,6 +5,7 @@ package runtime import ( + "internal/runtime/atomic" "internal/runtime/sys" ) @@ -112,3 +113,185 @@ func tailscaleNoteStackGrowth(gp *g, newsize uintptr) { gp.tsMaxStackOrder = order } } + +// The goroutine stack size histogram behind the +// /tailscale/sched/goroutines-by-stack-size:bytes metric counts live +// goroutines by the power-of-two size class of their stacks. Each P +// keeps its own small array of counters, p.tsStackHist, so that +// goroutine creation and exit never touch shared memory, and +// tailscaleStackHist holds whatever is not attributed to a P. The +// counts are maintained as goroutines are created, exit, and have +// their stacks resized, so reading the metric never walks the +// goroutines. +// +// The histogram distinguishes stack sizes from +// 1<= tailscaleStackHistLen { + i = tailscaleStackHistLen - 1 + } + return int(i) +} + +// tailscaleStackHistAdd records that delta more live goroutines (or +// fewer, if delta is negative) have a stack of size bytes. +// +// pp is the caller's P, or nil to use the global counter instead. If +// pp is non-nil, the caller must own it and must not be preemptible; +// all callers satisfy this by running on the system stack or with an +// M held. +func tailscaleStackHistAdd(pp *p, size uintptr, delta int16) { + i := tailscaleStackHistIndex(size) + if pp == nil { + tailscaleStackHist[i].Add(int64(delta)) + return + } + n := pp.tsStackHist[i] + delta + if n >= tailscaleStackHistSlack || n <= -tailscaleStackHistSlack { + tailscaleStackHist[i].Add(int64(n)) + n = 0 + } + pp.tsStackHist[i] = n +} + +// tailscaleNoteStackCopy records that a live goroutine's stack of +// oldsize bytes, of which used bytes are in use, is being replaced by +// one of newsize bytes. It moves the goroutine between size classes in +// the histogram and updates the cumulative growth, shrink, and copied +// byte counters. It is called from copystack. +func tailscaleNoteStackCopy(pp *p, oldsize, newsize, used uintptr) { + tailscaleStackHistAdd(pp, oldsize, -1) + tailscaleStackHistAdd(pp, newsize, 1) + if newsize > oldsize { + tailscaleStackGrowths.Add(1) + } else { + tailscaleStackShrinks.Add(1) + } + tailscaleStackBytesCopied.Add(int64(used)) +} + +// tailscaleStackHistFlush moves pp's stack size counts into +// tailscaleStackHist so that they survive pp being destroyed. +// +// The world must be stopped. +func tailscaleStackHistFlush(pp *p) { + assertWorldStopped() + for i := range pp.tsStackHist { + if n := pp.tsStackHist[i]; n != 0 { + tailscaleStackHist[i].Add(int64(n)) + pp.tsStackHist[i] = 0 + } + } +} + +// tailscaleStackHistFirst returns the index of the size class of the +// smallest stack a goroutine can have, which is the first bucket the +// metric reports. On most platforms fixedStack is 2 KiB and this is +// 0, but a larger fixedStack, as with the race detector, leaves the +// low entries permanently unused. +func tailscaleStackHistFirst() int { + return tailscaleStackHistIndex(fixedStack) +} + +// tailscaleStackHistBuckets returns the bucket boundaries of the +// /tailscale/sched/goroutines-by-stack-size:bytes histogram: every +// power of two from fixedStack through +// 1<<(tailscaleStackHistMaxOrder+1), then +Inf. Each bucket but the +// last therefore counts exactly the goroutines whose stack is the +// bucket's lower bound, and the last counts every goroutine with a +// larger stack. +func tailscaleStackHistBuckets() []float64 { + minOrder := sys.TrailingZeros64(uint64(fixedStack)) + buckets := make([]float64, 0, tailscaleStackHistMaxOrder+3-minOrder) + for o := minOrder; o <= tailscaleStackHistMaxOrder+1; o++ { + buckets = append(buckets, float64(uint64(1)<= float64(size) { + n += c + } + } + return n +} + +// checkStackHist requires that the incrementally maintained stack size +// histogram exactly matches one computed by walking every goroutine +// with the world stopped. +func checkStackHist(t *testing.T, what string) { + t.Helper() + want, got := runtime.TailscaleStackHistSlow() + if !slices.Equal(want, got) { + t.Errorf("%s: stack size histogram is out of sync with the goroutines\n got: %v\nwant: %v", what, got, want) + return + } + t.Logf("%s: %v", what, got) +} + +// tsParkDeep recursively consumes at least n bytes of stack, then calls +// parked.Done and blocks until release is closed, so that the goroutine +// stays parked with its grown stack. +// +//go:noinline +func tsParkDeep(n int, parked *sync.WaitGroup, release <-chan struct{}) byte { + var buf [256]byte + if n <= 0 { + parked.Done() + <-release + return buf[0] + } + buf[n%len(buf)] = byte(n) + return tsParkDeep(n-len(buf), parked, release) + buf[n%len(buf)] +} + +func TestTailscaleStackHistMetric(t *testing.T) { + hist := readStackHist(t) + if len(hist.Counts) != len(hist.Buckets)-1 { + t.Fatalf("len(Counts) = %d, len(Buckets) = %d; want Counts to be one shorter", len(hist.Counts), len(hist.Buckets)) + } + if len(hist.Buckets) < 2 { + t.Fatalf("Buckets = %v; want at least two boundaries", hist.Buckets) + } + bounds := hist.Buckets[:len(hist.Buckets)-1] + for i, b := range bounds { + if b <= 0 || b != math.Trunc(b) || bits.OnesCount64(uint64(b)) != 1 { + t.Errorf("Buckets[%d] = %v; want a power of two", i, b) + } + if i > 0 && b != 2*bounds[i-1] { + t.Errorf("Buckets[%d] = %v; want double Buckets[%d] = %v", i, b, i-1, bounds[i-1]) + } + } + if last := hist.Buckets[len(hist.Buckets)-1]; !math.IsInf(last, 1) { + t.Errorf("last bucket boundary = %v; want +Inf", last) + } + + // This goroutine's own stack must be counted in the bucket for + // its size. + var st runtime.TailscaleStackStats + runtime.TailscaleReadStackStats(&st) + if i := slices.Index(bounds, float64(st.Size)); i < 0 { + t.Errorf("no bucket for this goroutine's stack size %d in %v", st.Size, bounds) + } else if hist.Counts[i] == 0 { + t.Errorf("bucket for this goroutine's stack size %d is empty", st.Size) + } + + // The bucket counts sum to the goroutine count. Both are read + // without stopping the world, so retry in case goroutines came or + // went in between. + s := []metrics.Sample{{Name: stackHistMetric}, {Name: "/sched/goroutines:goroutines"}} + for attempt := 1; ; attempt++ { + metrics.Read(s) + var sum uint64 + for _, c := range s[0].Value.Float64Histogram().Counts { + sum += c + } + if sum == s[1].Value.Uint64() { + break + } + if attempt == 20 { + t.Fatalf("histogram sums to %d goroutines, /sched/goroutines:goroutines = %d; never matched", sum, s[1].Value.Uint64()) + } + t.Logf("histogram sums to %d goroutines, /sched/goroutines:goroutines = %d; retrying", sum, s[1].Value.Uint64()) + } +} + +func TestTailscaleStackHistTracking(t *testing.T) { + checkStackHist(t, "initial") + + release := make(chan struct{}) + var wg sync.WaitGroup + + // Create goroutines that park immediately with their initial + // stacks. + const idle = 500 + for range idle { + wg.Add(1) + go func() { + defer wg.Done() + <-release + }() + } + checkStackHist(t, "after creating goroutines") + + // Grow some stacks and keep them grown by parking deep in the + // recursion. + const grown = 10 + const growTo = 64 << 10 + before := stackHistCountAtLeast(readStackHist(t), growTo) + var parked sync.WaitGroup + for range grown { + wg.Add(1) + parked.Add(1) + go func() { + defer wg.Done() + tsParkDeep(growTo, &parked, release) + }() + } + parked.Wait() + checkStackHist(t, "after growing stacks") + if n := stackHistCountAtLeast(readStackHist(t), growTo); n < before+grown { + t.Errorf("goroutines with stacks of at least %d bytes: %d; want at least %d", growTo, n, before+grown) + } + + // A stack larger than the histogram's largest exact bucket lands + // in the catch-all last bucket. The last exact bucket is 16 MiB, + // so using 32 MiB of stack, which needs a 64 MiB stack, gets there + // with room to spare. + hist := readStackHist(t) + catchAll := uint64(hist.Buckets[len(hist.Buckets)-2]) + wg.Add(1) + parked.Add(1) + go func() { + defer wg.Done() + tsParkDeep(int(catchAll), &parked, release) + }() + parked.Wait() + checkStackHist(t, "after growing one stack past the largest bucket") + if hist = readStackHist(t); hist.Counts[len(hist.Counts)-1] == 0 { + t.Errorf("catch-all bucket [%d, +Inf) is empty; want the goroutine using %d bytes of stack", catchAll, catchAll) + } + + // Grow other stacks and then return to a shallow frame before + // parking, so that the garbage collector shrinks them. + for range grown { + wg.Add(1) + parked.Add(1) + go func() { + defer wg.Done() + var m runtime.TailscaleStackStats + tsUseStack(growTo, &m) + parked.Done() + <-release + }() + } + parked.Wait() + beforeGC := stackHistCountAtLeast(readStackHist(t), growTo) + for range 5 { + runtime.GC() + } + checkStackHist(t, "after GC shrank stacks") + if afterGC := stackHistCountAtLeast(readStackHist(t), growTo); beforeGC < before+2*grown { + t.Logf("skipping shrink check: only %d of %d grown stacks were still large before GC", beforeGC-before, 2*grown) + } else if afterGC >= beforeGC { + t.Errorf("goroutines with stacks of at least %d bytes after GC: %d; want fewer than %d", growTo, afterGC, beforeGC) + } + + // Changing GOMAXPROCS destroys and creates Ps. The counts held by + // destroyed Ps must survive. + procs := runtime.GOMAXPROCS(0) + defer runtime.GOMAXPROCS(procs) + runtime.GOMAXPROCS(1) + checkStackHist(t, "after GOMAXPROCS(1)") + runtime.GOMAXPROCS(procs + 2) + checkStackHist(t, "after growing GOMAXPROCS") + runtime.GOMAXPROCS(procs) + + close(release) + wg.Wait() + checkStackHist(t, "after goroutines exited") + + // The 64 MiB stack skewed the average stack size that the runtime + // starts new goroutines with (see gcComputeStartingStackSize), and + // it stays skewed until the next GC recomputes it from the + // goroutines that are still alive. Do that now rather than leaving + // every goroutine the next test creates with a huge stack. + runtime.GC() + done := make(chan struct{}) + go func() { + defer close(done) + var m runtime.TailscaleStackStats + runtime.TailscaleReadStackStats(&m) + t.Logf("fresh goroutine after GC: %+v", m) + }() + <-done +} + +// TestTailscaleStackHistCacheLine checks that the per-P histogram +// counters share a cache line with goroutinesCreated, which goroutine +// creation writes anyway, so that maintaining the histogram dirties no +// extra cache line. The layout differs on 32-bit platforms, where the +// runtime is not tuned this carefully, so only 64-bit ones are checked. +// If an upstream change to the p struct breaks this, move tsStackHist +// so that it is whole within goroutinesCreated's line again. +func TestTailscaleStackHistCacheLine(t *testing.T) { + if bits.UintSize != 64 { + t.Skip("layout is only tuned for 64-bit platforms") + } + const line = 64 + histOff, histSize, createdOff := runtime.TailscaleStackHistLayout() + t.Logf("p.tsStackHist at offset %d, %d bytes; p.goroutinesCreated at offset %d", histOff, histSize, createdOff) + if histOff/line != (histOff+histSize-1)/line { + t.Errorf("p.tsStackHist straddles a %d-byte cache line", line) + } + if histOff/line != createdOff/line { + t.Errorf("p.tsStackHist and p.goroutinesCreated are in different %d-byte cache lines", line) + } +} + +// readStackCopyMetrics returns the cumulative stack growth, shrink, and +// copied byte counters. +func readStackCopyMetrics(tb testing.TB) (growths, shrinks, copied uint64) { + tb.Helper() + s := []metrics.Sample{ + {Name: "/tailscale/sched/stacks/growths:events"}, + {Name: "/tailscale/sched/stacks/shrinks:events"}, + {Name: "/tailscale/sched/stacks/copied:bytes"}, + } + metrics.Read(s) + for _, v := range s { + if kind := v.Value.Kind(); kind != metrics.KindUint64 { + tb.Fatalf("%s: kind = %v; want KindUint64", v.Name, kind) + } + } + return s[0].Value.Uint64(), s[1].Value.Uint64(), s[2].Value.Uint64() +} + +func TestTailscaleStackCopyMetrics(t *testing.T) { + growths0, shrinks0, copied0 := readStackCopyMetrics(t) + + // Grow a fresh goroutine's stack several times, then let the GC + // shrink it back. TailscaleReadStackStats reports how many times + // that one goroutine grew and shrank, and the process-wide + // counters must have increased by at least that much. The + // recursion depth is relative to the initial stack size because + // the runtime adapts the size it starts goroutines with. + var initial, deep, shrunk runtime.TailscaleStackStats + done := make(chan struct{}) + go func() { + defer close(done) + runtime.TailscaleReadStackStats(&initial) + tsUseStack(int(initial.Size)*8, &deep) + for range 5 { + runtime.GC() + } + runtime.TailscaleReadStackStats(&shrunk) + }() + <-done + + growths1, shrinks1, copied1 := readStackCopyMetrics(t) + t.Logf("goroutine %+v; growths %d -> %d, shrinks %d -> %d, copied %d -> %d", + shrunk, growths0, growths1, shrinks0, shrinks1, copied0, copied1) + if deep.Growths == 0 || shrunk.Shrinks == 0 { + t.Fatalf("test goroutine did not both grow (%d) and shrink (%d) its stack", deep.Growths, shrunk.Shrinks) + } + if got, want := growths1-growths0, uint64(deep.Growths); got < want { + t.Errorf("growths increased by %d; want at least %d", got, want) + } + if got, want := shrinks1-shrinks0, uint64(shrunk.Shrinks); got < want { + t.Errorf("shrinks increased by %d; want at least %d", got, want) + } + // The final growth alone copied most of the previous stack, which + // was half of deep.Size, so a quarter of deep.Size is a safe lower + // bound on the bytes copied. + if got, want := copied1-copied0, deep.Size/4; got < want { + t.Errorf("copied bytes increased by %d; want at least %d", got, want) + } +} + +func BenchmarkTailscaleStackHistMetric(b *testing.B) { + for _, idle := range []int{0, 10000} { + b.Run(fmt.Sprintf("idle=%d", idle), func(b *testing.B) { + release := make(chan struct{}) + var wg sync.WaitGroup + for range idle { + wg.Add(1) + go func() { + defer wg.Done() + <-release + }() + } + s := []metrics.Sample{{Name: stackHistMetric}} + b.ResetTimer() + for range b.N { + metrics.Read(s) + } + b.StopTimer() + close(release) + wg.Wait() + }) + } +}