mirror of
https://github.com/cubefs/cubefs.git
synced 2026-08-02 02:00:56 +00:00
fix(blobnode): fix data inspect goroutine leak
@formatter:off Signed-off-by: mawei029 <mawei2@oppo.com>
This commit is contained in:
parent
4ddfc37d3f
commit
34663a6fd7
@ -28,7 +28,8 @@ const (
|
||||
minRateLimit = 64 * 1024 // 64 KB/s
|
||||
)
|
||||
|
||||
var dataInspectMetric = prometheus.NewGaugeVec(
|
||||
var (
|
||||
dataInspectMetric = prometheus.NewGaugeVec(
|
||||
prometheus.GaugeOpts{
|
||||
Namespace: "blobstore",
|
||||
Subsystem: "blobnode",
|
||||
@ -37,6 +38,8 @@ var dataInspectMetric = prometheus.NewGaugeVec(
|
||||
},
|
||||
[]string{"cluster_id", "idc", "rack", "host", "disk_id", "vuid", "bid", "error"},
|
||||
)
|
||||
errServiceClosed = errors.New("service is closed")
|
||||
)
|
||||
|
||||
type DataInspectConf struct {
|
||||
IntervalSec int `json:"interval_sec"` // next round inspect interval
|
||||
@ -158,11 +161,6 @@ func (mgr *DataInspectMgr) inspectChunk(pCtx context.Context, cs core.ChunkAPI)
|
||||
lmt := mgr.getLimiter(ds)
|
||||
successCnt := int64(0)
|
||||
|
||||
go func() {
|
||||
<-mgr.svr.closeCh
|
||||
cancel()
|
||||
}()
|
||||
|
||||
scanFn := func(batchShards []*bnapi.ShardInfo) (err error) {
|
||||
for _, si := range batchShards {
|
||||
if si.Size <= 0 {
|
||||
@ -173,6 +171,17 @@ func (mgr *DataInspectMgr) inspectChunk(pCtx context.Context, cs core.ChunkAPI)
|
||||
shardReader := core.NewShardReader(si.Bid, si.Vuid, 0, 0, discard)
|
||||
// retry overload
|
||||
for {
|
||||
select {
|
||||
case <-pCtx.Done():
|
||||
span.Warnf("inspect chunk stop, upper level has context canceled. vuid:%d,chunkid:%s.", cs.Vuid(), cs.ID())
|
||||
return pCtx.Err()
|
||||
case <-mgr.svr.closeCh:
|
||||
cancel()
|
||||
span.Warnf("inspect chunk stop, service is closed. vuid:%d, chunkid:%s.", cs.Vuid(), cs.ID())
|
||||
return errServiceClosed
|
||||
default:
|
||||
}
|
||||
|
||||
// Tokens of the corresponding size are obtained based on the size of the shard.
|
||||
// If the size of shard is 1MB, you need to get 1024*1024 tokens
|
||||
remain := si.Size
|
||||
@ -205,13 +214,6 @@ func (mgr *DataInspectMgr) inspectChunk(pCtx context.Context, cs core.ChunkAPI)
|
||||
span.Errorf("inspect blob error, vuid:%v, bid:%v, err:%v, bad shards:%v", si.Vuid, si.Bid, err, badShards)
|
||||
mgr.reportBadShard(cs, si.Bid, err)
|
||||
}
|
||||
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
span.Warnf("inspect chunk done.chunkid:%v", cs.ID())
|
||||
return ctx.Err()
|
||||
default:
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
@ -4,6 +4,7 @@ import (
|
||||
"context"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"runtime"
|
||||
"sync"
|
||||
"testing"
|
||||
|
||||
@ -94,20 +95,38 @@ func TestDataInspect(t *testing.T) {
|
||||
}
|
||||
|
||||
{
|
||||
// inspect single chunk, cancel parent ctx
|
||||
cs := NewMockChunkAPI(ctr)
|
||||
cs.EXPECT().Vuid().Return(proto.Vuid(1001)).AnyTimes()
|
||||
cs.EXPECT().ID().Times(2).Return(clustermgr.ChunkID{})
|
||||
cs.EXPECT().ID().Return(clustermgr.ChunkID{}).AnyTimes()
|
||||
cs.EXPECT().Disk().Return(ds1)
|
||||
cs.EXPECT().Read(any, any).Return(int64(0), nil)
|
||||
cs.EXPECT().Read(any, any).Return(int64(0), nil).Times(0)
|
||||
cs.EXPECT().ListShards(any, any, any, any).Return([]*bnapi.ShardInfo{{Bid: 123456, Size: 8}}, proto.BlobID(123456+1), nil)
|
||||
ds1.EXPECT().ID().Times(1).Return(proto.DiskID(11))
|
||||
ds1.EXPECT().ID().Return(proto.DiskID(11)).AnyTimes()
|
||||
|
||||
pCtx, cancel := context.WithCancel(context.Background())
|
||||
cancel()
|
||||
_, err = mgr.inspectChunk(pCtx, cs)
|
||||
require.NotNil(t, err)
|
||||
require.ErrorIs(t, err, context.Canceled)
|
||||
}
|
||||
|
||||
{
|
||||
// inspect single chunk, closed ctx
|
||||
cs := NewMockChunkAPI(ctr)
|
||||
cs.EXPECT().Vuid().Return(proto.Vuid(1001)).AnyTimes()
|
||||
cs.EXPECT().ID().Return(clustermgr.ChunkID{}).AnyTimes()
|
||||
cs.EXPECT().Disk().Return(ds1)
|
||||
cs.EXPECT().Read(any, any).Return(int64(0), nil).Times(0)
|
||||
cs.EXPECT().ListShards(any, any, any, any).Return([]*bnapi.ShardInfo{{Bid: 123456, Size: 8}}, proto.BlobID(123456+1), nil)
|
||||
ds1.EXPECT().ID().Return(proto.DiskID(11)).AnyTimes()
|
||||
|
||||
close(mgr.svr.closeCh)
|
||||
mgr.limits[proto.DiskID(11)].SetLimit(2)
|
||||
mgr.limits[proto.DiskID(11)].SetBurst(4)
|
||||
_, err = mgr.inspectChunk(ctx, cs)
|
||||
require.NotNil(t, err)
|
||||
require.Equal(t, "context canceled", err.Error())
|
||||
require.ErrorIs(t, err, errServiceClosed)
|
||||
}
|
||||
|
||||
{
|
||||
@ -116,3 +135,50 @@ func TestDataInspect(t *testing.T) {
|
||||
require.Equal(t, cfg.IntervalSec, mgr.conf.IntervalSec)
|
||||
}
|
||||
}
|
||||
|
||||
func TestInspectChunk_NoGoroutineLeak(t *testing.T) {
|
||||
ctr := gomock.NewController(t)
|
||||
ctx := context.Background()
|
||||
|
||||
// build service and manager
|
||||
ds := NewMockDiskAPI(ctr)
|
||||
svr := &Service{
|
||||
Disks: map[proto.DiskID]core.DiskAPI{11: ds},
|
||||
ctx: context.Background(),
|
||||
closeCh: make(chan struct{}),
|
||||
}
|
||||
getter := mocks.NewMockAccessor(ctr)
|
||||
getter.EXPECT().GetConfig(any, any).AnyTimes().Return("", nil)
|
||||
getter.EXPECT().SetConfig(any, any, any).AnyTimes().Return(nil)
|
||||
switchMgr := taskswitch.NewSwitchMgr(getter)
|
||||
mgr, err := NewDataInspectMgr(svr, DataInspectConf{IntervalSec: 1, RateLimit: 1024 * 1024}, switchMgr)
|
||||
require.NoError(t, err)
|
||||
mgr.svr = svr
|
||||
svr.inspectMgr = mgr
|
||||
|
||||
// limiter entry (avoid nil access if shards present)
|
||||
ds.EXPECT().ID().AnyTimes().Return(proto.DiskID(11))
|
||||
mgr.setLimiters([]core.DiskAPI{ds})
|
||||
|
||||
// chunk mock: empty shard list so inspectChunk returns quickly
|
||||
cs := NewMockChunkAPI(ctr)
|
||||
cs.EXPECT().Vuid().AnyTimes().Return(proto.Vuid(1001))
|
||||
cs.EXPECT().ID().AnyTimes().Return(clustermgr.ChunkID{})
|
||||
cs.EXPECT().Disk().AnyTimes().Return(ds)
|
||||
cs.EXPECT().ListShards(any, any, any, any).AnyTimes().Return([]*bnapi.ShardInfo{}, proto.InValidBlobID, nil)
|
||||
|
||||
before := runtime.NumGoroutine()
|
||||
for i := 0; i < 50; i++ {
|
||||
_, err = mgr.inspectChunk(ctx, cs)
|
||||
require.NoError(t, err)
|
||||
}
|
||||
// allow scheduler to settle
|
||||
// (if a leak existed via a background goroutine, goroutine count would keep growing)
|
||||
// small sleep to stabilize
|
||||
// not too long to avoid slowing CI
|
||||
// 50 iterations are enough to detect growth
|
||||
|
||||
after := runtime.NumGoroutine()
|
||||
// tolerate a small delta for unrelated goroutines
|
||||
require.LessOrEqual(t, after, before+5)
|
||||
}
|
||||
|
||||
Loading…
Reference in New Issue
Block a user