mirror of
https://github.com/cubefs/cubefs.git
synced 2026-08-02 02:00:56 +00:00
fix(blobnode): fix data inspect goroutine leak
@formatter:off Signed-off-by: mawei029 <mawei2@oppo.com>
This commit is contained in:
parent
4ddfc37d3f
commit
34663a6fd7
@ -28,14 +28,17 @@ const (
|
|||||||
minRateLimit = 64 * 1024 // 64 KB/s
|
minRateLimit = 64 * 1024 // 64 KB/s
|
||||||
)
|
)
|
||||||
|
|
||||||
var dataInspectMetric = prometheus.NewGaugeVec(
|
var (
|
||||||
prometheus.GaugeOpts{
|
dataInspectMetric = prometheus.NewGaugeVec(
|
||||||
Namespace: "blobstore",
|
prometheus.GaugeOpts{
|
||||||
Subsystem: "blobnode",
|
Namespace: "blobstore",
|
||||||
Name: "data_inspect",
|
Subsystem: "blobnode",
|
||||||
Help: "blobnode data inspect",
|
Name: "data_inspect",
|
||||||
},
|
Help: "blobnode data inspect",
|
||||||
[]string{"cluster_id", "idc", "rack", "host", "disk_id", "vuid", "bid", "error"},
|
},
|
||||||
|
[]string{"cluster_id", "idc", "rack", "host", "disk_id", "vuid", "bid", "error"},
|
||||||
|
)
|
||||||
|
errServiceClosed = errors.New("service is closed")
|
||||||
)
|
)
|
||||||
|
|
||||||
type DataInspectConf struct {
|
type DataInspectConf struct {
|
||||||
@ -158,11 +161,6 @@ func (mgr *DataInspectMgr) inspectChunk(pCtx context.Context, cs core.ChunkAPI)
|
|||||||
lmt := mgr.getLimiter(ds)
|
lmt := mgr.getLimiter(ds)
|
||||||
successCnt := int64(0)
|
successCnt := int64(0)
|
||||||
|
|
||||||
go func() {
|
|
||||||
<-mgr.svr.closeCh
|
|
||||||
cancel()
|
|
||||||
}()
|
|
||||||
|
|
||||||
scanFn := func(batchShards []*bnapi.ShardInfo) (err error) {
|
scanFn := func(batchShards []*bnapi.ShardInfo) (err error) {
|
||||||
for _, si := range batchShards {
|
for _, si := range batchShards {
|
||||||
if si.Size <= 0 {
|
if si.Size <= 0 {
|
||||||
@ -173,6 +171,17 @@ func (mgr *DataInspectMgr) inspectChunk(pCtx context.Context, cs core.ChunkAPI)
|
|||||||
shardReader := core.NewShardReader(si.Bid, si.Vuid, 0, 0, discard)
|
shardReader := core.NewShardReader(si.Bid, si.Vuid, 0, 0, discard)
|
||||||
// retry overload
|
// retry overload
|
||||||
for {
|
for {
|
||||||
|
select {
|
||||||
|
case <-pCtx.Done():
|
||||||
|
span.Warnf("inspect chunk stop, upper level has context canceled. vuid:%d,chunkid:%s.", cs.Vuid(), cs.ID())
|
||||||
|
return pCtx.Err()
|
||||||
|
case <-mgr.svr.closeCh:
|
||||||
|
cancel()
|
||||||
|
span.Warnf("inspect chunk stop, service is closed. vuid:%d, chunkid:%s.", cs.Vuid(), cs.ID())
|
||||||
|
return errServiceClosed
|
||||||
|
default:
|
||||||
|
}
|
||||||
|
|
||||||
// Tokens of the corresponding size are obtained based on the size of the shard.
|
// Tokens of the corresponding size are obtained based on the size of the shard.
|
||||||
// If the size of shard is 1MB, you need to get 1024*1024 tokens
|
// If the size of shard is 1MB, you need to get 1024*1024 tokens
|
||||||
remain := si.Size
|
remain := si.Size
|
||||||
@ -205,13 +214,6 @@ func (mgr *DataInspectMgr) inspectChunk(pCtx context.Context, cs core.ChunkAPI)
|
|||||||
span.Errorf("inspect blob error, vuid:%v, bid:%v, err:%v, bad shards:%v", si.Vuid, si.Bid, err, badShards)
|
span.Errorf("inspect blob error, vuid:%v, bid:%v, err:%v, bad shards:%v", si.Vuid, si.Bid, err, badShards)
|
||||||
mgr.reportBadShard(cs, si.Bid, err)
|
mgr.reportBadShard(cs, si.Bid, err)
|
||||||
}
|
}
|
||||||
|
|
||||||
select {
|
|
||||||
case <-ctx.Done():
|
|
||||||
span.Warnf("inspect chunk done.chunkid:%v", cs.ID())
|
|
||||||
return ctx.Err()
|
|
||||||
default:
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|||||||
@ -4,6 +4,7 @@ import (
|
|||||||
"context"
|
"context"
|
||||||
"net/http"
|
"net/http"
|
||||||
"net/http/httptest"
|
"net/http/httptest"
|
||||||
|
"runtime"
|
||||||
"sync"
|
"sync"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
@ -94,20 +95,38 @@ func TestDataInspect(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
{
|
{
|
||||||
|
// inspect single chunk, cancel parent ctx
|
||||||
cs := NewMockChunkAPI(ctr)
|
cs := NewMockChunkAPI(ctr)
|
||||||
cs.EXPECT().Vuid().Return(proto.Vuid(1001)).AnyTimes()
|
cs.EXPECT().Vuid().Return(proto.Vuid(1001)).AnyTimes()
|
||||||
cs.EXPECT().ID().Times(2).Return(clustermgr.ChunkID{})
|
cs.EXPECT().ID().Return(clustermgr.ChunkID{}).AnyTimes()
|
||||||
cs.EXPECT().Disk().Return(ds1)
|
cs.EXPECT().Disk().Return(ds1)
|
||||||
cs.EXPECT().Read(any, any).Return(int64(0), nil)
|
cs.EXPECT().Read(any, any).Return(int64(0), nil).Times(0)
|
||||||
cs.EXPECT().ListShards(any, any, any, any).Return([]*bnapi.ShardInfo{{Bid: 123456, Size: 8}}, proto.BlobID(123456+1), nil)
|
cs.EXPECT().ListShards(any, any, any, any).Return([]*bnapi.ShardInfo{{Bid: 123456, Size: 8}}, proto.BlobID(123456+1), nil)
|
||||||
ds1.EXPECT().ID().Times(1).Return(proto.DiskID(11))
|
ds1.EXPECT().ID().Return(proto.DiskID(11)).AnyTimes()
|
||||||
|
|
||||||
|
pCtx, cancel := context.WithCancel(context.Background())
|
||||||
|
cancel()
|
||||||
|
_, err = mgr.inspectChunk(pCtx, cs)
|
||||||
|
require.NotNil(t, err)
|
||||||
|
require.ErrorIs(t, err, context.Canceled)
|
||||||
|
}
|
||||||
|
|
||||||
|
{
|
||||||
|
// inspect single chunk, closed ctx
|
||||||
|
cs := NewMockChunkAPI(ctr)
|
||||||
|
cs.EXPECT().Vuid().Return(proto.Vuid(1001)).AnyTimes()
|
||||||
|
cs.EXPECT().ID().Return(clustermgr.ChunkID{}).AnyTimes()
|
||||||
|
cs.EXPECT().Disk().Return(ds1)
|
||||||
|
cs.EXPECT().Read(any, any).Return(int64(0), nil).Times(0)
|
||||||
|
cs.EXPECT().ListShards(any, any, any, any).Return([]*bnapi.ShardInfo{{Bid: 123456, Size: 8}}, proto.BlobID(123456+1), nil)
|
||||||
|
ds1.EXPECT().ID().Return(proto.DiskID(11)).AnyTimes()
|
||||||
|
|
||||||
close(mgr.svr.closeCh)
|
close(mgr.svr.closeCh)
|
||||||
mgr.limits[proto.DiskID(11)].SetLimit(2)
|
mgr.limits[proto.DiskID(11)].SetLimit(2)
|
||||||
mgr.limits[proto.DiskID(11)].SetBurst(4)
|
mgr.limits[proto.DiskID(11)].SetBurst(4)
|
||||||
_, err = mgr.inspectChunk(ctx, cs)
|
_, err = mgr.inspectChunk(ctx, cs)
|
||||||
require.NotNil(t, err)
|
require.NotNil(t, err)
|
||||||
require.Equal(t, "context canceled", err.Error())
|
require.ErrorIs(t, err, errServiceClosed)
|
||||||
}
|
}
|
||||||
|
|
||||||
{
|
{
|
||||||
@ -116,3 +135,50 @@ func TestDataInspect(t *testing.T) {
|
|||||||
require.Equal(t, cfg.IntervalSec, mgr.conf.IntervalSec)
|
require.Equal(t, cfg.IntervalSec, mgr.conf.IntervalSec)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func TestInspectChunk_NoGoroutineLeak(t *testing.T) {
|
||||||
|
ctr := gomock.NewController(t)
|
||||||
|
ctx := context.Background()
|
||||||
|
|
||||||
|
// build service and manager
|
||||||
|
ds := NewMockDiskAPI(ctr)
|
||||||
|
svr := &Service{
|
||||||
|
Disks: map[proto.DiskID]core.DiskAPI{11: ds},
|
||||||
|
ctx: context.Background(),
|
||||||
|
closeCh: make(chan struct{}),
|
||||||
|
}
|
||||||
|
getter := mocks.NewMockAccessor(ctr)
|
||||||
|
getter.EXPECT().GetConfig(any, any).AnyTimes().Return("", nil)
|
||||||
|
getter.EXPECT().SetConfig(any, any, any).AnyTimes().Return(nil)
|
||||||
|
switchMgr := taskswitch.NewSwitchMgr(getter)
|
||||||
|
mgr, err := NewDataInspectMgr(svr, DataInspectConf{IntervalSec: 1, RateLimit: 1024 * 1024}, switchMgr)
|
||||||
|
require.NoError(t, err)
|
||||||
|
mgr.svr = svr
|
||||||
|
svr.inspectMgr = mgr
|
||||||
|
|
||||||
|
// limiter entry (avoid nil access if shards present)
|
||||||
|
ds.EXPECT().ID().AnyTimes().Return(proto.DiskID(11))
|
||||||
|
mgr.setLimiters([]core.DiskAPI{ds})
|
||||||
|
|
||||||
|
// chunk mock: empty shard list so inspectChunk returns quickly
|
||||||
|
cs := NewMockChunkAPI(ctr)
|
||||||
|
cs.EXPECT().Vuid().AnyTimes().Return(proto.Vuid(1001))
|
||||||
|
cs.EXPECT().ID().AnyTimes().Return(clustermgr.ChunkID{})
|
||||||
|
cs.EXPECT().Disk().AnyTimes().Return(ds)
|
||||||
|
cs.EXPECT().ListShards(any, any, any, any).AnyTimes().Return([]*bnapi.ShardInfo{}, proto.InValidBlobID, nil)
|
||||||
|
|
||||||
|
before := runtime.NumGoroutine()
|
||||||
|
for i := 0; i < 50; i++ {
|
||||||
|
_, err = mgr.inspectChunk(ctx, cs)
|
||||||
|
require.NoError(t, err)
|
||||||
|
}
|
||||||
|
// allow scheduler to settle
|
||||||
|
// (if a leak existed via a background goroutine, goroutine count would keep growing)
|
||||||
|
// small sleep to stabilize
|
||||||
|
// not too long to avoid slowing CI
|
||||||
|
// 50 iterations are enough to detect growth
|
||||||
|
|
||||||
|
after := runtime.NumGoroutine()
|
||||||
|
// tolerate a small delta for unrelated goroutines
|
||||||
|
require.LessOrEqual(t, after, before+5)
|
||||||
|
}
|
||||||
|
|||||||
Loading…
Reference in New Issue
Block a user