NOTE
retry
1. What it is A failed remote-service call may succeed when attempted again 2. When retries are needed Network errors can be retried; logic errors do not need retries Long-lasting network failures should not be retried indefinitely Non-idempotent requests cannot be retried 3. Retry strategy 4. Go retry 5. References
This is a historical learning note and may contain outdated or incomplete understanding.
1. What It Is
When a remote-service call fails, calling it again may succeed.
2. When Retries Are Needed
- Network errors can be retried; logic errors do not need retries.
- When a network error persists for a long time, retries should not be used indefinitely, to avoid putting even more pressure on the service.
- Non-idempotent requests cannot be retried.
3. Retry Strategy
- Trade off fast failure against retries.
- If repeated retries still fail, do not keep retrying. See the circuit-breaker pattern.
4. Go Retry
4.1. retry
Equivalent to a Go version of Ribbon: define a retry policy.
const (
defaultDelay = time.Millisecond * 100
defaultMaxDelay = time.Second * 1
defaultAttempts = uint(3)
)
// GetRetryOptions returns retry parameters.
func GetRetryOptions(ctx context.Context, funcName string) []retry.Option {
delay := defaultDelay
maxDelay := defaultMaxDelay
attempts := defaultAttempts
deadline, ok := ctx.Deadline()
// If the upstream caller set a deadline, calculate the retry count from that deadline.
if ok {
remainTime := deadline.Sub(time.Now())
if remainTime < defaultDelay {
return []retry.Option{
retry.Context(ctx),
retry.Attempts(1),
}
}
attempts = uint(remainTime / defaultDelay)
maxDelay = remainTime
}
return []retry.Option{
retry.Context(ctx),
retry.Attempts(attempts),
retry.Delay(delay),
retry.MaxDelay(maxDelay),
retry.RetryIf(func(err error) bool {
return constant.IsNetworkError(err)
}),
retry.OnRetry(func(n uint, err error) {
log.InfoContextf(ctx, "%s request attempt %v, err:%v", funcName, n+1, err)
if err != nil {
metrics.Counter(fmt.Sprintf("%s request attempt %v failed", funcName, n+1)).Incr()
}
}),
}
}
const (
// Success
Suc = 0
// Non-retryable
NonRetry = 1
// Retryable
Retry = 2
)
func TestGetRetryOptions(t *testing.T) {
convey.Convey("Retry", t, func() {
convey.Convey("[Success] request once", func() {
ctx := trpc.BackgroundContext()
requestTimes := 0
err := retry.Do(func() error {
return logic(ctx, Suc, &requestTimes)
}, GetRetryOptions(ctx, "logic")...)
convey.So(requestTimes == 1, convey.ShouldBeTrue)
convey.So(err, convey.ShouldBeNil)
})
convey.Convey("[Failure] non-retryable error - request only once", func() {
ctx := trpc.BackgroundContext()
requestTimes := 0
err := retry.Do(func() error {
return logic(ctx, NonRetry, &requestTimes)
}, GetRetryOptions(ctx, "logic")...)
convey.So(requestTimes == 1, convey.ShouldBeTrue)
convey.So(err, convey.ShouldNotBeNil)
})
convey.Convey("[Failure] retryable error with no upstream timeout - default to three requests", func() {
ctx := trpc.BackgroundContext()
requestTimes := 0
err := retry.Do(func() error {
return logic(ctx, Retry, &requestTimes)
}, GetRetryOptions(ctx, "logic")...)
convey.So(requestTimes == 3, convey.ShouldBeTrue)
convey.So(err, convey.ShouldNotBeNil)
})
convey.Convey("[Failure] retryable error with upstream timeout - calculate request count from upstream time", func() {
ctx, cancel := context.WithTimeout(trpc.BackgroundContext(), time.Millisecond*200)
defer cancel()
requestTimes := 0
err := retry.Do(func() error {
return logic(ctx, Retry, &requestTimes)
}, GetRetryOptions(ctx, "logic")...)
convey.So(requestTimes == 2, convey.ShouldBeTrue)
convey.So(err, convey.ShouldNotBeNil)
})
convey.Convey("[Failure] retryable error with upstream timeout - make at least one request", func() {
ctx, cancel := context.WithTimeout(trpc.BackgroundContext(), time.Millisecond*100)
defer cancel()
requestTimes := 0
err := retry.Do(func() error {
return logic(ctx, Retry, &requestTimes)
}, GetRetryOptions(ctx, "logic")...)
convey.So(requestTimes == 1, convey.ShouldBeTrue)
convey.So(err, convey.ShouldNotBeNil)
})
convey.Convey("[Failure] retryable error with upstream timeout - canceled context returns an error with zero requests", func() {
ctx, cancel := context.WithTimeout(trpc.BackgroundContext(), time.Millisecond*100)
cancel()
requestTimes := 0
err := retry.Do(func() error {
return logic(ctx, Retry, &requestTimes)
}, GetRetryOptions(ctx, "logic")...)
convey.So(requestTimes == 0, convey.ShouldBeTrue)
convey.So(err, convey.ShouldNotBeNil)
})
})
}
func logic(ctx context.Context, t int, requestTimes *int) error {
*requestTimes++
log.InfoContextf(ctx, "doing logic...")
switch t {
case Retry:
return errs.NewFrameError(errs.RetClientNetErr, "retryable error")
case NonRetry:
return errs.New(-1, "non-retryable error")
default:
return nil
}
}
// IsNetworkError reports whether an error is a network error.
func IsNetworkError(e error) bool {
if e == nil {
return false
}
var code int
err, ok := e.(*errs.Error)
if ok {
code = errs.Code(err)
} else {
code = errors.Code(e)
}
return code == errs.RetClientNetErr || code == errs.RetClientTimeout ||
code == errs.RetClientConnectFail || code == errs.RetServerTimeout || code == errs.RetServerSystemErr
}
Discussion
Sign in with GitHub to comment. Discussions are stored as GitHub Issues.View on GitHub