mirror of
https://gitea.com/gitea/act_runner
synced 2026-09-21 19:37:07 +02:00
perf: cut redundant work out of job setup and teardown (#1218)
Implement speedups to job start and shutdown. - Create the job container while services are still becoming healthy, and poll their health at a flat one second instead of a 2s to 32s doubling backoff - Pull each service image once instead of twice, and fetch a warm action cache once instead of twice - Report the job result before reclaiming its volumes, and reap volumes stranded by a runner that died mid-job | Step | Scenario | Before | After | | --- | --- | --- | --- | | Complete job | Large workspace volume | 4.2s | 0.4s | | Set up job | One service, 2s health interval | 7.08s | 3.26s | | Set up job | Two cached actions from github.com | 1.81s | 1.34s | | Set up job | Two cached actions from gitea.com | 2.42s | 1.95s | | Set up job | Two actions, cold action cache | 6.62s | unchanged | | Set up job | Minimal job, no services or actions | 0.62s | unchanged | Assisted-by: Claude Code:Opus 5 Reviewed-on: https://gitea.com/gitea/runner/pulls/1218 Reviewed-by: bircni <bircni@icloud.com> Co-authored-by: silverwind <me@silverwind.io>
This commit is contained in:
@@ -146,6 +146,8 @@ func (r *Runner) Close() error {
|
||||
// removeOrphanNetworks is a variable so tests can substitute one that needs no Docker daemon.
|
||||
var removeOrphanNetworks = container.RemoveOrphanNetworks
|
||||
|
||||
var removeOrphanJobVolumes = container.RemoveOrphanJobVolumes
|
||||
|
||||
// OnIdle performs lightweight maintenance during polling idle windows.
|
||||
// It runs synchronously on the poller goroutine; shouldRunIdleCleanup
|
||||
// throttles invocations to runner.idle_cleanup_interval so the impact on
|
||||
@@ -167,21 +169,20 @@ func (r *Runner) OnIdle(ctx context.Context) {
|
||||
if hostRoot := filepath.FromSlash(r.cfg.Host.WorkdirParent); hostRoot != "" {
|
||||
r.cleanupStaleDirs(ctx, hostRoot, isHostScratchDir)
|
||||
}
|
||||
r.cleanupOrphanNetworks(ctx)
|
||||
r.cleanupOrphanDockerResources(ctx)
|
||||
}
|
||||
|
||||
// cleanupOrphanNetworks reclaims the per-job networks of jobs this runner did not live to
|
||||
// tear down. A labelled network with no containers on it is finished with, and as for the
|
||||
// directories above, a task beginning during the pass is safe because the cutoff keeps a
|
||||
// network it has created but not yet attached a container to out of scope.
|
||||
func (r *Runner) cleanupOrphanNetworks(ctx context.Context) {
|
||||
if r.uuid == "" || !r.requiresDocker() {
|
||||
func (r *Runner) cleanupOrphanDockerResources(ctx context.Context) {
|
||||
if r.uuid == "" || (!r.requiresDocker() && !dockerReachable(ctx)) {
|
||||
return
|
||||
}
|
||||
cutoff := r.now().Add(-r.cfg.Runner.WorkdirCleanupAge)
|
||||
if err := removeOrphanNetworks(ctx, r.uuid, cutoff); err != nil {
|
||||
log.Warnf("failed to clean up networks left behind by earlier jobs: %v", err)
|
||||
}
|
||||
if err := removeOrphanJobVolumes(ctx, r.uuid, cutoff); err != nil {
|
||||
log.Warnf("failed to clean up volumes left behind by earlier jobs: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func (r *Runner) shouldRunIdleCleanup() bool {
|
||||
@@ -287,6 +288,7 @@ func (r *Runner) Run(ctx context.Context, task *runnerv1.Task) error {
|
||||
defer r.runningTasks.Delete(task.Id)
|
||||
|
||||
r.runningCount.Add(1)
|
||||
defer r.runningCount.Add(-1)
|
||||
|
||||
start := time.Now()
|
||||
|
||||
@@ -294,15 +296,25 @@ func (r *Runner) Run(ctx context.Context, task *runnerv1.Task) error {
|
||||
defer cancel()
|
||||
// A proxy URL may carry credentials, and every job is given it; keep them out of the log.
|
||||
reporter := report.NewReporter(ctx, cancel, r.client, task, r.cfg, proxyPasswords()...)
|
||||
var volumeCleanup []common.Executor
|
||||
var volumeCleanupMu sync.Mutex
|
||||
if r.cfg.Runner.PostTaskScript == "" {
|
||||
ctx = runner.WithJobVolumeCleanup(ctx, func(cleanup common.Executor) {
|
||||
volumeCleanupMu.Lock()
|
||||
defer volumeCleanupMu.Unlock()
|
||||
volumeCleanup = append(volumeCleanup, cleanup)
|
||||
})
|
||||
}
|
||||
var runErr error
|
||||
defer func() {
|
||||
r.runningCount.Add(-1)
|
||||
|
||||
lastWords := ""
|
||||
if runErr != nil {
|
||||
lastWords = runErr.Error()
|
||||
}
|
||||
_ = reporter.Close(lastWords)
|
||||
if err := cleanupJobVolumes(ctx, volumeCleanup); err != nil {
|
||||
log.Warnf("task %d volume cleanup after reporting: %v", task.Id, err)
|
||||
}
|
||||
|
||||
metrics.JobDuration.Observe(time.Since(start).Seconds())
|
||||
metrics.JobsTotal.WithLabelValues(metrics.ResultToStatusLabel(reporter.Result())).Inc()
|
||||
@@ -313,6 +325,16 @@ func (r *Runner) Run(ctx context.Context, task *runnerv1.Task) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
func cleanupJobVolumes(ctx context.Context, cleanups []common.Executor) error {
|
||||
ctx, cancel := context.WithTimeout(context.WithoutCancel(ctx), time.Minute)
|
||||
defer cancel()
|
||||
var errs []error
|
||||
for _, cleanup := range cleanups {
|
||||
errs = append(errs, cleanup(ctx))
|
||||
}
|
||||
return errors.Join(errs...)
|
||||
}
|
||||
|
||||
func (r *Runner) cloneEnvs() map[string]string {
|
||||
// Reserve space for the per-task keys injected by run():
|
||||
// ACTIONS_ID_TOKEN_REQUEST_URL, ACTIONS_ID_TOKEN_REQUEST_TOKEN, ACTIONS_RUNTIME_TOKEN,
|
||||
|
||||
Reference in New Issue
Block a user