From 81f58f636df6feaf4cecc0d494e1e227667c8d21 Mon Sep 17 00:00:00 2001 From: comonad Date: Thu, 25 Jun 2026 22:57:28 +0800 Subject: [PATCH 01/15] feat(nix): init gomod2nix --- .envrc | 6 ++ .gitignore | 5 ++ flake.lock | 204 +++++++++++++++++++++++++++++++++++++++++++++++++ flake.nix | 116 ++++++++++++++++++++++++++++ go.mod | 45 +++++------ go.sum | 95 ++++++++++++----------- gomod2nix.toml | 122 +++++++++++++++++++++++++++++ package.nix | 11 +++ shell.nix | 12 +++ 9 files changed, 547 insertions(+), 69 deletions(-) create mode 100644 .envrc create mode 100644 flake.lock create mode 100644 flake.nix create mode 100644 gomod2nix.toml create mode 100644 package.nix create mode 100644 shell.nix diff --git a/.envrc b/.envrc new file mode 100644 index 0000000..d5359eb --- /dev/null +++ b/.envrc @@ -0,0 +1,6 @@ +# shellcheck shell=bash +# ^ make editor happy + +export GO111MODULE="on" +export GOPROXY="https://goproxy.cn" +use flake . -Lv --fallback --show-trace diff --git a/.gitignore b/.gitignore index 506db39..f491022 100644 --- a/.gitignore +++ b/.gitignore @@ -35,3 +35,8 @@ _testmain.go /Caddyfile /gomplate* /_vendor* + +/.direnv +/.pre-commit-config.flake.yaml + + diff --git a/flake.lock b/flake.lock new file mode 100644 index 0000000..3648c61 --- /dev/null +++ b/flake.lock @@ -0,0 +1,204 @@ +{ + "nodes": { + "flake-compat": { + "flake": false, + "locked": { + "lastModified": 1767039857, + "narHash": "sha256-vNpUSpF5Nuw8xvDLj2KCwwksIbjua2LZCqhV1LNRDns=", + "owner": "NixOS", + "repo": "flake-compat", + "rev": "5edf11c44bc78a0d334f6334cdaf7d60d732daab", + "type": "github" + }, + "original": { + "owner": "NixOS", + "repo": "flake-compat", + "type": "github" + } + }, + "flake-compat_2": { + "flake": false, + "locked": { + "lastModified": 1767039857, + "narHash": "sha256-vNpUSpF5Nuw8xvDLj2KCwwksIbjua2LZCqhV1LNRDns=", + "owner": "NixOS", + "repo": "flake-compat", + "rev": "5edf11c44bc78a0d334f6334cdaf7d60d732daab", + "type": "github" + }, + "original": { + "owner": "NixOS", + "repo": "flake-compat", + "type": "github" + } + }, + "flake-parts": { + "inputs": { + "nixpkgs-lib": [ + "nixpkgs" + ] + }, + "locked": { + "lastModified": 1778716662, + "narHash": "sha256-m1Yf0wZ8j1OHjTc2UwHwyQRSnNeSgLJOd7q5Y45hzi4=", + "owner": "hercules-ci", + "repo": "flake-parts", + "rev": "f7c1a2d347e4c52d5fb8d10cb4d94b5884e546fb", + "type": "github" + }, + "original": { + "owner": "hercules-ci", + "repo": "flake-parts", + "type": "github" + } + }, + "flake-utils": { + "inputs": { + "systems": "systems" + }, + "locked": { + "lastModified": 1731533236, + "narHash": "sha256-l0KFg5HjrsfsO/JpG+r7fRrqm12kzFHyUHqHCVpMMbI=", + "owner": "numtide", + "repo": "flake-utils", + "rev": "11707dc2f618dd54ca8739b309ec4fc024de578b", + "type": "github" + }, + "original": { + "owner": "numtide", + "repo": "flake-utils", + "type": "github" + } + }, + "gitignore": { + "inputs": { + "nixpkgs": [ + "pre-commit-hooks", + "nixpkgs" + ] + }, + "locked": { + "lastModified": 1709087332, + "narHash": "sha256-HG2cCnktfHsKV0s4XW83gU3F57gaTljL9KNSuG6bnQs=", + "owner": "hercules-ci", + "repo": "gitignore.nix", + "rev": "637db329424fd7e46cf4185293b9cc8c88c95394", + "type": "github" + }, + "original": { + "owner": "hercules-ci", + "repo": "gitignore.nix", + "type": "github" + } + }, + "gomod2nix": { + "inputs": { + "flake-utils": [ + "flake-utils" + ], + "nixpkgs": [ + "nixpkgs" + ] + }, + "locked": { + "lastModified": 1770585520, + "narHash": "sha256-yBz9Ozd5Wb56i3e3cHZ8WcbzCQ9RlVaiW18qDYA/AzA=", + "owner": "nix-community", + "repo": "gomod2nix", + "rev": "1201ddd1279c35497754f016ef33d5e060f3da8d", + "type": "github" + }, + "original": { + "owner": "nix-community", + "repo": "gomod2nix", + "type": "github" + } + }, + "nixpkgs": { + "locked": { + "lastModified": 1781607440, + "narHash": "sha256-rxO+uc/KFbSJp+pgyXRuAX6QlG9hJdnt0BXpEQRXY+U=", + "owner": "NixOS", + "repo": "nixpkgs", + "rev": "3e41b24abd260e8f71dbe2f5737d24122f972158", + "type": "github" + }, + "original": { + "owner": "NixOS", + "ref": "nixpkgs-unstable", + "repo": "nixpkgs", + "type": "github" + } + }, + "pre-commit-hooks": { + "inputs": { + "flake-compat": "flake-compat_2", + "gitignore": "gitignore", + "nixpkgs": [ + "nixpkgs" + ] + }, + "locked": { + "lastModified": 1781733627, + "narHash": "sha256-U3yTuGBnmXvXoQI3qkpfEDsn9RovQPAjN7ndRco+3u0=", + "owner": "cachix", + "repo": "git-hooks.nix", + "rev": "3bbec39bc90eadfa031e6f3b77272f3f60803e39", + "type": "github" + }, + "original": { + "owner": "cachix", + "repo": "git-hooks.nix", + "type": "github" + } + }, + "root": { + "inputs": { + "flake-compat": "flake-compat", + "flake-parts": "flake-parts", + "flake-utils": "flake-utils", + "gomod2nix": "gomod2nix", + "nixpkgs": "nixpkgs", + "pre-commit-hooks": "pre-commit-hooks", + "treefmt-nix": "treefmt-nix" + } + }, + "systems": { + "locked": { + "lastModified": 1681028828, + "narHash": "sha256-Vy1rq5AaRuLzOxct8nz4T6wlgyUR7zLU309k9mBC768=", + "owner": "nix-systems", + "repo": "default", + "rev": "da67096a3b9bf56a91d16901293e51ba5b49a27e", + "type": "github" + }, + "original": { + "owner": "nix-systems", + "repo": "default", + "type": "github" + } + }, + "treefmt-nix": { + "inputs": { + "nixpkgs": [ + "nixpkgs" + ] + }, + "locked": { + "lastModified": 1780220602, + "narHash": "sha256-eynAfOmbmxJnkp7YewvCEbShNnnYJ9gLLqkzsYtBPeM=", + "owner": "numtide", + "repo": "treefmt-nix", + "rev": "db947814a175b7ca6ded66e21383d938df01c227", + "type": "github" + }, + "original": { + "owner": "numtide", + "repo": "treefmt-nix", + "type": "github" + } + } + }, + "root": "root", + "version": 7 +} diff --git a/flake.nix b/flake.nix new file mode 100644 index 0000000..aca92ca --- /dev/null +++ b/flake.nix @@ -0,0 +1,116 @@ +{ + description = "Go development environment"; + + inputs = { + nixpkgs.url = "github:NixOS/nixpkgs/nixpkgs-unstable"; + flake-parts = { + url = "github:hercules-ci/flake-parts"; + inputs.nixpkgs-lib.follows = "nixpkgs"; + }; + flake-utils.url = "github:numtide/flake-utils"; + gomod2nix = { + url = "github:nix-community/gomod2nix"; + inputs.nixpkgs.follows = "nixpkgs"; + inputs.flake-utils.follows = "flake-utils"; + }; + flake-compat = { + url = "github:NixOS/flake-compat"; + flake = false; + }; + pre-commit-hooks = { + url = "github:cachix/git-hooks.nix"; + inputs.nixpkgs.follows = "nixpkgs"; + }; + treefmt-nix = { + url = "github:numtide/treefmt-nix"; + inputs.nixpkgs.follows = "nixpkgs"; + }; + }; + + outputs = + inputs@{ flake-parts, ... }: + flake-parts.lib.mkFlake { inherit inputs; } { + imports = [ + inputs.treefmt-nix.flakeModule + inputs.pre-commit-hooks.flakeModule + ]; + + systems = [ + "x86_64-linux" + "aarch64-linux" + "aarch64-darwin" + ]; + + perSystem = + { + config, + pkgs, + lib, + system, + ... + }: + let + goVersion = "1.25"; + in + { + _module.args.pkgs = import inputs.nixpkgs { + inherit system; + overlays = [ + inputs.gomod2nix.overlays.default + (_final: prev: { + go = prev."go_${lib.versions.major goVersion}_${lib.versions.minor goVersion}"; + }) + ]; + }; + + # https://flake.parts/options/treefmt-nix.html + # Example: https://github.com/nix-community/buildbot-nix/blob/main/nix/treefmt/flake-module.nix + treefmt = { + projectRootFile = "flake.nix"; + settings.global.excludes = [ ]; + + programs = { + goimports.enable = true; + nixfmt.enable = true; + }; + }; + + # https://flake.parts/options/git-hooks-nix.html + # Example: https://github.com/cachix/git-hooks.nix/blob/master/template/flake.nix + pre-commit.settings.package = pkgs.prek; + pre-commit.settings.configPath = ".pre-commit-config.flake.yaml"; + pre-commit.settings.hooks = { + commitizen.enable = true; + golangci-lint.enable = true; + treefmt.enable = true; + }; + + packages.default = pkgs.callPackage ./package.nix { + inherit (inputs.gomod2nix.legacyPackages.${system}) buildGoApplication; + }; + + devShells.default = pkgs.mkShell { + inputsFrom = [ + config.treefmt.build.devShell + config.pre-commit.devShell + ]; + + shellHook = '' + echo 1>&2 "Welcome to the development shell!" + ''; + + packages = with pkgs; [ + (mkGoEnv { pwd = ./.; }) + gopls + gomod2nix + gotools + go-mod-upgrade + govulncheck + go-junit-report + go-task + delve + ]; + }; + }; + }; +} diff --git a/go.mod b/go.mod index a79af1a..99a5e2c 100644 --- a/go.mod +++ b/go.mod @@ -1,42 +1,39 @@ module github.com/sjtug/lug -go 1.23.0 - -toolchain go1.23.7 +go 1.25.0 require ( github.com/ant0ine/go-json-rest v3.3.2+incompatible github.com/cheshir/logrustash v0.0.0-20230213210745-aca6961b250d github.com/davecgh/go-spew v1.1.1 github.com/dustin/go-humanize v1.0.1 - github.com/prometheus/client_golang v1.21.1 - github.com/sirupsen/logrus v1.9.3 - github.com/spf13/pflag v1.0.6 - github.com/spf13/viper v1.20.0 - github.com/stretchr/testify v1.10.0 - mvdan.cc/sh/v3 v3.11.0 + github.com/prometheus/client_golang v1.23.2 + github.com/sirupsen/logrus v1.9.4 + github.com/spf13/pflag v1.0.10 + github.com/spf13/viper v1.21.0 + github.com/stretchr/testify v1.11.1 + mvdan.cc/sh/v3 v3.13.1 ) require ( github.com/beorn7/perks v1.0.1 // indirect github.com/cespare/xxhash/v2 v2.3.0 // indirect - github.com/fsnotify/fsnotify v1.8.0 // indirect - github.com/go-viper/mapstructure/v2 v2.2.1 // indirect - github.com/klauspost/compress v1.18.0 // indirect + github.com/fsnotify/fsnotify v1.10.1 // indirect + github.com/go-viper/mapstructure/v2 v2.5.0 // indirect github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822 // indirect - github.com/pelletier/go-toml/v2 v2.2.3 // indirect + github.com/pelletier/go-toml/v2 v2.4.2 // indirect github.com/pmezard/go-difflib v1.0.0 // indirect - github.com/prometheus/client_model v0.6.1 // indirect - github.com/prometheus/common v0.63.0 // indirect - github.com/prometheus/procfs v0.16.0 // indirect - github.com/sagikazarmark/locafero v0.8.0 // indirect - github.com/sourcegraph/conc v0.3.0 // indirect - github.com/spf13/afero v1.14.0 // indirect - github.com/spf13/cast v1.7.1 // indirect + github.com/prometheus/client_model v0.6.2 // indirect + github.com/prometheus/common v0.69.0 // indirect + github.com/prometheus/procfs v0.20.1 // indirect + github.com/sagikazarmark/locafero v0.12.0 // indirect + github.com/spf13/afero v1.15.0 // indirect + github.com/spf13/cast v1.10.0 // indirect github.com/subosito/gotenv v1.6.0 // indirect - go.uber.org/multierr v1.11.0 // indirect - golang.org/x/sys v0.31.0 // indirect - golang.org/x/text v0.23.0 // indirect - google.golang.org/protobuf v1.36.5 // indirect + go.yaml.in/yaml/v2 v2.4.4 // indirect + go.yaml.in/yaml/v3 v3.0.4 // indirect + golang.org/x/sys v0.46.0 // indirect + golang.org/x/text v0.38.0 // indirect + google.golang.org/protobuf v1.36.11 // indirect gopkg.in/yaml.v3 v3.0.1 // indirect ) diff --git a/go.sum b/go.sum index 57212ed..33d553c 100644 --- a/go.sum +++ b/go.sum @@ -6,19 +6,22 @@ github.com/cespare/xxhash/v2 v2.3.0 h1:UL815xU9SqsFlibzuggzjXhog7bL6oX9BbNZnL2UF github.com/cespare/xxhash/v2 v2.3.0/go.mod h1:VGX0DQ3Q6kWi7AoAeZDth3/j3BFtOZR5XLFGgcrjCOs= github.com/cheshir/logrustash v0.0.0-20230213210745-aca6961b250d h1:d/UzZmpXS1dZvb90oSdCT8BnbGlbzo6+9JM44zJyzyc= github.com/cheshir/logrustash v0.0.0-20230213210745-aca6961b250d/go.mod h1:J+idqV/m19ccuMARuOeEJ9KeZS0Vdrwqv3gNx195CBg= -github.com/davecgh/go-spew v1.1.0/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38= github.com/davecgh/go-spew v1.1.1 h1:vj9j/u1bqnvCEfJOwUhtlOARqs3+rkHYY13jYWTU97c= github.com/davecgh/go-spew v1.1.1/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38= github.com/dustin/go-humanize v1.0.1 h1:GzkhY7T5VNhEkwH0PVJgjz+fX1rhBrR7pRT3mDkpeCY= github.com/dustin/go-humanize v1.0.1/go.mod h1:Mu1zIs6XwVuF/gI1OepvI0qD18qycQx+mFykh5fBlto= github.com/frankban/quicktest v1.14.6 h1:7Xjx+VpznH+oBnejlPUj8oUpdxnVs4f8XU8WnHkI4W8= github.com/frankban/quicktest v1.14.6/go.mod h1:4ptaffx2x8+WTWXmUCuVU6aPUX1/Mz7zb5vbUoiM6w0= -github.com/fsnotify/fsnotify v1.8.0 h1:dAwr6QBTBZIkG8roQaJjGof0pp0EeF+tNV7YBP3F/8M= -github.com/fsnotify/fsnotify v1.8.0/go.mod h1:8jBTzvmWwFyi3Pb8djgCCO5IBqzKJ/Jwo8TRcHyHii0= +github.com/fsnotify/fsnotify v1.9.0 h1:2Ml+OJNzbYCTzsxtv8vKSFD9PbJjmhYF14k/jKC7S9k= +github.com/fsnotify/fsnotify v1.9.0/go.mod h1:8jBTzvmWwFyi3Pb8djgCCO5IBqzKJ/Jwo8TRcHyHii0= +github.com/fsnotify/fsnotify v1.10.1 h1:b0/UzAf9yR5rhf3RPm9gf3ehBPpf0oZKIjtpKrx59Ho= +github.com/fsnotify/fsnotify v1.10.1/go.mod h1:TLheqan6HD6GBK6PrDWyDPBaEV8LspOxvPSjC+bVfgo= github.com/go-quicktest/qt v1.101.0 h1:O1K29Txy5P2OK0dGo59b7b0LR6wKfIhttaAhHUyn7eI= github.com/go-quicktest/qt v1.101.0/go.mod h1:14Bz/f7NwaXPtdYEgzsx46kqSxVwTbzVZsDC26tQJow= -github.com/go-viper/mapstructure/v2 v2.2.1 h1:ZAaOCxANMuZx5RCeg0mBdEZk7DZasvvZIxtHqx8aGss= -github.com/go-viper/mapstructure/v2 v2.2.1/go.mod h1:oJDH3BJKyqBA2TXFhDsKDGDTlndYOZ6rGS0BRZIxGhM= +github.com/go-viper/mapstructure/v2 v2.4.0 h1:EBsztssimR/CONLSZZ04E8qAkxNYq4Qp9LvH92wZUgs= +github.com/go-viper/mapstructure/v2 v2.4.0/go.mod h1:oJDH3BJKyqBA2TXFhDsKDGDTlndYOZ6rGS0BRZIxGhM= +github.com/go-viper/mapstructure/v2 v2.5.0 h1:vM5IJoUAy3d7zRSVtIwQgBj7BiWtMPfmPEgAXnvj1Ro= +github.com/go-viper/mapstructure/v2 v2.5.0/go.mod h1:oJDH3BJKyqBA2TXFhDsKDGDTlndYOZ6rGS0BRZIxGhM= github.com/google/go-cmp v0.7.0 h1:wk8382ETsv4JYUZwIsn6YpYiWiBsYLSJiTsyBybVuN8= github.com/google/go-cmp v0.7.0/go.mod h1:pXiqmnSA92OHEEa9HXL2W4E7lf9JzCmGVUdgjX3N/iU= github.com/klauspost/compress v1.18.0 h1:c/Cqfb0r+Yi+JtIEq73FWXVkRonBlf0CRNYc8Zttxdo= @@ -31,54 +34,56 @@ github.com/kylelemons/godebug v1.1.0 h1:RPNrshWIDI6G2gRW9EHilWtl7Z6Sb1BR0xunSBf0 github.com/kylelemons/godebug v1.1.0/go.mod h1:9/0rRGxNHcop5bhtWyNeEfOS8JIWk580+fNqagV/RAw= github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822 h1:C3w9PqII01/Oq1c1nUAm88MOHcQC9l5mIlSMApZMrHA= github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822/go.mod h1:+n7T8mK8HuQTcFwEeznm/DIxMOiR9yIdICNftLE1DvQ= -github.com/pelletier/go-toml/v2 v2.2.3 h1:YmeHyLY8mFWbdkNWwpr+qIL2bEqT0o95WSdkNHvL12M= -github.com/pelletier/go-toml/v2 v2.2.3/go.mod h1:MfCQTFTvCcUyyvvwm1+G6H/jORL20Xlb6rzQu9GuUkc= +github.com/pelletier/go-toml/v2 v2.2.4 h1:mye9XuhQ6gvn5h28+VilKrrPoQVanw5PMw/TB0t5Ec4= +github.com/pelletier/go-toml/v2 v2.2.4/go.mod h1:2gIqNv+qfxSVS7cM2xJQKtLSTLUE9V8t9Stt+h56mCY= +github.com/pelletier/go-toml/v2 v2.4.2 h1:M2fKKbmyvI+hGId/D0W64qDBMVhJnNR10O5gIbMc//Q= +github.com/pelletier/go-toml/v2 v2.4.2/go.mod h1:2gIqNv+qfxSVS7cM2xJQKtLSTLUE9V8t9Stt+h56mCY= github.com/pmezard/go-difflib v1.0.0 h1:4DBwDE0NGyQoBHbLQYPwSUPoCMWR5BEzIk/f1lZbAQM= github.com/pmezard/go-difflib v1.0.0/go.mod h1:iKH77koFhYxTK1pcRnkKkqfTogsbg7gZNVY4sRDYZ/4= -github.com/prometheus/client_golang v1.21.1 h1:DOvXXTqVzvkIewV/CDPFdejpMCGeMcbGCQ8YOmu+Ibk= -github.com/prometheus/client_golang v1.21.1/go.mod h1:U9NM32ykUErtVBxdvD3zfi+EuFkkaBvMb09mIfe0Zgg= -github.com/prometheus/client_model v0.6.1 h1:ZKSh/rekM+n3CeS952MLRAdFwIKqeY8b62p8ais2e9E= -github.com/prometheus/client_model v0.6.1/go.mod h1:OrxVMOVHjw3lKMa8+x6HeMGkHMQyHDk9E3jmP2AmGiY= -github.com/prometheus/common v0.63.0 h1:YR/EIY1o3mEFP/kZCD7iDMnLPlGyuU2Gb3HIcXnA98k= -github.com/prometheus/common v0.63.0/go.mod h1:VVFF/fBIoToEnWRVkYoXEkq3R3paCoxG9PXP74SnV18= -github.com/prometheus/procfs v0.16.0 h1:xh6oHhKwnOJKMYiYBDWmkHqQPyiY40sny36Cmx2bbsM= -github.com/prometheus/procfs v0.16.0/go.mod h1:8veyXUu3nGP7oaCxhX6yeaM5u4stL2FeMXnCqhDthZg= +github.com/prometheus/client_golang v1.23.2 h1:Je96obch5RDVy3FDMndoUsjAhG5Edi49h0RJWRi/o0o= +github.com/prometheus/client_golang v1.23.2/go.mod h1:Tb1a6LWHB3/SPIzCoaDXI4I8UHKeFTEQ1YCr+0Gyqmg= +github.com/prometheus/client_model v0.6.2 h1:oBsgwpGs7iVziMvrGhE53c/GrLUsZdHnqNwqPLxwZyk= +github.com/prometheus/client_model v0.6.2/go.mod h1:y3m2F6Gdpfy6Ut/GBsUqTWZqCUvMVzSfMLjcu6wAwpE= +github.com/prometheus/common v0.66.1 h1:h5E0h5/Y8niHc5DlaLlWLArTQI7tMrsfQjHV+d9ZoGs= +github.com/prometheus/common v0.66.1/go.mod h1:gcaUsgf3KfRSwHY4dIMXLPV0K/Wg1oZ8+SbZk/HH/dA= +github.com/prometheus/common v0.69.0 h1:OA85nJQS/T/MaYh/Q2CcgDKSGWqNIgrBDvDH85CuiNk= +github.com/prometheus/common v0.69.0/go.mod h1:ZzL3f6u94qUxh9p+tJTrF+FvBS1XXbbRAZCQkytAL0Y= +github.com/prometheus/procfs v0.20.1 h1:XwbrGOIplXW/AU3YhIhLODXMJYyC1isLFfYCsTEycfc= +github.com/prometheus/procfs v0.20.1/go.mod h1:o9EMBZGRyvDrSPH1RqdxhojkuXstoe4UlK79eF5TGGo= github.com/rogpeppe/go-internal v1.14.1 h1:UQB4HGPB6osV0SQTLymcB4TgvyWu6ZyliaW0tI/otEQ= github.com/rogpeppe/go-internal v1.14.1/go.mod h1:MaRKkUm5W0goXpeCfT7UZI6fk/L7L7so1lCWt35ZSgc= -github.com/sagikazarmark/locafero v0.8.0 h1:mXaMVw7IqxNBxfv3LdWt9MDmcWDQ1fagDH918lOdVaQ= -github.com/sagikazarmark/locafero v0.8.0/go.mod h1:UBUyz37V+EdMS3hDF3QWIiVr/2dPrx49OMO0Bn0hJqk= -github.com/sirupsen/logrus v1.9.3 h1:dueUQJ1C2q9oE3F7wvmSGAaVtTmUizReu6fjN8uqzbQ= -github.com/sirupsen/logrus v1.9.3/go.mod h1:naHLuLoDiP4jHNo9R0sCBMtWGeIprob74mVsIT4qYEQ= -github.com/sourcegraph/conc v0.3.0 h1:OQTbbt6P72L20UqAkXXuLOj79LfEanQ+YQFNpLA9ySo= -github.com/sourcegraph/conc v0.3.0/go.mod h1:Sdozi7LEKbFPqYX2/J+iBAM6HpqSLTASQIKqDmF7Mt0= -github.com/spf13/afero v1.14.0 h1:9tH6MapGnn/j0eb0yIXiLjERO8RB6xIVZRDCX7PtqWA= -github.com/spf13/afero v1.14.0/go.mod h1:acJQ8t0ohCGuMN3O+Pv0V0hgMxNYDlvdk+VTfyZmbYo= -github.com/spf13/cast v1.7.1 h1:cuNEagBQEHWN1FnbGEjCXL2szYEXqfJPbP2HNUaca9Y= -github.com/spf13/cast v1.7.1/go.mod h1:ancEpBxwJDODSW/UG4rDrAqiKolqNNh2DX3mk86cAdo= -github.com/spf13/pflag v1.0.6 h1:jFzHGLGAlb3ruxLB8MhbI6A8+AQX/2eW4qeyNZXNp2o= -github.com/spf13/pflag v1.0.6/go.mod h1:McXfInJRrz4CZXVZOBLb0bTZqETkiAhM9Iw0y3An2Bg= -github.com/spf13/viper v1.20.0 h1:zrxIyR3RQIOsarIrgL8+sAvALXul9jeEPa06Y0Ph6vY= -github.com/spf13/viper v1.20.0/go.mod h1:P9Mdzt1zoHIG8m2eZQinpiBjo6kCmZSKBClNNqjJvu4= -github.com/stretchr/objx v0.1.0/go.mod h1:HFkY916IF+rwdDfMAkV7OtwuqBVzrE8GR6GFx+wExME= -github.com/stretchr/testify v1.7.0/go.mod h1:6Fq8oRcR53rry900zMqJjRRixrwX3KX962/h/Wwjteg= -github.com/stretchr/testify v1.10.0 h1:Xv5erBjTwe/5IxqUQTdXv5kgmIvbHo3QQyRwhJsOfJA= -github.com/stretchr/testify v1.10.0/go.mod h1:r2ic/lqez/lEtzL7wO/rwa5dbSLXVDPFyf8C91i36aY= +github.com/sagikazarmark/locafero v0.12.0 h1:/NQhBAkUb4+fH1jivKHWusDYFjMOOKU88eegjfxfHb4= +github.com/sagikazarmark/locafero v0.12.0/go.mod h1:sZh36u/YSZ918v0Io+U9ogLYQJ9tLLBmM4eneO6WwsI= +github.com/sirupsen/logrus v1.9.4 h1:TsZE7l11zFCLZnZ+teH4Umoq5BhEIfIzfRDZ1Uzql2w= +github.com/sirupsen/logrus v1.9.4/go.mod h1:ftWc9WdOfJ0a92nsE2jF5u5ZwH8Bv2zdeOC42RjbV2g= +github.com/spf13/afero v1.15.0 h1:b/YBCLWAJdFWJTN9cLhiXXcD7mzKn9Dm86dNnfyQw1I= +github.com/spf13/afero v1.15.0/go.mod h1:NC2ByUVxtQs4b3sIUphxK0NioZnmxgyCrfzeuq8lxMg= +github.com/spf13/cast v1.10.0 h1:h2x0u2shc1QuLHfxi+cTJvs30+ZAHOGRic8uyGTDWxY= +github.com/spf13/cast v1.10.0/go.mod h1:jNfB8QC9IA6ZuY2ZjDp0KtFO2LZZlg4S/7bzP6qqeHo= +github.com/spf13/pflag v1.0.10 h1:4EBh2KAYBwaONj6b2Ye1GiHfwjqyROoF4RwYO+vPwFk= +github.com/spf13/pflag v1.0.10/go.mod h1:McXfInJRrz4CZXVZOBLb0bTZqETkiAhM9Iw0y3An2Bg= +github.com/spf13/viper v1.21.0 h1:x5S+0EU27Lbphp4UKm1C+1oQO+rKx36vfCoaVebLFSU= +github.com/spf13/viper v1.21.0/go.mod h1:P0lhsswPGWD/1lZJ9ny3fYnVqxiegrlNrEmgLjbTCAY= +github.com/stretchr/testify v1.11.1 h1:7s2iGBzp5EwR7/aIZr8ao5+dra3wiQyKjjFuvgVKu7U= +github.com/stretchr/testify v1.11.1/go.mod h1:wZwfW3scLgRK+23gO65QZefKpKQRnfz6sD981Nm4B6U= github.com/subosito/gotenv v1.6.0 h1:9NlTDc1FTs4qu0DDq7AEtTPNw6SVm7uBMsUCUjABIf8= github.com/subosito/gotenv v1.6.0/go.mod h1:Dk4QP5c2W3ibzajGcXpNraDfq2IrhjMIvMSWPKKo0FU= -go.uber.org/multierr v1.11.0 h1:blXXJkSxSSfBVBlC76pxqeO+LN3aDfLQo+309xJstO0= -go.uber.org/multierr v1.11.0/go.mod h1:20+QtiLqy0Nd6FdQB9TLXag12DsQkrbs3htMFfDN80Y= -golang.org/x/sys v0.0.0-20220715151400-c0bba94af5f8/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= -golang.org/x/sys v0.31.0 h1:ioabZlmFYtWhL+TRYpcnNlLwhyxaM9kWTDEmfnprqik= -golang.org/x/sys v0.31.0/go.mod h1:BJP2sWEmIv4KK5OTEluFJCKSidICx8ciO85XgH3Ak8k= -golang.org/x/text v0.23.0 h1:D71I7dUrlY+VX0gQShAThNGHFxZ13dGLBHQLVl1mJlY= -golang.org/x/text v0.23.0/go.mod h1:/BLNzu4aZCJ1+kcD0DNRotWKage4q2rGVAg4o22unh4= -google.golang.org/protobuf v1.36.5 h1:tPhr+woSbjfYvY6/GPufUoYizxw1cF/yFoxJ2fmpwlM= -google.golang.org/protobuf v1.36.5/go.mod h1:9fA7Ob0pmnwhb644+1+CVWFRbNajQ6iRojtC/QF5bRE= +go.uber.org/goleak v1.3.0 h1:2K3zAYmnTNqV73imy9J1T3WC+gmCePx2hEGkimedGto= +go.uber.org/goleak v1.3.0/go.mod h1:CoHD4mav9JJNrW/WLlf7HGZPjdw8EucARQHekz1X6bE= +go.yaml.in/yaml/v2 v2.4.4 h1:tuyd0P+2Ont/d6e2rl3be67goVK4R6deVxCUX5vyPaQ= +go.yaml.in/yaml/v2 v2.4.4/go.mod h1:gMZqIpDtDqOfM0uNfy0SkpRhvUryYH0Z6wdMYcacYXQ= +go.yaml.in/yaml/v3 v3.0.4 h1:tfq32ie2Jv2UxXFdLJdh3jXuOzWiL1fo0bu/FbuKpbc= +go.yaml.in/yaml/v3 v3.0.4/go.mod h1:DhzuOOF2ATzADvBadXxruRBLzYTpT36CKvDb3+aBEFg= +golang.org/x/sys v0.46.0 h1:noSf2Fq6F8DBgS+LysIkx7rIExoNHJsxOAtPp4rthXw= +golang.org/x/sys v0.46.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw= +golang.org/x/text v0.38.0 h1:sXmwo9DwP3OK9EZ7PqAdaooSGozfl/3a6/xJcbzPRhE= +golang.org/x/text v0.38.0/go.mod h1:YXZt3QhHUKYT53r2lLKFIVi6Ao1jdzrTR/KQ09qyxF4= +google.golang.org/protobuf v1.36.11 h1:fV6ZwhNocDyBLK0dj+fg8ektcVegBBuEolpbTQyBNVE= +google.golang.org/protobuf v1.36.11/go.mod h1:HTf+CrKn2C3g5S8VImy6tdcUvCska2kB7j23XfzDpco= gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0= gopkg.in/check.v1 v1.0.0-20201130134442-10cb98267c6c h1:Hei/4ADfdWqJk1ZMxUNpqntNwaWcugrBjAiHlqqRiVk= gopkg.in/check.v1 v1.0.0-20201130134442-10cb98267c6c/go.mod h1:JHkPIbrfpd72SG/EVd6muEfDQjcINNoR0C8j2r3qZ4Q= -gopkg.in/yaml.v3 v3.0.0-20200313102051-9f266ea9e77c/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM= gopkg.in/yaml.v3 v3.0.1 h1:fxVm/GzAzEWqLHuvctI91KS9hhNmmWOoWu0XTYJS7CA= gopkg.in/yaml.v3 v3.0.1/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM= -mvdan.cc/sh/v3 v3.11.0 h1:q5h+XMDRfUGUedCqFFsjoFjrhwf2Mvtt1rkMvVz0blw= -mvdan.cc/sh/v3 v3.11.0/go.mod h1:LRM+1NjoYCzuq/WZ6y44x14YNAI0NK7FLPeQSaFagGg= +mvdan.cc/sh/v3 v3.13.1 h1:DP3TfgZhDkT7lerUdnp6PTGKyxxzz6T+cOlY/xEvfWk= +mvdan.cc/sh/v3 v3.13.1/go.mod h1:lXJ8SexMvEVcHCoDvAGLZgFJ9Wsm2sulmoNEXGhYZD0= diff --git a/gomod2nix.toml b/gomod2nix.toml new file mode 100644 index 0000000..8954af3 --- /dev/null +++ b/gomod2nix.toml @@ -0,0 +1,122 @@ +schema = 3 + +[mod] + [mod.'github.com/ant0ine/go-json-rest'] + version = 'v3.3.2+incompatible' + hash = 'sha256-yPHFG+m81N2lHr2KzGpR5utbzZLZxRfMYwJMWw/+aXc=' + + [mod.'github.com/beorn7/perks'] + version = 'v1.0.1' + hash = 'sha256-h75GUqfwJKngCJQVE5Ao5wnO3cfKD9lSIteoLp/3xJ4=' + + [mod.'github.com/cespare/xxhash/v2'] + version = 'v2.3.0' + hash = 'sha256-7hRlwSR+fos1kx4VZmJ/7snR7zHh8ZFKX+qqqqGcQpY=' + + [mod.'github.com/cheshir/logrustash'] + version = 'v0.0.0-20230213210745-aca6961b250d' + hash = 'sha256-8Wpg1T3R9HpUYNSkm1gtCXHwoqBMKzB3AJPC7mm+v94=' + + [mod.'github.com/davecgh/go-spew'] + version = 'v1.1.1' + hash = 'sha256-nhzSUrE1fCkN0+RL04N4h8jWmRFPPPWbCuDc7Ss0akI=' + + [mod.'github.com/dustin/go-humanize'] + version = 'v1.0.1' + hash = 'sha256-yuvxYYngpfVkUg9yAmG99IUVmADTQA0tMbBXe0Fq0Mc=' + + [mod.'github.com/fsnotify/fsnotify'] + version = 'v1.10.1' + hash = 'sha256-6LBLgsh4nKkMpgRKVsYFEaGDSU1fncBcWVSjKBdfgjU=' + + [mod.'github.com/go-viper/mapstructure/v2'] + version = 'v2.5.0' + hash = 'sha256-LbrCBANBprVI84M0CWrXc7rriJL5ac5VKbh58LBTw7U=' + + [mod.'github.com/munnerz/goautoneg'] + version = 'v0.0.0-20191010083416-a7dc8b61c822' + hash = 'sha256-79URDDFenmGc9JZu+5AXHToMrtTREHb3BC84b/gym9Q=' + + [mod.'github.com/pelletier/go-toml/v2'] + version = 'v2.4.2' + hash = 'sha256-5o0FLXD0pMCl5OMqdTZGuPz8C6M6qO7xK7QHi1X7bZU=' + + [mod.'github.com/pmezard/go-difflib'] + version = 'v1.0.0' + hash = 'sha256-/FtmHnaGjdvEIKAJtrUfEhV7EVo5A/eYrtdnUkuxLDA=' + + [mod.'github.com/prometheus/client_golang'] + version = 'v1.23.2' + hash = 'sha256-3GD4fBFa1tJu8MS4TNP6r2re2eViUE+kWUaieIOQXCg=' + + [mod.'github.com/prometheus/client_model'] + version = 'v0.6.2' + hash = 'sha256-q6Fh6v8iNJN9ypD47LjWmx66YITa3FyRjZMRsuRTFeQ=' + + [mod.'github.com/prometheus/common'] + version = 'v0.69.0' + hash = 'sha256-rqui1G7KMtVtGXDFY7XpWD9tA721UtoPC+/JLh8axrA=' + + [mod.'github.com/prometheus/procfs'] + version = 'v0.20.1' + hash = 'sha256-L6RuVGYgdBCCke8BibA7cXgfwizGE2pA0gNo9dGJ8WQ=' + + [mod.'github.com/sagikazarmark/locafero'] + version = 'v0.12.0' + hash = 'sha256-EXk9S5Z5sYyApAzCgHIugsGMbt/pHWRfHYFZH5D+5Ws=' + + [mod.'github.com/sirupsen/logrus'] + version = 'v1.9.4' + hash = 'sha256-ltRvmtM3XTCAFwY0IesfRqYIivyXPPuvkFjL4ARh1wg=' + + [mod.'github.com/spf13/afero'] + version = 'v1.15.0' + hash = 'sha256-LhcezbOqfuBzacytbqck0hNUxi6NbWNhifUc5/9uHQ8=' + + [mod.'github.com/spf13/cast'] + version = 'v1.10.0' + hash = 'sha256-dQ6Qqf26IZsa6XsGKP7GDuCj+WmSsBmkBwGTDfue/rk=' + + [mod.'github.com/spf13/pflag'] + version = 'v1.0.10' + hash = 'sha256-uDPnWjHpSrzXr17KEYEA1yAbizfcsfo5AyztY2tS6ZU=' + + [mod.'github.com/spf13/viper'] + version = 'v1.21.0' + hash = 'sha256-A9A8i7HH/ge4j3hw7G++HNj8BjhhpZKvxHhfY+QAxkI=' + + [mod.'github.com/stretchr/testify'] + version = 'v1.11.1' + hash = 'sha256-sWfjkuKJyDllDEtnM8sb/pdLzPQmUYWYtmeWz/5suUc=' + + [mod.'github.com/subosito/gotenv'] + version = 'v1.6.0' + hash = 'sha256-LspbjTniiq2xAICSXmgqP7carwlNaLqnCTQfw2pa80A=' + + [mod.'go.yaml.in/yaml/v2'] + version = 'v2.4.4' + hash = 'sha256-ecT2ZXw7iT+63J4210xA6sMz0fUFXmDzLwZe2FzaNFU=' + + [mod.'go.yaml.in/yaml/v3'] + version = 'v3.0.4' + hash = 'sha256-NkGFiDPoCxbr3LFsI6OCygjjkY0rdmg5ggvVVwpyDQ4=' + + [mod.'golang.org/x/sys'] + version = 'v0.46.0' + hash = 'sha256-NzRXMSEk6upeudJvUEPVnw6clJ3d8UdC/vdfANWAc8g=' + + [mod.'golang.org/x/text'] + version = 'v0.38.0' + hash = 'sha256-PzREcn7yzTAJ0WBvyYL/2/r+tfoeKTY/E5HVXi/CJsU=' + + [mod.'google.golang.org/protobuf'] + version = 'v1.36.11' + hash = 'sha256-7W+6jntfI/awWL3JP6yQedxqP5S9o3XvPgJ2XxxsIeE=' + + [mod.'gopkg.in/yaml.v3'] + version = 'v3.0.1' + hash = 'sha256-FqL9TKYJ0XkNwJFnq9j0VvJ5ZUU1RvH/52h/f5bkYAU=' + + [mod.'mvdan.cc/sh/v3'] + version = 'v3.13.1' + hash = 'sha256-FM+2xpGELZLm4Gc09OE5Z3JeNLPMvzeb1wRcdAIeKy4=' diff --git a/package.nix b/package.nix new file mode 100644 index 0000000..5874397 --- /dev/null +++ b/package.nix @@ -0,0 +1,11 @@ +{ + buildGoApplication, +}: + +buildGoApplication { + pname = "lug"; + version = "0.1.0"; + pwd = ./.; + src = ./.; + modules = ./gomod2nix.toml; +} diff --git a/shell.nix b/shell.nix new file mode 100644 index 0000000..692cd4d --- /dev/null +++ b/shell.nix @@ -0,0 +1,12 @@ +(import ( + let + lock = builtins.fromJSON (builtins.readFile ./flake.lock); + nodeName = lock.nodes.root.inputs.flake-compat; + in + fetchTarball { + url = + lock.nodes.${nodeName}.locked.url + or "https://github.com/NixOS/flake-compat/archive/${lock.nodes.${nodeName}.locked.rev}.tar.gz"; + sha256 = lock.nodes.${nodeName}.locked.narHash; + } +) { src = ./.; }).shellNix From cb35c056f88b3f829c7111d983af148c6d6a25b7 Mon Sep 17 00:00:00 2001 From: comonad Date: Fri, 26 Jun 2026 00:24:29 +0800 Subject: [PATCH 02/15] chore: lint --- cli/lug/license.go | 7 ++++--- cli/lug/main.go | 12 ++++++++---- pkg/config/config.go | 5 +++-- pkg/config/config_test.go | 2 +- pkg/exporter/exporter.go | 2 +- pkg/helper/disk_usage.go | 2 +- pkg/manager/json_rest.go | 8 ++++++-- pkg/manager/manager.go | 5 +++-- pkg/worker/executor_invoke_worker.go | 7 ++++--- pkg/worker/external_worker.go | 2 +- pkg/worker/shell_script_executor.go | 4 ++-- pkg/worker/utility_rlimit.go | 3 +-- pkg/worker/worker.go | 2 +- pkg/worker/worker_test.go | 4 ++-- 14 files changed, 38 insertions(+), 27 deletions(-) diff --git a/cli/lug/license.go b/cli/lug/license.go index 76d55e7..456787e 100644 --- a/cli/lug/license.go +++ b/cli/lug/license.go @@ -1,6 +1,7 @@ -package main - /* Generated by script scripts/gen_license.sh */ - const licenseText = ` ==> ./vendor/gopkg.in/yaml.v3/LICENSE <== +package main + +/* Generated by script scripts/gen_license.sh */ +const licenseText = ` ==> ./vendor/gopkg.in/yaml.v3/LICENSE <== This project is covered by two different licenses: MIT and Apache. diff --git a/cli/lug/main.go b/cli/lug/main.go index 0e3e9b8..879ffdd 100644 --- a/cli/lug/main.go +++ b/cli/lug/main.go @@ -67,8 +67,8 @@ func init() { flags := getFlags() cfgViper := config.CfgViper - cfgViper.BindPFlag("json_api.address", flag.Lookup("jsonapi")) - cfgViper.BindPFlag("exporter_address", flag.Lookup("exporter")) + _ = cfgViper.BindPFlag("json_api.address", flag.Lookup("jsonapi")) + _ = cfgViper.BindPFlag("exporter_address", flag.Lookup("exporter")) if flags.version { fmt.Print(lugVersionInfo) @@ -86,7 +86,7 @@ func init() { fmt.Print(configHelp) os.Exit(0) } - defer file.Close() + defer func() { _ = file.Close() }() cfg = config.Config{} err = cfg.Parse(file) @@ -105,7 +105,11 @@ func main() { } jsonapi := manager.NewRestfulAPI(m) handler := jsonapi.GetAPIHandler() - go http.ListenAndServe(cfg.JsonAPIConfig.Address, handler) + go func() { + if err := http.ListenAndServe(cfg.JsonAPIConfig.Address, handler); err != nil { + log.Error(err) + } + }() go exporter.Expose(cfg.ExporterAddr) m.Run() diff --git a/pkg/config/config.go b/pkg/config/config.go index 90c8cfa..a6c4a2b 100644 --- a/pkg/config/config.go +++ b/pkg/config/config.go @@ -70,9 +70,10 @@ func (c *Config) Parse(in io.Reader) (err error) { err = CfgViper.UnmarshalExact(&c) if err == nil { if c.Interval < 0 { - return errors.New("Interval can't be negative") + return errors.New("interval can't be negative") } - if c.LogLevel < 0 || c.LogLevel > 5 { + // logrus.Level is uint32 (non-negative) + if c.LogLevel > 5 { return errors.New("loglevel must be 0-5") } if c.ConcurrentLimit <= 0 { diff --git a/pkg/config/config_test.go b/pkg/config/config_test.go index aea418f..3708ddf 100644 --- a/pkg/config/config_test.go +++ b/pkg/config/config_test.go @@ -98,7 +98,7 @@ repos: err = c.Parse(strings.NewReader(testStr)) asrt := assert.New(t) - asrt.Equal("Interval can't be negative", err.Error()) + asrt.Equal("interval can't be negative", err.Error()) testStr = `interval: 25 loglevel: 6 diff --git a/pkg/exporter/exporter.go b/pkg/exporter/exporter.go index 035e19a..d3b8365 100644 --- a/pkg/exporter/exporter.go +++ b/pkg/exporter/exporter.go @@ -107,7 +107,7 @@ func (e *Exporter) UpdateDiskUsage(worker string, path string) { "path": path, }) logger.WithField("event", "update_disk_usage").Info("Invoke UpdateDiskUsage") - if !found || time.Now().Sub(lastUpdateTime) > updateDiskUsageThrottle { + if !found || time.Since(lastUpdateTime) > updateDiskUsageThrottle { logger.Debug("background update_disk_usage launched") // first, we set it to infinity (2037-01-01) // Note that we cannot use a larger value due to Y2038 problem on *nix diff --git a/pkg/helper/disk_usage.go b/pkg/helper/disk_usage.go index 8b1b607..20acb38 100644 --- a/pkg/helper/disk_usage.go +++ b/pkg/helper/disk_usage.go @@ -13,7 +13,7 @@ func DiskUsage(curPath string) (int64, error) { if err != nil { return size, err } - defer dir.Close() + defer func() { _ = dir.Close() }() files, err := dir.Readdir(-1) if err != nil { diff --git a/pkg/manager/json_rest.go b/pkg/manager/json_rest.go index 442a554..9996e48 100644 --- a/pkg/manager/json_rest.go +++ b/pkg/manager/json_rest.go @@ -55,7 +55,9 @@ type MangerStatusSimple struct { func (r *RestfulAPI) getManagerStatusCommon(w rest.ResponseWriter, req *rest.Request, detailed bool) { rawStatus := r.manager.GetStatus() if detailed { - w.WriteJson(rawStatus) + if err := w.WriteJson(rawStatus); err != nil { + log.Error(err) + } return } managerStatusSimple := MangerStatusSimple{ @@ -70,7 +72,9 @@ func (r *RestfulAPI) getManagerStatusCommon(w rest.ResponseWriter, req *rest.Req Idle: rawWorkerStatus.Idle, } } - w.WriteJson(managerStatusSimple) + if err := w.WriteJson(managerStatusSimple); err != nil { + log.Error(err) + } } func (r *RestfulAPI) getManagerStatusDetail(w rest.ResponseWriter, req *rest.Request) { diff --git a/pkg/manager/manager.go b/pkg/manager/manager.go index 7c1f543..f5e1e77 100644 --- a/pkg/manager/manager.go +++ b/pkg/manager/manager.go @@ -4,11 +4,12 @@ package manager import ( "encoding/json" "fmt" - "github.com/davecgh/go-spew/spew" "io" "os" "time" + "github.com/davecgh/go-spew/spew" + "github.com/sirupsen/logrus" "github.com/sjtug/lug/pkg/config" @@ -361,7 +362,7 @@ func (m *Manager) GetStatus() *Status { for _, w := range m.workers { wConfig := w.GetConfig() wStatus := w.GetStatus() - if hidden, ok := wConfig["hidden"].(bool); !(ok && hidden) { + if hidden, ok := wConfig["hidden"].(bool); !ok || !hidden { status.WorkerStatus[wConfig["name"].(string)] = wStatus } } diff --git a/pkg/worker/executor_invoke_worker.go b/pkg/worker/executor_invoke_worker.go index 3173336..82875aa 100644 --- a/pkg/worker/executor_invoke_worker.go +++ b/pkg/worker/executor_invoke_worker.go @@ -2,13 +2,14 @@ package worker import ( "errors" + "sync" + "time" + "github.com/davecgh/go-spew/spew" log "github.com/sirupsen/logrus" "github.com/sjtug/lug/pkg/config" "github.com/sjtug/lug/pkg/exporter" "github.com/sjtug/lug/pkg/helper" - "sync" - "time" ) type executorInvokeWorker struct { @@ -35,7 +36,7 @@ func NewExecutorInvokeWorker(exector executor, status Status, signal chan int) (*executorInvokeWorker, error) { name, ok := cfg["name"].(string) if !ok { - return nil, errors.New("No name in config") + return nil, errors.New("no name in config") } w := &executorInvokeWorker{ idle: status.Idle, diff --git a/pkg/worker/external_worker.go b/pkg/worker/external_worker.go index b4b7bda..3fbd944 100644 --- a/pkg/worker/external_worker.go +++ b/pkg/worker/external_worker.go @@ -20,7 +20,7 @@ type ExternalWorker struct { func NewExternalWorker(cfg config.RepoConfig) (*ExternalWorker, error) { rawName, ok := cfg["name"] if !ok { - return nil, errors.New("Name is required for external worker") + return nil, errors.New("name is required for external worker") } name := rawName.(string) return &ExternalWorker{ diff --git a/pkg/worker/shell_script_executor.go b/pkg/worker/shell_script_executor.go index ec233fc..0cf6e51 100644 --- a/pkg/worker/shell_script_executor.go +++ b/pkg/worker/shell_script_executor.go @@ -29,11 +29,11 @@ func newShellScriptExecutor(cfg config.RepoConfig) *shellScriptExecutor { func convertMapToEnvVars(m map[string]interface{}) (map[string]string, error) { result := map[string]string{} for k, v := range m { - switch v.(type) { + switch v := v.(type) { case nil: // skip case bool: - if v.(bool) { + if v { result["LUG_"+k] = "1" } case int, uint, float32, float64, string: diff --git a/pkg/worker/utility_rlimit.go b/pkg/worker/utility_rlimit.go index 1d4e893..31336d6 100644 --- a/pkg/worker/utility_rlimit.go +++ b/pkg/worker/utility_rlimit.go @@ -31,8 +31,7 @@ func (r *rlimit) preHook() error { } if rlimitMem, ok := cfg["rlimit_mem"]; ok { if bytes, err := humanize.ParseBytes(rlimitMem.(string)); err == nil { - var rlimitNew syscall.Rlimit - rlimitNew = r.oldRlimit + rlimitNew := r.oldRlimit rlimitNew.Cur = bytes err := syscall.Setrlimit(syscall.RLIMIT_AS, &rlimitNew) if err != nil { diff --git a/pkg/worker/worker.go b/pkg/worker/worker.go index 8627834..d88ad68 100644 --- a/pkg/worker/worker.go +++ b/pkg/worker/worker.go @@ -64,5 +64,5 @@ func NewWorker(cfg config.RepoConfig, lastFinished time.Time, Result bool) (Work return w, nil } } - return nil, errors.New("Fail to create a new worker") + return nil, errors.New("fail to create a new worker") } diff --git a/pkg/worker/worker_test.go b/pkg/worker/worker_test.go index 0fd2f0e..1224bf1 100644 --- a/pkg/worker/worker_test.go +++ b/pkg/worker/worker_test.go @@ -171,9 +171,9 @@ func TestUtilityRlimit(t *testing.T) { cmd := exec.Command("rev") cmd.Stdin = newLimitReader(20000000) // > 10M = 10485760 - rlimitUtility.preHook() + asrt.NoError(rlimitUtility.preHook()) err1 := cmd.Start() - rlimitUtility.postHook() + asrt.NoError(rlimitUtility.postHook()) var err2 error if err1 == nil { err2 = cmd.Wait() From 1152fb02e44354a25446db8d195d03a155c171a0 Mon Sep 17 00:00:00 2001 From: comonad Date: Fri, 26 Jun 2026 01:56:48 +0800 Subject: [PATCH 03/15] feat(log): structured JSON logging w/ repo/sync_id - prepareLogger: JSONFormatter to stdout (12-factor XI, event stream); logstash hook Fatal -> Warn so a dead sink never takes down the scheduler - executorInvokeWorker.RunSync: stamp a short sync_id per sync attempt and thread it through every log line + into executor.RunOnce, so all lines of one run correlate in the stream - rename logrus field "worker" -> "repo" on the two worker loggers (executorInvokeWorker, ExternalWorker); Prometheus label "worker" and manager's "manager" field unchanged for metric/API compatibility --- cli/lug/main.go | 26 ++++++++++---- pkg/worker/executor_invoke_worker.go | 52 ++++++++++++++++++++-------- pkg/worker/external_worker.go | 2 +- 3 files changed, 59 insertions(+), 21 deletions(-) diff --git a/cli/lug/main.go b/cli/lug/main.go index 879ffdd..c253742 100644 --- a/cli/lug/main.go +++ b/cli/lug/main.go @@ -48,16 +48,30 @@ func getFlags() (flags CommandFlags) { // Register Logger and set logLevel func prepareLogger(logLevel log.Level, logStashAddr string, additionalFields map[string]interface{}) { log.SetLevel(logLevel) + + // 12-factor XI: logs are an event stream written unbuffered to stdout. + // Docker's logging driver (journald/loki/json-file) handles routing. + log.SetFormatter(&log.JSONFormatter{ + TimestampFormat: time.RFC3339Nano, + FieldMap: log.FieldMap{ + log.FieldKeyMsg: "msg", + }, + }) + log.SetOutput(os.Stdout) + if logStashAddr != "" { hook, err := logrustash.NewAsyncHookWithFields("tcp", logStashAddr, "lug", additionalFields) if err != nil { - log.Fatal(err) + // A misconfigured logstash sink must not take the scheduler down; + // the stdout event stream remains the source of truth. + log.WithError(err).Warn("logstash hook disabled") + } else { + hook.WaitUntilBufferFrees = true + hook.ReconnectBaseDelay = time.Second + hook.ReconnectDelayMultiplier = 2 + hook.MaxSendRetries = 10 + log.AddHook(hook) } - hook.WaitUntilBufferFrees = true - hook.ReconnectBaseDelay = time.Second - hook.ReconnectDelayMultiplier = 2 - hook.MaxSendRetries = 10 - log.AddHook(hook) } } diff --git a/pkg/worker/executor_invoke_worker.go b/pkg/worker/executor_invoke_worker.go index 82875aa..87e9b8b 100644 --- a/pkg/worker/executor_invoke_worker.go +++ b/pkg/worker/executor_invoke_worker.go @@ -1,6 +1,8 @@ package worker import ( + "crypto/rand" + "encoding/hex" "errors" "sync" "time" @@ -49,7 +51,7 @@ func NewExecutorInvokeWorker(exector executor, status Status, cfg: cfg, signal: signal, name: name, - logger: log.WithField("worker", name), + logger: log.WithField("repo", name), executor: exector, } if retry_generic, ok := cfg["retry"]; ok { @@ -95,39 +97,46 @@ func (eiw *executorInvokeWorker) GetConfig() config.RepoConfig { func (w *executorInvokeWorker) RunSync() { for { - w.logger.WithField("event", "start_wait_signal").Debug("start waiting for signal") + // Before the signal there is no sync_id yet; reuse the repo-scoped logger. + logger := w.logger + logger.WithField("event", "start_wait_signal").Debug("start waiting for signal") func() { w.rwmutex.Lock() defer w.rwmutex.Unlock() w.idle = true }() <-w.signal - w.logger.WithField("event", "signal_received").Debug("finished waiting for signal") + + // A new sync run begins: stamp a sync_id so every line of this run + // can be correlated in the event stream (see PLAN, Phase 0). + syncID := shortID() + logger = w.logger.WithField("sync_id", syncID) + logger.WithField("event", "signal_received").Debug("finished waiting for signal") func() { w.rwmutex.Lock() defer w.rwmutex.Unlock() w.idle = false }() - w.logger.WithField("event", "start_execution").Info("start execution") + logger.WithField("event", "start_execution").Info("sync started") retry_limit := w.retry var result execResult var err error for retry_cnt := 1; retry_cnt <= retry_limit; retry_cnt++ { - w.logger.WithField("event", "invoke_executor").WithField( + logger.WithField("event", "invoke_executor").WithField( "try_cnt", retry_cnt).Debugf("Invoke executor for the %v time", retry_cnt) utilities := []utility{newRlimit(w)} - result, err = w.executor.RunOnce(w.logger, utilities) + result, err = w.executor.RunOnce(logger, utilities) if err == nil { break } - w.logger.WithField("event", "invoke_executor_fail").WithField( + logger.WithField("event", "invoke_executor_fail").WithField( "try_cnt", retry_cnt).Infof( "Failed on the %v-th executor. Error: %v", retry_cnt, err.Error()) - w.logger.Debug("Stderr: ", result.Stderr) + logger.Debug("Stderr: ", result.Stderr) time.Sleep(w.retry_interval) } if err != nil { - w.logger.WithField("event", "execution_fail").Error(err.Error()) + logger.WithField("event", "execution_fail").Error(err.Error()) exporter.GetInstance().SyncFail(w.name) func() { w.rwmutex.Lock() @@ -135,24 +144,39 @@ func (w *executorInvokeWorker) RunSync() { w.result = false w.stdout.Put(result.Stdout) w.stderr.Put(result.Stderr) - w.logger.Infof("Stderr: %s", result.Stderr) - w.logger.Debugf("Stdout: %s", result.Stdout) + logger.Infof("Stderr: %s", result.Stderr) + logger.Debugf("Stdout: %s", result.Stdout) w.idle = true }() + logger.WithField("event", "execution_ended").Info("sync ended") continue } exporter.GetInstance().SyncSuccess(w.name) - w.logger.WithField("event", "execution_succeed").Info("succeed") - w.logger.Infof("Stderr: %s", result.Stderr) + logger.WithField("event", "execution_succeed").Info("sync succeeded") + logger.Infof("Stderr: %s", result.Stderr) func() { w.rwmutex.Lock() defer w.rwmutex.Unlock() w.stderr.Put(result.Stderr) - w.logger.Debugf("Stdout: %s", result.Stdout) + logger.Debugf("Stdout: %s", result.Stdout) w.stdout.Put(result.Stdout) w.result = true w.lastFinished = time.Now() }() + logger.WithField("event", "execution_ended").Info("sync ended") + } +} + +// shortID generates a short unique ID without external dependencies. +// It returns 4 bytes of crypto-random hex (8 chars), enough to correlate +// all log lines within a single sync attempt without collision risk. +func shortID() string { + b := make([]byte, 4) + if _, err := rand.Read(b); err != nil { + // rand.Read should not fail in practice; fall back to a fixed marker + // so logging never breaks on ID generation. + return "00000000" } + return hex.EncodeToString(b) } diff --git a/pkg/worker/external_worker.go b/pkg/worker/external_worker.go index 3fbd944..8d11ff6 100644 --- a/pkg/worker/external_worker.go +++ b/pkg/worker/external_worker.go @@ -25,7 +25,7 @@ func NewExternalWorker(cfg config.RepoConfig) (*ExternalWorker, error) { name := rawName.(string) return &ExternalWorker{ name: name, - logger: log.WithField("worker", name), + logger: log.WithField("repo", name), cfg: cfg, }, nil } From f68b7a84b91776767ec182e4f66973cf62173a85 Mon Sep 17 00:00:00 2001 From: comonad Date: Sun, 13 Sep 2026 23:10:34 +0800 Subject: [PATCH 04/15] chore: update nix toolchain and go dependencies Pin go-version, refresh flake.lock/gomod2nix and module deps. --- .envrc | 2 ++ .go-version | 1 + flake.lock | 46 ++++++++++++---------------------------------- flake.nix | 6 +----- go.mod | 20 ++++++++++---------- go.sum | 22 ++++++++++++++++++++++ gomod2nix.toml | 36 ++++++++++++++++++------------------ 7 files changed, 66 insertions(+), 67 deletions(-) create mode 100644 .go-version diff --git a/.envrc b/.envrc index d5359eb..e576402 100644 --- a/.envrc +++ b/.envrc @@ -3,4 +3,6 @@ export GO111MODULE="on" export GOPROXY="https://goproxy.cn" + +watch_file .go-version use flake . -Lv --fallback --show-trace diff --git a/.go-version b/.go-version new file mode 100644 index 0000000..24cffb8 --- /dev/null +++ b/.go-version @@ -0,0 +1 @@ +1.26 diff --git a/flake.lock b/flake.lock index 3648c61..5fc279e 100644 --- a/flake.lock +++ b/flake.lock @@ -39,11 +39,11 @@ ] }, "locked": { - "lastModified": 1778716662, - "narHash": "sha256-m1Yf0wZ8j1OHjTc2UwHwyQRSnNeSgLJOd7q5Y45hzi4=", + "lastModified": 1788450739, + "narHash": "sha256-glZLQlzIn1fXH6PazR2iUmTo7kzzyYSshrWhLS9TqCU=", "owner": "hercules-ci", "repo": "flake-parts", - "rev": "f7c1a2d347e4c52d5fb8d10cb4d94b5884e546fb", + "rev": "31729ca8cbdb4fa927b34e5f4353e6a83f39e993", "type": "github" }, "original": { @@ -70,27 +70,6 @@ "type": "github" } }, - "gitignore": { - "inputs": { - "nixpkgs": [ - "pre-commit-hooks", - "nixpkgs" - ] - }, - "locked": { - "lastModified": 1709087332, - "narHash": "sha256-HG2cCnktfHsKV0s4XW83gU3F57gaTljL9KNSuG6bnQs=", - "owner": "hercules-ci", - "repo": "gitignore.nix", - "rev": "637db329424fd7e46cf4185293b9cc8c88c95394", - "type": "github" - }, - "original": { - "owner": "hercules-ci", - "repo": "gitignore.nix", - "type": "github" - } - }, "gomod2nix": { "inputs": { "flake-utils": [ @@ -116,11 +95,11 @@ }, "nixpkgs": { "locked": { - "lastModified": 1781607440, - "narHash": "sha256-rxO+uc/KFbSJp+pgyXRuAX6QlG9hJdnt0BXpEQRXY+U=", + "lastModified": 1789073787, + "narHash": "sha256-xfX/toC2QV707s06GbP4II/TxYF0fNQj7s5/LClNDKc=", "owner": "NixOS", "repo": "nixpkgs", - "rev": "3e41b24abd260e8f71dbe2f5737d24122f972158", + "rev": "aff8a0b28396750446e5537a96461bc4facdb287", "type": "github" }, "original": { @@ -133,17 +112,16 @@ "pre-commit-hooks": { "inputs": { "flake-compat": "flake-compat_2", - "gitignore": "gitignore", "nixpkgs": [ "nixpkgs" ] }, "locked": { - "lastModified": 1781733627, - "narHash": "sha256-U3yTuGBnmXvXoQI3qkpfEDsn9RovQPAjN7ndRco+3u0=", + "lastModified": 1788267358, + "narHash": "sha256-nt+lUqYVpc9Y6JeMd2WmXzCDojasdadKo0mWcluvY2Y=", "owner": "cachix", "repo": "git-hooks.nix", - "rev": "3bbec39bc90eadfa031e6f3b77272f3f60803e39", + "rev": "27555e2624241fb116b49095df4caaee85a25691", "type": "github" }, "original": { @@ -185,11 +163,11 @@ ] }, "locked": { - "lastModified": 1780220602, - "narHash": "sha256-eynAfOmbmxJnkp7YewvCEbShNnnYJ9gLLqkzsYtBPeM=", + "lastModified": 1786901030, + "narHash": "sha256-WSFCsDSE5ffgD2MqzkM2CYjeFiKhRF/dJUN8uedb6YE=", "owner": "numtide", "repo": "treefmt-nix", - "rev": "db947814a175b7ca6ded66e21383d938df01c227", + "rev": "27b3b12a8e6375f28ebe122f07d230ca5459bbfa", "type": "github" }, "original": { diff --git a/flake.nix b/flake.nix index aca92ca..282b3a1 100644 --- a/flake.nix +++ b/flake.nix @@ -50,7 +50,7 @@ ... }: let - goVersion = "1.25"; + goVersion = lib.fileContents ./.go-version; in { _module.args.pkgs = import inputs.nixpkgs { @@ -95,10 +95,6 @@ config.pre-commit.devShell ]; - shellHook = '' - echo 1>&2 "Welcome to the development shell!" - ''; - packages = with pkgs; [ (mkGoEnv { pwd = ./.; }) gopls diff --git a/go.mod b/go.mod index 99a5e2c..33db85b 100644 --- a/go.mod +++ b/go.mod @@ -1,18 +1,18 @@ module github.com/sjtug/lug -go 1.25.0 +go 1.26.0 require ( github.com/ant0ine/go-json-rest v3.3.2+incompatible github.com/cheshir/logrustash v0.0.0-20230213210745-aca6961b250d github.com/davecgh/go-spew v1.1.1 github.com/dustin/go-humanize v1.0.1 - github.com/prometheus/client_golang v1.23.2 - github.com/sirupsen/logrus v1.9.4 + github.com/prometheus/client_golang v1.24.1 + github.com/sirupsen/logrus v1.10.2 github.com/spf13/pflag v1.0.10 github.com/spf13/viper v1.21.0 - github.com/stretchr/testify v1.11.1 - mvdan.cc/sh/v3 v3.13.1 + github.com/stretchr/testify v1.12.1 + mvdan.cc/sh/v3 v3.14.1 ) require ( @@ -24,16 +24,16 @@ require ( github.com/pelletier/go-toml/v2 v2.4.2 // indirect github.com/pmezard/go-difflib v1.0.0 // indirect github.com/prometheus/client_model v0.6.2 // indirect - github.com/prometheus/common v0.69.0 // indirect - github.com/prometheus/procfs v0.20.1 // indirect + github.com/prometheus/common v0.70.1 // indirect + github.com/prometheus/procfs v0.21.1 // indirect github.com/sagikazarmark/locafero v0.12.0 // indirect github.com/spf13/afero v1.15.0 // indirect github.com/spf13/cast v1.10.0 // indirect github.com/subosito/gotenv v1.6.0 // indirect go.yaml.in/yaml/v2 v2.4.4 // indirect - go.yaml.in/yaml/v3 v3.0.4 // indirect - golang.org/x/sys v0.46.0 // indirect - golang.org/x/text v0.38.0 // indirect + go.yaml.in/yaml/v3 v3.0.5 // indirect + golang.org/x/sys v0.47.0 // indirect + golang.org/x/text v0.40.0 // indirect google.golang.org/protobuf v1.36.11 // indirect gopkg.in/yaml.v3 v3.0.1 // indirect ) diff --git a/go.sum b/go.sum index 33d553c..16c3acf 100644 --- a/go.sum +++ b/go.sum @@ -18,6 +18,7 @@ github.com/fsnotify/fsnotify v1.10.1 h1:b0/UzAf9yR5rhf3RPm9gf3ehBPpf0oZKIjtpKrx5 github.com/fsnotify/fsnotify v1.10.1/go.mod h1:TLheqan6HD6GBK6PrDWyDPBaEV8LspOxvPSjC+bVfgo= github.com/go-quicktest/qt v1.101.0 h1:O1K29Txy5P2OK0dGo59b7b0LR6wKfIhttaAhHUyn7eI= github.com/go-quicktest/qt v1.101.0/go.mod h1:14Bz/f7NwaXPtdYEgzsx46kqSxVwTbzVZsDC26tQJow= +github.com/go-quicktest/qt v1.102.0 h1:HSQxCeh5YZH3EL3W39ixjtyaEhcWSXQHtHnMBzSs474= github.com/go-viper/mapstructure/v2 v2.4.0 h1:EBsztssimR/CONLSZZ04E8qAkxNYq4Qp9LvH92wZUgs= github.com/go-viper/mapstructure/v2 v2.4.0/go.mod h1:oJDH3BJKyqBA2TXFhDsKDGDTlndYOZ6rGS0BRZIxGhM= github.com/go-viper/mapstructure/v2 v2.5.0 h1:vM5IJoUAy3d7zRSVtIwQgBj7BiWtMPfmPEgAXnvj1Ro= @@ -26,6 +27,7 @@ github.com/google/go-cmp v0.7.0 h1:wk8382ETsv4JYUZwIsn6YpYiWiBsYLSJiTsyBybVuN8= github.com/google/go-cmp v0.7.0/go.mod h1:pXiqmnSA92OHEEa9HXL2W4E7lf9JzCmGVUdgjX3N/iU= github.com/klauspost/compress v1.18.0 h1:c/Cqfb0r+Yi+JtIEq73FWXVkRonBlf0CRNYc8Zttxdo= github.com/klauspost/compress v1.18.0/go.mod h1:2Pp+KzxcywXVXMr50+X0Q/Lsb43OQHYWRCY2AiWywWQ= +github.com/klauspost/compress v1.19.1 h1:VsB4HPswih7mmZ8WleSFQ75c/Ui1M4trX5oAsJnhSlk= github.com/kr/pretty v0.3.1 h1:flRD4NNwYAUpkphVc1HcthR4KEIFJ65n8Mw5qdRn3LE= github.com/kr/pretty v0.3.1/go.mod h1:hoEshYVHaxMs3cyo3Yncou5ZscifuDolrwPKZanG3xk= github.com/kr/text v0.2.0 h1:5Nx0Ya0ZqY2ygV366QzturHI13Jq95ApcVaJBhpS+AY= @@ -42,20 +44,29 @@ github.com/pmezard/go-difflib v1.0.0 h1:4DBwDE0NGyQoBHbLQYPwSUPoCMWR5BEzIk/f1lZb github.com/pmezard/go-difflib v1.0.0/go.mod h1:iKH77koFhYxTK1pcRnkKkqfTogsbg7gZNVY4sRDYZ/4= github.com/prometheus/client_golang v1.23.2 h1:Je96obch5RDVy3FDMndoUsjAhG5Edi49h0RJWRi/o0o= github.com/prometheus/client_golang v1.23.2/go.mod h1:Tb1a6LWHB3/SPIzCoaDXI4I8UHKeFTEQ1YCr+0Gyqmg= +github.com/prometheus/client_golang v1.24.1 h1:JnJkREXzWxUdCuPFpIWZiPispT9xVV59uiuyR2bPlnU= +github.com/prometheus/client_golang v1.24.1/go.mod h1:F+oSRECHg4sse5ucfYpYDeIv/hu68Zo0uoHKetWnzcE= github.com/prometheus/client_model v0.6.2 h1:oBsgwpGs7iVziMvrGhE53c/GrLUsZdHnqNwqPLxwZyk= github.com/prometheus/client_model v0.6.2/go.mod h1:y3m2F6Gdpfy6Ut/GBsUqTWZqCUvMVzSfMLjcu6wAwpE= github.com/prometheus/common v0.66.1 h1:h5E0h5/Y8niHc5DlaLlWLArTQI7tMrsfQjHV+d9ZoGs= github.com/prometheus/common v0.66.1/go.mod h1:gcaUsgf3KfRSwHY4dIMXLPV0K/Wg1oZ8+SbZk/HH/dA= github.com/prometheus/common v0.69.0 h1:OA85nJQS/T/MaYh/Q2CcgDKSGWqNIgrBDvDH85CuiNk= github.com/prometheus/common v0.69.0/go.mod h1:ZzL3f6u94qUxh9p+tJTrF+FvBS1XXbbRAZCQkytAL0Y= +github.com/prometheus/common v0.70.1 h1:1HvjP4D5oL3t8RsPlwxA9onvvStjtIHYE5XuuwOi/PY= +github.com/prometheus/common v0.70.1/go.mod h1:VdFUQDMZK3VLkurFUVhia6uys/0suUp86TJz5qbJRhc= github.com/prometheus/procfs v0.20.1 h1:XwbrGOIplXW/AU3YhIhLODXMJYyC1isLFfYCsTEycfc= github.com/prometheus/procfs v0.20.1/go.mod h1:o9EMBZGRyvDrSPH1RqdxhojkuXstoe4UlK79eF5TGGo= +github.com/prometheus/procfs v0.21.1 h1:GljZCt+zSTS+NZq88cyQ1LjZ+RCHp3uVuabBWA5+OJI= +github.com/prometheus/procfs v0.21.1/go.mod h1:aB55Cww9pdSJVHk0hUf0inxWyyjPogFIjmHKYgMKmtY= github.com/rogpeppe/go-internal v1.14.1 h1:UQB4HGPB6osV0SQTLymcB4TgvyWu6ZyliaW0tI/otEQ= github.com/rogpeppe/go-internal v1.14.1/go.mod h1:MaRKkUm5W0goXpeCfT7UZI6fk/L7L7so1lCWt35ZSgc= +github.com/rogpeppe/go-internal v1.15.0 h1:D0RCU5rMAp+SpgkiNdrjfJ+LX4J1M32V2NeCY7EJ6hc= github.com/sagikazarmark/locafero v0.12.0 h1:/NQhBAkUb4+fH1jivKHWusDYFjMOOKU88eegjfxfHb4= github.com/sagikazarmark/locafero v0.12.0/go.mod h1:sZh36u/YSZ918v0Io+U9ogLYQJ9tLLBmM4eneO6WwsI= github.com/sirupsen/logrus v1.9.4 h1:TsZE7l11zFCLZnZ+teH4Umoq5BhEIfIzfRDZ1Uzql2w= github.com/sirupsen/logrus v1.9.4/go.mod h1:ftWc9WdOfJ0a92nsE2jF5u5ZwH8Bv2zdeOC42RjbV2g= +github.com/sirupsen/logrus v1.10.2 h1:G2SED73/qrAu6YwbdxOD6peLkCBI3z7L+ykJFTXJBBo= +github.com/sirupsen/logrus v1.10.2/go.mod h1:SLEg8TqYulVKKfIGHldVp2K2aYz2DKSVBq4g/H5bR7Q= github.com/spf13/afero v1.15.0 h1:b/YBCLWAJdFWJTN9cLhiXXcD7mzKn9Dm86dNnfyQw1I= github.com/spf13/afero v1.15.0/go.mod h1:NC2ByUVxtQs4b3sIUphxK0NioZnmxgyCrfzeuq8lxMg= github.com/spf13/cast v1.10.0 h1:h2x0u2shc1QuLHfxi+cTJvs30+ZAHOGRic8uyGTDWxY= @@ -66,6 +77,8 @@ github.com/spf13/viper v1.21.0 h1:x5S+0EU27Lbphp4UKm1C+1oQO+rKx36vfCoaVebLFSU= github.com/spf13/viper v1.21.0/go.mod h1:P0lhsswPGWD/1lZJ9ny3fYnVqxiegrlNrEmgLjbTCAY= github.com/stretchr/testify v1.11.1 h1:7s2iGBzp5EwR7/aIZr8ao5+dra3wiQyKjjFuvgVKu7U= github.com/stretchr/testify v1.11.1/go.mod h1:wZwfW3scLgRK+23gO65QZefKpKQRnfz6sD981Nm4B6U= +github.com/stretchr/testify v1.12.1 h1:EuwCh5fleGS7H32xRwO3wRGT7DxrDhLAT6FF8MpWDWE= +github.com/stretchr/testify v1.12.1/go.mod h1:MDEgiDPPsNp5cuIrHPPCyornHKgEVbtFUmoNlxoYthg= github.com/subosito/gotenv v1.6.0 h1:9NlTDc1FTs4qu0DDq7AEtTPNw6SVm7uBMsUCUjABIf8= github.com/subosito/gotenv v1.6.0/go.mod h1:Dk4QP5c2W3ibzajGcXpNraDfq2IrhjMIvMSWPKKo0FU= go.uber.org/goleak v1.3.0 h1:2K3zAYmnTNqV73imy9J1T3WC+gmCePx2hEGkimedGto= @@ -74,12 +87,19 @@ go.yaml.in/yaml/v2 v2.4.4 h1:tuyd0P+2Ont/d6e2rl3be67goVK4R6deVxCUX5vyPaQ= go.yaml.in/yaml/v2 v2.4.4/go.mod h1:gMZqIpDtDqOfM0uNfy0SkpRhvUryYH0Z6wdMYcacYXQ= go.yaml.in/yaml/v3 v3.0.4 h1:tfq32ie2Jv2UxXFdLJdh3jXuOzWiL1fo0bu/FbuKpbc= go.yaml.in/yaml/v3 v3.0.4/go.mod h1:DhzuOOF2ATzADvBadXxruRBLzYTpT36CKvDb3+aBEFg= +go.yaml.in/yaml/v3 v3.0.5 h1:N6y/pJk8buWs9NY5ERU2HSMfm+IuD/OtfdAnq6kESPw= +go.yaml.in/yaml/v3 v3.0.5/go.mod h1:HVTZu1O7/Vkt2N+BFy8Zza+lnLsABggaTM2ZpNIGuKg= golang.org/x/sys v0.46.0 h1:noSf2Fq6F8DBgS+LysIkx7rIExoNHJsxOAtPp4rthXw= golang.org/x/sys v0.46.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw= +golang.org/x/sys v0.47.0 h1:o7XGOvZQCADBQQ4Y7VNq2dRWQR7JmOUW8Kxx4ZsNgWs= +golang.org/x/sys v0.47.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw= golang.org/x/text v0.38.0 h1:sXmwo9DwP3OK9EZ7PqAdaooSGozfl/3a6/xJcbzPRhE= golang.org/x/text v0.38.0/go.mod h1:YXZt3QhHUKYT53r2lLKFIVi6Ao1jdzrTR/KQ09qyxF4= +golang.org/x/text v0.40.0 h1:Ub2Z6/xjgF1WrYQz2nuITOEegKFtiIy+rieRJ5lHZKs= +golang.org/x/text v0.40.0/go.mod h1:hpnzDAfGV753zIKo+wk3u1bVKCGPbrnF7+7LBF/UHVY= google.golang.org/protobuf v1.36.11 h1:fV6ZwhNocDyBLK0dj+fg8ektcVegBBuEolpbTQyBNVE= google.golang.org/protobuf v1.36.11/go.mod h1:HTf+CrKn2C3g5S8VImy6tdcUvCska2kB7j23XfzDpco= +gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405 h1:yhCVgyC4o1eVCa2tZl7eS0r+SDo693bJlVdllGtEeKM= gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0= gopkg.in/check.v1 v1.0.0-20201130134442-10cb98267c6c h1:Hei/4ADfdWqJk1ZMxUNpqntNwaWcugrBjAiHlqqRiVk= gopkg.in/check.v1 v1.0.0-20201130134442-10cb98267c6c/go.mod h1:JHkPIbrfpd72SG/EVd6muEfDQjcINNoR0C8j2r3qZ4Q= @@ -87,3 +107,5 @@ gopkg.in/yaml.v3 v3.0.1 h1:fxVm/GzAzEWqLHuvctI91KS9hhNmmWOoWu0XTYJS7CA= gopkg.in/yaml.v3 v3.0.1/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM= mvdan.cc/sh/v3 v3.13.1 h1:DP3TfgZhDkT7lerUdnp6PTGKyxxzz6T+cOlY/xEvfWk= mvdan.cc/sh/v3 v3.13.1/go.mod h1:lXJ8SexMvEVcHCoDvAGLZgFJ9Wsm2sulmoNEXGhYZD0= +mvdan.cc/sh/v3 v3.14.1 h1:bXkhQWNHCs0KZEChF8hYS6FC+T2N9mUZLbQv9blditI= +mvdan.cc/sh/v3 v3.14.1/go.mod h1:syYCoFET8w9tvevxiXUtY8/ICrU+l26jHmhJDra3Vwo= diff --git a/gomod2nix.toml b/gomod2nix.toml index 8954af3..29d4284 100644 --- a/gomod2nix.toml +++ b/gomod2nix.toml @@ -46,28 +46,28 @@ schema = 3 hash = 'sha256-/FtmHnaGjdvEIKAJtrUfEhV7EVo5A/eYrtdnUkuxLDA=' [mod.'github.com/prometheus/client_golang'] - version = 'v1.23.2' - hash = 'sha256-3GD4fBFa1tJu8MS4TNP6r2re2eViUE+kWUaieIOQXCg=' + version = 'v1.24.1' + hash = 'sha256-HAOFVYyPiU7hVS1XXMMjkbGgTT7/UN0zVHXYOnjr7is=' [mod.'github.com/prometheus/client_model'] version = 'v0.6.2' hash = 'sha256-q6Fh6v8iNJN9ypD47LjWmx66YITa3FyRjZMRsuRTFeQ=' [mod.'github.com/prometheus/common'] - version = 'v0.69.0' - hash = 'sha256-rqui1G7KMtVtGXDFY7XpWD9tA721UtoPC+/JLh8axrA=' + version = 'v0.70.1' + hash = 'sha256-xlhVEswCWaBnAXn53KOUzquoDEnZVd/NcTbj52EJ6rE=' [mod.'github.com/prometheus/procfs'] - version = 'v0.20.1' - hash = 'sha256-L6RuVGYgdBCCke8BibA7cXgfwizGE2pA0gNo9dGJ8WQ=' + version = 'v0.21.1' + hash = 'sha256-5SpWprdX29oVntHJyadiICOk9UAP+hu+xJM51/A8Ig4=' [mod.'github.com/sagikazarmark/locafero'] version = 'v0.12.0' hash = 'sha256-EXk9S5Z5sYyApAzCgHIugsGMbt/pHWRfHYFZH5D+5Ws=' [mod.'github.com/sirupsen/logrus'] - version = 'v1.9.4' - hash = 'sha256-ltRvmtM3XTCAFwY0IesfRqYIivyXPPuvkFjL4ARh1wg=' + version = 'v1.10.2' + hash = 'sha256-1BKin1NpY1kFawnHkY0FxQ152aeB82RRKIDWJLctZB4=' [mod.'github.com/spf13/afero'] version = 'v1.15.0' @@ -86,8 +86,8 @@ schema = 3 hash = 'sha256-A9A8i7HH/ge4j3hw7G++HNj8BjhhpZKvxHhfY+QAxkI=' [mod.'github.com/stretchr/testify'] - version = 'v1.11.1' - hash = 'sha256-sWfjkuKJyDllDEtnM8sb/pdLzPQmUYWYtmeWz/5suUc=' + version = 'v1.12.1' + hash = 'sha256-9MTDdVjZMh1MJ5EH3HmAFrm24YqZayeMBgKwtaURZCc=' [mod.'github.com/subosito/gotenv'] version = 'v1.6.0' @@ -98,16 +98,16 @@ schema = 3 hash = 'sha256-ecT2ZXw7iT+63J4210xA6sMz0fUFXmDzLwZe2FzaNFU=' [mod.'go.yaml.in/yaml/v3'] - version = 'v3.0.4' - hash = 'sha256-NkGFiDPoCxbr3LFsI6OCygjjkY0rdmg5ggvVVwpyDQ4=' + version = 'v3.0.5' + hash = 'sha256-ygho+GU5kE7vPMx+dZYyNfCaeMjNxj66XmrcVf3afFE=' [mod.'golang.org/x/sys'] - version = 'v0.46.0' - hash = 'sha256-NzRXMSEk6upeudJvUEPVnw6clJ3d8UdC/vdfANWAc8g=' + version = 'v0.47.0' + hash = 'sha256-TpbRyWWqHjddP6QzUgAbaLd2EE0S+GYNRUIDJd18r98=' [mod.'golang.org/x/text'] - version = 'v0.38.0' - hash = 'sha256-PzREcn7yzTAJ0WBvyYL/2/r+tfoeKTY/E5HVXi/CJsU=' + version = 'v0.40.0' + hash = 'sha256-LJfnki46XEreGbSgjl+DeqgcTsINTOu2owyXNvijMcA=' [mod.'google.golang.org/protobuf'] version = 'v1.36.11' @@ -118,5 +118,5 @@ schema = 3 hash = 'sha256-FqL9TKYJ0XkNwJFnq9j0VvJ5ZUU1RvH/52h/f5bkYAU=' [mod.'mvdan.cc/sh/v3'] - version = 'v3.13.1' - hash = 'sha256-FM+2xpGELZLm4Gc09OE5Z3JeNLPMvzeb1wRcdAIeKy4=' + version = 'v3.14.1' + hash = 'sha256-+739dk4UfUQAJZd4uwOJ3LjBcuUG2v/wB81wmWB2Stk=' From ad571a7cc085f7708008da9a7030dee0234d112e Mon Sep 17 00:00:00 2001 From: comonad Date: Sun, 13 Sep 2026 23:13:29 +0800 Subject: [PATCH 05/15] feat(worker): cgroup v2 job isolation with cancellation and telemetry Replace RLIMIT_AS pre/post-hook utility with ephemeral cgroup v2 job isolation. Each sync attempt now runs in its own cgroup: memory.max is enforced with swap disabled and oom.group set, so a job exceeding its budget is killed atomically instead of silently spilling to swap. The child enters the cgroup atomically via clone3 (CgroupFD). Per-attempt telemetry (duration, memory.peak, oom_kill) is collected from the cgroup and accumulated into a per-repo EWMA, exposed via new Prometheus gauges (lug_job_*). --- config.example.yaml | 4 + pkg/exporter/exporter.go | 91 +++++++++++ pkg/manager/manager.go | 15 +- pkg/worker/cgroup_runner.go | 225 +++++++++++++++++++++++++++ pkg/worker/cgroup_runner_stub.go | 17 ++ pkg/worker/cgroup_test.go | 52 +++++++ pkg/worker/executor.go | 17 +- pkg/worker/executor_invoke_worker.go | 57 ++++++- pkg/worker/shell_script_executor.go | 122 +++++++++++---- pkg/worker/telemetry.go | 51 ++++++ pkg/worker/utilities.go | 6 - pkg/worker/utility_rlimit.go | 53 ------- pkg/worker/worker.go | 21 ++- pkg/worker/worker_test.go | 111 ++++++++----- 14 files changed, 706 insertions(+), 136 deletions(-) create mode 100644 pkg/worker/cgroup_runner.go create mode 100644 pkg/worker/cgroup_runner_stub.go create mode 100644 pkg/worker/cgroup_test.go create mode 100644 pkg/worker/telemetry.go delete mode 100644 pkg/worker/utilities.go delete mode 100644 pkg/worker/utility_rlimit.go diff --git a/config.example.yaml b/config.example.yaml index f08e32f..facc4c4 100644 --- a/config.example.yaml +++ b/config.example.yaml @@ -19,6 +19,10 @@ repos: script: rsync -av rsync://rsync.chiark.greenend.org.uk/ftp/users/sgtatham/putty-website-mirror/ /tmp/putty name: putty interval: 600 + # Optional per-job resource controls (enforced via cgroup v2 when a + # delegated subtree is available; ignored gracefully otherwise): + # rlimit_mem: 1G # cgroup memory.max for the job; the whole job tree is OOM-killed above it + # timeout: 7200 # wall-clock budget in seconds; the job tree is killed on expiry - type: shell_script script: bash -c 'printenv | grep ^LUG' name: printenv diff --git a/pkg/exporter/exporter.go b/pkg/exporter/exporter.go index d3b8365..751f72f 100644 --- a/pkg/exporter/exporter.go +++ b/pkg/exporter/exporter.go @@ -18,6 +18,12 @@ type Exporter struct { successCounter *prometheus.CounterVec failCounter *prometheus.CounterVec diskUsage *prometheus.GaugeVec + peakMem *prometheus.GaugeVec + peakMemEWMA *prometheus.GaugeVec + duration *prometheus.GaugeVec + durationEWMA *prometheus.GaugeVec + oomKills *prometheus.GaugeVec + admissions *prometheus.CounterVec // stores worker_name -> last time that updates its disk usage diskUsageLastUpdateTime map[string]time.Time // guard the exporter @@ -52,11 +58,59 @@ func newExporter() *Exporter { }, []string{"worker"}, ), + peakMem: prometheus.NewGaugeVec( + prometheus.GaugeOpts{ + Namespace: "lug", Subsystem: "job", Name: "peak_mem_bytes", + Help: "Peak memory (bytes) of the last sync run, partitioned by workers.", + }, + []string{"worker"}, + ), + peakMemEWMA: prometheus.NewGaugeVec( + prometheus.GaugeOpts{ + Namespace: "lug", Subsystem: "job", Name: "peak_mem_ewma_bytes", + Help: "EWMA of peak memory (bytes) across sync runs, partitioned by workers.", + }, + []string{"worker"}, + ), + duration: prometheus.NewGaugeVec( + prometheus.GaugeOpts{ + Namespace: "lug", Subsystem: "job", Name: "duration_seconds", + Help: "Wall time (seconds) of the last sync run, partitioned by workers.", + }, + []string{"worker"}, + ), + durationEWMA: prometheus.NewGaugeVec( + prometheus.GaugeOpts{ + Namespace: "lug", Subsystem: "job", Name: "duration_ewma_seconds", + Help: "EWMA of sync run wall time (seconds), partitioned by workers.", + }, + []string{"worker"}, + ), + oomKills: prometheus.NewGaugeVec( + prometheus.GaugeOpts{ + Namespace: "lug", Subsystem: "job", Name: "oom_kills_total", + Help: "Cumulative OOM kills observed for sync runs, partitioned by workers.", + }, + []string{"worker"}, + ), + admissions: prometheus.NewCounterVec( + prometheus.CounterOpts{ + Namespace: "lug", Subsystem: "admission", Name: "verdicts_total", + Help: "Admission verdicts, partitioned by outcome (admitted/deferred).", + }, + []string{"outcome"}, + ), diskUsageLastUpdateTime: map[string]time.Time{}, } prometheus.MustRegister(newExporter.successCounter) prometheus.MustRegister(newExporter.failCounter) prometheus.MustRegister(newExporter.diskUsage) + prometheus.MustRegister(newExporter.peakMem) + prometheus.MustRegister(newExporter.peakMemEWMA) + prometheus.MustRegister(newExporter.duration) + prometheus.MustRegister(newExporter.durationEWMA) + prometheus.MustRegister(newExporter.oomKills) + prometheus.MustRegister(newExporter.admissions) log.Info("Exporter initialized") return &newExporter } @@ -93,6 +147,43 @@ func (e *Exporter) SyncFail(worker string) { e.successCounter.With(prometheus.Labels{"worker": worker}).Add(0) } +// AdmissionVerdict reports one admission decision. +func (e *Exporter) AdmissionVerdict(admitted bool) { + e.mutex.Lock() + defer e.mutex.Unlock() + outcome := "deferred" + if admitted { + outcome = "admitted" + } + e.admissions.With(prometheus.Labels{"outcome": outcome}).Inc() +} + +// TelemetrySample carries per-run resource telemetry into the exporter. +// Defined here (not in pkg/worker) to avoid an import cycle. +type TelemetrySample struct { + LastDuration time.Duration + DurationEWMA time.Duration + LastPeakMemBytes uint64 + PeakMemEWMABytes uint64 + OOMKills uint64 +} + +// SyncTelemetry reports accumulated per-worker resource telemetry after a run. +func (e *Exporter) SyncTelemetry(workerName string, t TelemetrySample) { + e.mutex.Lock() + defer e.mutex.Unlock() + labels := prometheus.Labels{"worker": workerName} + if t.LastPeakMemBytes > 0 { + e.peakMem.With(labels).Set(float64(t.LastPeakMemBytes)) + e.peakMemEWMA.With(labels).Set(float64(t.PeakMemEWMABytes)) + } + if t.LastDuration > 0 { + e.duration.With(labels).Set(t.LastDuration.Seconds()) + e.durationEWMA.With(labels).Set(t.DurationEWMA.Seconds()) + } + e.oomKills.With(labels).Set(float64(t.OOMKills)) +} + // need at least 1min to rescan disk const updateDiskUsageThrottle time.Duration = time.Minute diff --git a/pkg/manager/manager.go b/pkg/manager/manager.go index f5e1e77..91dd2cf 100644 --- a/pkg/manager/manager.go +++ b/pkg/manager/manager.go @@ -55,6 +55,9 @@ type WorkerCheckPoint struct { LastInvokeTime time.Time `json:"last_invoke_time"` LastFinished *time.Time `json:"last_finished,omitempty"` Result *bool `json:"result,omitempty"` + // Telemetry persists learned resource usage (peak mem / duration EWMAs) + // across restarts. Additive; absent in v1 checkpoints. + Telemetry *worker.Telemetry `json:"telemetry,omitempty"` } type CheckPoint struct { @@ -101,7 +104,11 @@ func workerFromCheckpoint(repoConfig config.RepoConfig, checkpoint *CheckPoint, if info.LastFinished != nil { lastFinished = *info.LastFinished } - return worker.NewWorker(repoConfig, lastFinished, result) + telemetry := worker.Telemetry{} + if info.Telemetry != nil { + telemetry = *info.Telemetry + } + return worker.NewWorkerWithTelemetry(repoConfig, lastFinished, result, telemetry) } // NewManager creates a new manager with attached workers from config @@ -143,6 +150,10 @@ func NewManager(config *config.Config) (*Manager, error) { } func (m *Manager) checkpoint() error { + if m.config.Checkpoint == "" { + // checkpointing disabled (e.g. tests); learned telemetry only lives in memory + return nil + } ckptObj := &CheckPoint{WorkerInfo: make(map[string]WorkerCheckPoint)} for _, w := range m.workers { name := w.GetConfig()["name"].(string) @@ -152,10 +163,12 @@ func (m *Manager) checkpoint() error { lastInvokeTime = time.Now().AddDate(-1, 0, 0) } + telemetry := status.Telemetry ckptObj.WorkerInfo[name] = WorkerCheckPoint{ LastInvokeTime: lastInvokeTime, Result: &status.Result, LastFinished: &status.LastFinished, + Telemetry: &telemetry, } } diff --git a/pkg/worker/cgroup_runner.go b/pkg/worker/cgroup_runner.go new file mode 100644 index 0000000..da86e06 --- /dev/null +++ b/pkg/worker/cgroup_runner.go @@ -0,0 +1,225 @@ +//go:build linux + +package worker + +import ( + "bufio" + "errors" + "fmt" + "os" + "os/exec" + "path/filepath" + "strconv" + "strings" + "sync" + "syscall" + "time" + + log "github.com/sirupsen/logrus" +) + +// cgroup v2 job isolation. +// +// Each sync job runs in an ephemeral cgroup `/jobs/` created +// right before exec (the child enters it atomically via clone3+CgroupFD) and +// removed right after collection. This gives us: +// - enforceable per-job memory.max (replaces the racy RLIMIT_AS utility), +// - kill-the-whole-tree reclamation via cgroup.kill on ctx cancellation, +// - free telemetry: memory.peak and memory.events oom_kill per run. +// +// Requirements: cgroup v2 unified hierarchy and a delegated subtree (systemd +// service with Delegate=yes, or a container started with a private, writable +// cgroup namespace). When unavailable, executors fall back to plain process +// execution with rusage-based telemetry; see fallback paths in the executor. + +const cgroupMountpoint = "/sys/fs/cgroup" + +var ( + cgroupRootOnce sync.Once + cgroupJobsRoot string // "" means cgroup isolation unavailable +) + +// jobsCgroupRoot lazily prepares `/jobs` with the memory and pids +// controllers enabled. To satisfy the cgroup v2 "no internal processes" rule, +// all of lug's own processes are first moved into a `supervisor` leaf so that +// controllers can be enabled on the subtree. +// Returns "" if the environment does not support it. +func jobsCgroupRoot() string { + cgroupRootOnce.Do(func() { + root, err := setupJobsCgroup() + if err != nil { + log.WithField("event", "cgroup_unavailable").WithError(err). + Warn("cgroup v2 job isolation unavailable; falling back to plain process execution " + + "(no memory.max enforcement, rusage-only telemetry)") + return + } + log.WithField("event", "cgroup_jobs_root").WithField("path", root). + Info("cgroup v2 job isolation enabled") + cgroupJobsRoot = root + }) + return cgroupJobsRoot +} + +func setupJobsCgroup() (string, error) { + own, err := ownCgroupPath() + if err != nil { + return "", err + } + base := filepath.Join(cgroupMountpoint, own) + + // Ensure the unified hierarchy is really there and delegated to us. + if err := unixAccessWritable(filepath.Join(base, "cgroup.subtree_control")); err != nil { + return "", fmt.Errorf("cgroup subtree not delegated at %s: %w", base, err) + } + + // Move ourselves (and any sibling procs) to a leaf so controllers can be + // enabled on `base` (cgroup v2 forbids processes in non-leaf cgroups with + // enabled subtree controllers). + supervisor := filepath.Join(base, "supervisor") + if err := os.MkdirAll(supervisor, 0o755); err != nil { + return "", err + } + if err := moveAllProcs(base, supervisor); err != nil { + return "", fmt.Errorf("failed to move lug into supervisor leaf: %w", err) + } + + // memory and pids are required for enforcement and cleanup; + // cpu/io may be unavailable in unprivileged setups, so enable best-effort. + if err := os.WriteFile(filepath.Join(base, "cgroup.subtree_control"), + []byte("+memory +pids"), 0); err != nil { + return "", fmt.Errorf("failed to enable memory controller: %w", err) + } + _ = os.WriteFile(filepath.Join(base, "cgroup.subtree_control"), []byte("+cpu +io"), 0) + + jobs := filepath.Join(base, "jobs") + if err := os.MkdirAll(jobs, 0o755); err != nil { + return "", err + } + if err := os.WriteFile(filepath.Join(jobs, "cgroup.subtree_control"), + []byte("+memory +pids"), 0); err != nil { + return "", fmt.Errorf("failed to enable memory controller on jobs: %w", err) + } + return jobs, nil +} + +// ownCgroupPath returns this process's cgroup2 path relative to the mountpoint. +func ownCgroupPath() (string, error) { + f, err := os.Open("/proc/self/cgroup") + if err != nil { + return "", err + } + defer func() { _ = f.Close() }() + scanner := bufio.NewScanner(f) + for scanner.Scan() { + // cgroup v2 entry: "0::/path" + if rest, ok := strings.CutPrefix(scanner.Text(), "0::"); ok { + return strings.TrimSuffix(rest, " (deleted)"), nil + } + } + return "", errors.New("no cgroup v2 entry in /proc/self/cgroup") +} + +func unixAccessWritable(path string) error { + return syscall.Access(path, 0x2 /* W_OK */) +} + +func moveAllProcs(from, to string) error { + data, err := os.ReadFile(filepath.Join(from, "cgroup.procs")) + if err != nil { + return err + } + for _, pid := range strings.Fields(string(data)) { + if err := os.WriteFile(filepath.Join(to, "cgroup.procs"), []byte(pid), 0); err != nil { + return err + } + } + return nil +} + +// jobCgroup is one ephemeral per-run cgroup. +type jobCgroup struct { + dir string + fd int +} + +// newJobCgroup creates `/` with memory.max set to memMax +// (0 = unlimited) and returns a handle whose fd can be passed to +// SysProcAttr.CgroupFD. Returns (nil, nil) when cgroup isolation is +// unavailable (caller should fall back). +func newJobCgroup(name string, memMax uint64) (*jobCgroup, error) { + root := jobsCgroupRoot() + if root == "" { + return nil, nil + } + dir := filepath.Join(root, name) + if err := os.Mkdir(dir, 0o755); err != nil { + return nil, err + } + cleanup := func() { _ = os.Remove(dir) } + if memMax > 0 { + if err := os.WriteFile(filepath.Join(dir, "memory.max"), + []byte(strconv.FormatUint(memMax, 10)), 0); err != nil { + cleanup() + return nil, fmt.Errorf("failed to set memory.max: %w", err) + } + // A budget that silently spills to swap is not a budget: without this + // a job exceeding memory.max degrades the whole host via swap I/O + // instead of failing loudly. Best-effort (swap controller may be absent). + _ = os.WriteFile(filepath.Join(dir, "memory.swap.max"), []byte("0"), 0) + // Kill the whole job atomically on OOM instead of leaving a + // half-dead process tree behind. + _ = os.WriteFile(filepath.Join(dir, "memory.oom.group"), []byte("1"), 0) + } + fd, err := syscall.Open(dir, syscall.O_DIRECTORY|syscall.O_RDONLY|syscall.O_CLOEXEC, 0) + if err != nil { + cleanup() + return nil, err + } + return &jobCgroup{dir: dir, fd: fd}, nil +} + +// Kill terminates every process in the job cgroup (cgroup.kill, Linux >= 5.14). +func (j *jobCgroup) Kill() error { + return os.WriteFile(filepath.Join(j.dir, "cgroup.kill"), []byte("1"), 0) +} + +// Collect reads telemetry after the job's main process has exited. +func (j *jobCgroup) Collect() (peakMem uint64, oomKilled bool) { + if data, err := os.ReadFile(filepath.Join(j.dir, "memory.peak")); err == nil { + peakMem, _ = strconv.ParseUint(strings.TrimSpace(string(data)), 10, 64) + } + if data, err := os.ReadFile(filepath.Join(j.dir, "memory.events")); err == nil { + for _, line := range strings.Split(string(data), "\n") { + if cnt, ok := strings.CutPrefix(line, "oom_kill "); ok { + n, _ := strconv.ParseUint(strings.TrimSpace(cnt), 10, 64) + oomKilled = n > 0 + } + } + } + return +} + +// Close releases the fd and removes the cgroup. Descendant processes may +// still be draining after cgroup.kill, so removal is retried briefly. +func (j *jobCgroup) Close() { + _ = syscall.Close(j.fd) + for i := 0; i < 10; i++ { + if err := os.Remove(j.dir); err == nil || errors.Is(err, os.ErrNotExist) { + return + } + time.Sleep(100 * time.Millisecond) + } + log.WithField("event", "cgroup_remove_failed").WithField("dir", j.dir). + Warn("failed to remove job cgroup; it will leak until manual cleanup") +} + +// attachToCgroup makes cmd's child enter the job cgroup atomically at +// clone3 time (CLONE_INTO_CGROUP), so not a single instruction runs outside +// the resource-limited scope. +func attachToCgroup(cmd *exec.Cmd, cg *jobCgroup) { + if cmd.SysProcAttr == nil { + cmd.SysProcAttr = &syscall.SysProcAttr{} + } + cmd.SysProcAttr.UseCgroupFD = true + cmd.SysProcAttr.CgroupFD = cg.fd +} diff --git a/pkg/worker/cgroup_runner_stub.go b/pkg/worker/cgroup_runner_stub.go new file mode 100644 index 0000000..1f3ea82 --- /dev/null +++ b/pkg/worker/cgroup_runner_stub.go @@ -0,0 +1,17 @@ +//go:build !linux + +package worker + +import "os/exec" + +// Non-Linux stub: cgroup job isolation is Linux-only; executors fall back to +// plain subprocess execution with no memory enforcement or peak telemetry. + +type jobCgroup struct{} + +func newJobCgroup(name string, memMax uint64) (*jobCgroup, error) { return nil, nil } + +func (j *jobCgroup) Kill() error { return nil } +func (j *jobCgroup) Collect() (peakMem uint64, oom bool) { return 0, false } +func (j *jobCgroup) Close() {} +func attachToCgroup(cmd *exec.Cmd, cg *jobCgroup) {} diff --git a/pkg/worker/cgroup_test.go b/pkg/worker/cgroup_test.go new file mode 100644 index 0000000..5f77549 --- /dev/null +++ b/pkg/worker/cgroup_test.go @@ -0,0 +1,52 @@ +package worker + +import ( + "context" + "testing" + "time" + + "github.com/sirupsen/logrus" + "github.com/sjtug/lug/pkg/config" +) + +// TestCgroupEnforcement verifies memory.max enforcement and OOM telemetry. +// It skips unless run under a delegated cgroup subtree, e.g.: +// +// systemd-run --user --scope -p Delegate=yes go test -run TestCgroup ./pkg/worker/ +func TestCgroupEnforcement(t *testing.T) { + logrus.SetLevel(logrus.DebugLevel) + e, err := newShellScriptExecutor(config.RepoConfig{ + "name": "cg_mem_test", + "script": `python3 -c "x = bytearray(100*1024*1024); import time; time.sleep(1)"`, + "rlimit_mem": "20M", + }) + if err != nil { + t.Fatal(err) + } + result, err := e.RunOnce(context.Background(), logrus.WithField("repo", "cg_mem_test")) + t.Logf("err=%v peak=%d oom=%v dur=%v", err, result.PeakMemBytes, result.OOMKilled, result.Duration) + if result.PeakMemBytes == 0 { + t.Skip("cgroup delegation unavailable in this environment") + } + if err == nil || !result.OOMKilled { + t.Fatal("expected OOM kill under 20M memory.max") + } +} + +func TestCgroupTelemetryHappy(t *testing.T) { + e, err := newShellScriptExecutor(config.RepoConfig{ + "name": "cg_ok_test", + "script": "sleep 0.2", + }) + if err != nil { + t.Fatal(err) + } + result, err := e.RunOnce(context.Background(), logrus.WithField("repo", "cg_ok_test")) + if err != nil { + t.Fatal(err) + } + if result.Duration < 100*time.Millisecond { + t.Fatal("duration not measured") + } + t.Logf("peak=%d dur=%v", result.PeakMemBytes, result.Duration) +} diff --git a/pkg/worker/executor.go b/pkg/worker/executor.go index c61d4e4..a756184 100644 --- a/pkg/worker/executor.go +++ b/pkg/worker/executor.go @@ -1,14 +1,25 @@ package worker -import "github.com/sirupsen/logrus" +import ( + "context" + "time" + + "github.com/sirupsen/logrus" +) type execResult struct { Stdout string Stderr string + // Telemetry for this attempt. Zero values mean "unavailable" + // (e.g. cgroup isolation disabled). + Duration time.Duration + PeakMemBytes uint64 + OOMKilled bool } -// executor is a layer beneath worker, called by executorInvokeWorker +// executor is a layer beneath worker, called by executorInvokeWorker. +// ctx cancellation must terminate the whole job (including descendants). type executor interface { // When called, the executor performs sync for one time - RunOnce(logger *logrus.Entry, utilities []utility) (execResult, error) + RunOnce(ctx context.Context, logger *logrus.Entry) (execResult, error) } diff --git a/pkg/worker/executor_invoke_worker.go b/pkg/worker/executor_invoke_worker.go index 87e9b8b..f323eb3 100644 --- a/pkg/worker/executor_invoke_worker.go +++ b/pkg/worker/executor_invoke_worker.go @@ -1,6 +1,7 @@ package worker import ( + "context" "crypto/rand" "encoding/hex" "errors" @@ -21,13 +22,17 @@ type executorInvokeWorker struct { retry int retry_interval time.Duration lastFinished time.Time + telemetry Telemetry stdout *helper.MaxLengthStringSliceAdaptor stderr *helper.MaxLengthStringSliceAdaptor cfg config.RepoConfig name string signal chan int - logger *log.Entry - rwmutex sync.RWMutex + // cancelMu guards cancelRun, which aborts the in-flight sync (if any). + cancelMu sync.Mutex + cancelRun context.CancelFunc + logger *log.Entry + rwmutex sync.RWMutex } // creates a new executorInvokeWorker, which encapsules an executor @@ -46,6 +51,7 @@ func NewExecutorInvokeWorker(exector executor, status Status, retry: 3, retry_interval: 3 * time.Second, lastFinished: status.LastFinished, + telemetry: status.Telemetry, stdout: helper.NewMaxLengthSlice(status.Stdout, 20), stderr: helper.NewMaxLengthSlice(status.Stderr, 20), cfg: cfg, @@ -77,6 +83,17 @@ func (eiw *executorInvokeWorker) TriggerSync() { eiw.signal <- 1 } +// AbortSync cancels the in-flight sync run, if any. The cancellation +// propagates through the executor's context and (with cgroup isolation) +// kills the entire job process tree. +func (eiw *executorInvokeWorker) AbortSync() { + eiw.cancelMu.Lock() + defer eiw.cancelMu.Unlock() + if eiw.cancelRun != nil { + eiw.cancelRun() + } +} + func (eiw *executorInvokeWorker) GetStatus() Status { eiw.rwmutex.RLock() defer eiw.rwmutex.RUnlock() @@ -84,6 +101,7 @@ func (eiw *executorInvokeWorker) GetStatus() Status { Idle: eiw.idle, Result: eiw.result, LastFinished: eiw.lastFinished, + Telemetry: eiw.telemetry, Stdout: eiw.stdout.GetAll(), Stderr: eiw.stderr.GetAll(), } @@ -108,7 +126,7 @@ func (w *executorInvokeWorker) RunSync() { <-w.signal // A new sync run begins: stamp a sync_id so every line of this run - // can be correlated in the event stream (see PLAN, Phase 0). + // can be correlated in the event stream. syncID := shortID() logger = w.logger.WithField("sync_id", syncID) logger.WithField("event", "signal_received").Debug("finished waiting for signal") @@ -117,6 +135,12 @@ func (w *executorInvokeWorker) RunSync() { defer w.rwmutex.Unlock() w.idle = false }() + + ctx, cancel := context.WithCancel(context.Background()) + w.cancelMu.Lock() + w.cancelRun = cancel + w.cancelMu.Unlock() + logger.WithField("event", "start_execution").Info("sync started") retry_limit := w.retry var result execResult @@ -124,9 +148,9 @@ func (w *executorInvokeWorker) RunSync() { for retry_cnt := 1; retry_cnt <= retry_limit; retry_cnt++ { logger.WithField("event", "invoke_executor").WithField( "try_cnt", retry_cnt).Debugf("Invoke executor for the %v time", retry_cnt) - utilities := []utility{newRlimit(w)} - result, err = w.executor.RunOnce(logger, utilities) - if err == nil { + result, err = w.executor.RunOnce(ctx, logger) + if err == nil || ctx.Err() != nil { + // success, or the whole run was aborted: retrying is pointless break } logger.WithField("event", "invoke_executor_fail").WithField( @@ -135,6 +159,12 @@ func (w *executorInvokeWorker) RunSync() { logger.Debug("Stderr: ", result.Stderr) time.Sleep(w.retry_interval) } + + w.cancelMu.Lock() + w.cancelRun = nil + w.cancelMu.Unlock() + cancel() + if err != nil { logger.WithField("event", "execution_fail").Error(err.Error()) exporter.GetInstance().SyncFail(w.name) @@ -142,12 +172,14 @@ func (w *executorInvokeWorker) RunSync() { w.rwmutex.Lock() defer w.rwmutex.Unlock() w.result = false + w.telemetry.Observe(result.Duration, result.PeakMemBytes, result.OOMKilled) w.stdout.Put(result.Stdout) w.stderr.Put(result.Stderr) logger.Infof("Stderr: %s", result.Stderr) logger.Debugf("Stdout: %s", result.Stdout) w.idle = true }() + w.exportTelemetry() logger.WithField("event", "execution_ended").Info("sync ended") continue } @@ -163,11 +195,24 @@ func (w *executorInvokeWorker) RunSync() { w.stdout.Put(result.Stdout) w.result = true w.lastFinished = time.Now() + w.telemetry.Observe(result.Duration, result.PeakMemBytes, result.OOMKilled) }() + w.exportTelemetry() logger.WithField("event", "execution_ended").Info("sync ended") } } +func (w *executorInvokeWorker) exportTelemetry() { + t := w.GetStatus().Telemetry + exporter.GetInstance().SyncTelemetry(w.name, exporter.TelemetrySample{ + LastDuration: t.LastDuration, + DurationEWMA: t.DurationEWMA, + LastPeakMemBytes: t.LastPeakMem, + PeakMemEWMABytes: t.PeakMemEWMA, + OOMKills: t.OOMKills, + }) +} + // shortID generates a short unique ID without external dependencies. // It returns 4 bytes of crypto-random hex (8 chars), enough to correlate // all log lines within a single sync attempt without collision risk. diff --git a/pkg/worker/shell_script_executor.go b/pkg/worker/shell_script_executor.go index 0cf6e51..7442479 100644 --- a/pkg/worker/shell_script_executor.go +++ b/pkg/worker/shell_script_executor.go @@ -2,14 +2,17 @@ package worker import ( "bytes" + "context" "encoding/json" "errors" "fmt" "os" "os/exec" "strings" + "time" "github.com/davecgh/go-spew/spew" + "github.com/dustin/go-humanize" "github.com/sirupsen/logrus" "github.com/sjtug/lug/pkg/config" "mvdan.cc/sh/v3/shell" @@ -18,12 +21,35 @@ import ( // shellScriptExecutor implements executor interface type shellScriptExecutor struct { cfg config.RepoConfig + // memMax is the per-job memory.max in bytes (0 = unlimited), parsed + // from `rlimit_mem` and enforced via the job cgroup. + memMax uint64 + // timeout is the per-attempt wall clock budget (0 = no timeout), parsed + // from `timeout` (seconds). + timeout time.Duration } -func newShellScriptExecutor(cfg config.RepoConfig) *shellScriptExecutor { - return &shellScriptExecutor{ - cfg: cfg, +func newShellScriptExecutor(cfg config.RepoConfig) (*shellScriptExecutor, error) { + e := &shellScriptExecutor{cfg: cfg} + if raw, ok := cfg["rlimit_mem"]; ok { + s, ok := raw.(string) + if !ok { + return nil, errors.New("rlimit_mem should be a size string when present") + } + bytes_, err := humanize.ParseBytes(s) + if err != nil { + return nil, fmt.Errorf("invalid rlimit_mem: %w", err) + } + e.memMax = bytes_ + } + if raw, ok := cfg["timeout"]; ok { + sec, ok := raw.(int) + if !ok || sec < 0 { + return nil, errors.New("timeout should be a non-negative integer (seconds) when present") + } + e.timeout = time.Duration(sec) * time.Second } + return e, nil } func convertMapToEnvVars(m map[string]interface{}) (map[string]string, error) { @@ -54,7 +80,7 @@ func getOsEnvsAsMap() (result map[string]string) { envs := os.Environ() result = map[string]string{} for _, e := range envs { - pair := strings.Split(e, "=") + pair := strings.SplitN(e, "=", 2) key := pair[0] val := pair[1] result[key] = val @@ -62,11 +88,13 @@ func getOsEnvsAsMap() (result map[string]string) { return } -// RunSync launches the worker -func (w *shellScriptExecutor) RunOnce(logger *logrus.Entry, utilities []utility) (execResult, error) { +// RunOnce performs one sync attempt inside an ephemeral cgroup (when +// available): memory.max enforcement, whole-tree kill on ctx cancellation, +// and memory.peak / oom_kill telemetry are collected per run. +func (w *shellScriptExecutor) RunOnce(ctx context.Context, logger *logrus.Entry) (execResult, error) { script, ok := w.cfg["script"] if !ok { - return execResult{"", ""}, errors.New("script not found in config") + return execResult{}, errors.New("script not found in config") } // Split the command string into fields, respecting shell quoting rules @@ -74,53 +102,93 @@ func (w *shellScriptExecutor) RunOnce(logger *logrus.Entry, utilities []utility) return getOsEnvsAsMap()[name] }) if err != nil { - return execResult{"", ""}, fmt.Errorf("failed to parse command: %w", err) + return execResult{}, fmt.Errorf("failed to parse command: %w", err) } if len(fields) == 0 { - return execResult{"", ""}, errors.New("empty command") + return execResult{}, errors.New("empty command") + } + + if w.timeout > 0 { + var cancel context.CancelFunc + ctx, cancel = context.WithTimeout(ctx, w.timeout) + defer cancel() } logger.Debug("Invoking command:", fields[0], "with args:", fields[1:]) - cmd := exec.Command(fields[0], fields[1:]...) + cmd := exec.CommandContext(ctx, fields[0], fields[1:]...) + cmd.WaitDelay = 30 * time.Second // don't hang forever on inherited pipes // Forwarding config items to shell script as environmental variables // Adds a LUG_ prefix to their key env := os.Environ() envvars, err := convertMapToEnvVars(w.cfg) if err != nil { - return execResult{"", ""}, errors.New(fmt.Sprint("cannot convert w.cfg to env vars: ", err)) + return execResult{}, fmt.Errorf("cannot convert w.cfg to env vars: %w", err) } for k, v := range envvars { env = append(env, fmt.Sprintf("%s=%s", k, v)) } cmd.Env = env - for _, utility := range utilities { - logger.WithField("event", "exec_prehook").Debug("Executing prehook of ", utility) - if err := utility.preHook(); err != nil { - logger.Error("Failed to execute preHook:", err) - } - } - var bufErr, bufOut bytes.Buffer cmd.Stdout = &bufOut cmd.Stderr = &bufErr - err = cmd.Start() - - for _, utility := range utilities { - logger.WithField("event", "exec_posthook").Debug("Executing postHook of ", utility) - if err := utility.postHook(); err != nil { - logger.Error("Failed to execute postHook:", err) + // Ephemeral per-job cgroup; nil when the environment lacks cgroup v2 + // delegation, in which case the job runs unconfined (plain subprocess). + jobName := fmt.Sprintf("%v-%d", w.cfg["name"], time.Now().UnixNano()) + cg, err := newJobCgroup(jobName, w.memMax) + if err != nil { + logger.WithField("event", "job_cgroup_failed").WithError(err). + Warn("failed to create job cgroup; running unconfined") + } + if cg != nil { + defer cg.Close() + attachToCgroup(cmd, cg) + // Cancellation must reap the whole process tree, not just the direct + // child (rsync wrappers fork). cgroup.kill does exactly that. + cmd.Cancel = func() error { + logger.WithField("event", "job_cgroup_kill").Info("killing job cgroup") + return cg.Kill() } } + + start := time.Now() + err = cmd.Start() if err != nil { - return execResult{"", ""}, errors.New("execution cannot start") + return execResult{}, fmt.Errorf("execution cannot start: %w", err) } err = cmd.Wait() + + result := execResult{ + Stdout: bufOut.String(), + Stderr: bufErr.String(), + Duration: time.Since(start), + } + if cg != nil { + result.PeakMemBytes, result.OOMKilled = cg.Collect() + } + logger.WithFields(logrus.Fields{ + "event": "attempt_telemetry", + "duration_sec": result.Duration.Seconds(), + "peak_mem": result.PeakMemBytes, + "oom_killed": result.OOMKilled, + "ctx_err": ctx.Err(), + "cgroup_scoped": cg != nil, + }).Info("sync attempt finished") + if err != nil { - return execResult{bufOut.String(), bufErr.String()}, errors.New("execution failed") + switch { + case result.OOMKilled: + return result, fmt.Errorf("execution failed: killed by OOM (memory.max=%d)", w.memMax) + case errors.Is(ctx.Err(), context.DeadlineExceeded): + return result, fmt.Errorf("execution failed: timeout after %v", w.timeout) + case ctx.Err() != nil: + return result, fmt.Errorf("execution canceled: %w", ctx.Err()) + default: + return result, fmt.Errorf("execution failed: %w", err) + } } - return execResult{bufOut.String(), bufErr.String()}, nil + return result, nil } diff --git a/pkg/worker/telemetry.go b/pkg/worker/telemetry.go new file mode 100644 index 0000000..80d9375 --- /dev/null +++ b/pkg/worker/telemetry.go @@ -0,0 +1,51 @@ +package worker + +import "time" + +// ewmaAlpha is the smoothing factor for per-repo resource telemetry. +// 0.3 weights recent runs enough to track upstream growth while damping +// one-off spikes (e.g. a full re-sync after checkpoint loss). +const ewmaAlpha = 0.3 + +// Telemetry accumulates per-repo resource usage across sync runs. +// It is persisted in the manager checkpoint so learned budgets survive +// restarts, and feeds the admission controller's memory estimates. +type Telemetry struct { + // LastDuration is the wall time of the last completed attempt chain. + LastDuration time.Duration `json:"last_duration_ns,omitempty"` + // DurationEWMA is the exponentially weighted moving average of durations. + DurationEWMA time.Duration `json:"duration_ewma_ns,omitempty"` + // LastPeakMem is the peak memory (bytes) of the last run, when measurable. + LastPeakMem uint64 `json:"last_peak_mem,omitempty"` + // PeakMemEWMA is the EWMA of peak memory in bytes. + PeakMemEWMA uint64 `json:"peak_mem_ewma,omitempty"` + // OOMKills counts runs terminated by the kernel OOM killer (cgroup runner only). + OOMKills uint64 `json:"oom_kills,omitempty"` + // Runs counts observed runs contributing to the EWMAs. + Runs uint64 `json:"runs,omitempty"` +} + +// Observe folds one run's measurements into the telemetry. +// Zero-valued measurements (unavailable) do not disturb the EWMAs. +func (t *Telemetry) Observe(duration time.Duration, peakMem uint64, oomKilled bool) { + t.Runs++ + if duration > 0 { + t.LastDuration = duration + if t.DurationEWMA == 0 { + t.DurationEWMA = duration + } else { + t.DurationEWMA = time.Duration(ewmaAlpha*float64(duration) + (1-ewmaAlpha)*float64(t.DurationEWMA)) + } + } + if peakMem > 0 { + t.LastPeakMem = peakMem + if t.PeakMemEWMA == 0 { + t.PeakMemEWMA = peakMem + } else { + t.PeakMemEWMA = uint64(ewmaAlpha*float64(peakMem) + (1-ewmaAlpha)*float64(t.PeakMemEWMA)) + } + } + if oomKilled { + t.OOMKills++ + } +} diff --git a/pkg/worker/utilities.go b/pkg/worker/utilities.go deleted file mode 100644 index 6c0fc55..0000000 --- a/pkg/worker/utilities.go +++ /dev/null @@ -1,6 +0,0 @@ -package worker - -type utility interface { - preHook() error - postHook() error -} diff --git a/pkg/worker/utility_rlimit.go b/pkg/worker/utility_rlimit.go deleted file mode 100644 index 31336d6..0000000 --- a/pkg/worker/utility_rlimit.go +++ /dev/null @@ -1,53 +0,0 @@ -package worker - -import ( - "fmt" - "syscall" - - "github.com/dustin/go-humanize" -) - -type rlimit struct { - oldRlimit syscall.Rlimit - w Worker -} - -func newRlimit(w Worker) *rlimit { - return &rlimit{ - w: w, - } -} - -type rlimitError string - -func (re rlimitError) Error() string { - return string(re) -} - -func (r *rlimit) preHook() error { - cfg := r.w.GetConfig() - if err := syscall.Getrlimit(syscall.RLIMIT_AS, &r.oldRlimit); err != nil { - return rlimitError(fmt.Sprint("Failed to getrlimit:", err)) - } - if rlimitMem, ok := cfg["rlimit_mem"]; ok { - if bytes, err := humanize.ParseBytes(rlimitMem.(string)); err == nil { - rlimitNew := r.oldRlimit - rlimitNew.Cur = bytes - err := syscall.Setrlimit(syscall.RLIMIT_AS, &rlimitNew) - if err != nil { - return rlimitError(fmt.Sprint("Failed to setrlimit:", err)) - } - } else { - return rlimitError(fmt.Sprint("Invalid rlimit_mem: must be size:", err)) - } - } - return nil -} - -func (r *rlimit) postHook() error { - err := syscall.Setrlimit(syscall.RLIMIT_AS, &r.oldRlimit) - if err != nil { - return rlimitError(fmt.Sprint("Failed to restore rlimit:", err)) - } - return nil -} diff --git a/pkg/worker/worker.go b/pkg/worker/worker.go index d88ad68..8c145a7 100644 --- a/pkg/worker/worker.go +++ b/pkg/worker/worker.go @@ -19,12 +19,20 @@ type Worker interface { GetConfig() config.RepoConfig } +// Aborter is implemented by workers whose in-flight sync can be canceled. +type Aborter interface { + // AbortSync cancels the current sync run, if any. Thread-safe. + AbortSync() +} + // Status shows sync result and last timestamp. type Status struct { // Result is true if sync succeed, else false Result bool // LastFinished indicates last success time LastFinished time.Time + // Telemetry accumulates learned resource usage across runs + Telemetry Telemetry // Idle stands for whether worker is idle, false if syncing Idle bool // Last stdout(s) for admin. Internal implementation may vary to provide it in Status() @@ -35,17 +43,28 @@ type Status struct { // NewWorker generates a worker by config and log. func NewWorker(cfg config.RepoConfig, lastFinished time.Time, Result bool) (Worker, error) { + return NewWorkerWithTelemetry(cfg, lastFinished, Result, Telemetry{}) +} + +// NewWorkerWithTelemetry generates a worker restoring learned resource +// telemetry (e.g. from a checkpoint). +func NewWorkerWithTelemetry(cfg config.RepoConfig, lastFinished time.Time, Result bool, telemetry Telemetry) (Worker, error) { if syncType, ok := cfg["type"]; ok { switch syncType { case "rsync": return nil, errors.New("rsync worker has been removed since 0.10. " + "Use rsync.sh with shell_script worker at https://github.com/sjtug/mirror-docker instead") case "shell_script": + exec, err := newShellScriptExecutor(cfg) + if err != nil { + return nil, err + } w, err := NewExecutorInvokeWorker( - newShellScriptExecutor(cfg), + exec, Status{ Result: Result, LastFinished: lastFinished, + Telemetry: telemetry, Idle: true, Stdout: make([]string, 0), Stderr: make([]string, 0), diff --git a/pkg/worker/worker_test.go b/pkg/worker/worker_test.go index 1224bf1..de86016 100644 --- a/pkg/worker/worker_test.go +++ b/pkg/worker/worker_test.go @@ -1,8 +1,7 @@ package worker import ( - "io" - "os/exec" + "context" "strings" "testing" "time" @@ -100,9 +99,9 @@ type dummyExecutor struct { RunCnt int32 } -func (d *dummyExecutor) RunOnce(logger *logrus.Entry, utilities []utility) (execResult, error) { +func (d *dummyExecutor) RunOnce(ctx context.Context, logger *logrus.Entry) (execResult, error) { atomic.AddInt32(&d.RunCnt, 1) - return execResult{"", ""}, errors.New("dummy error") + return execResult{}, errors.New("dummy error") } func TestExecutorInvokeWorker(t *testing.T) { @@ -137,48 +136,82 @@ func TestExecutorInvokeWorker(t *testing.T) { asrt.Equal(2, int(atomic.LoadInt32(&d.RunCnt))) } -type limitReader struct { - cnt int - limit int +func TestShellScriptExecutorTimeout(t *testing.T) { + asrt := assert.New(t) + e, err := newShellScriptExecutor(config.RepoConfig{ + "name": "timeout_test", + "script": "sleep 30", + "timeout": 1, + }) + asrt.Nil(err) + start := time.Now() + result, err := e.RunOnce(context.Background(), logrus.WithField("repo", "timeout_test")) + asrt.NotNil(err) + asrt.Contains(err.Error(), "timeout") + asrt.Less(time.Since(start), 10*time.Second) + asrt.Greater(result.Duration, time.Duration(0)) } -func newLimitReader(limit int) *limitReader { - return &limitReader{ - cnt: 0, - limit: limit, - } +func TestShellScriptExecutorCancel(t *testing.T) { + asrt := assert.New(t) + e, err := newShellScriptExecutor(config.RepoConfig{ + "name": "cancel_test", + "script": "sleep 30", + }) + asrt.Nil(err) + ctx, cancel := context.WithCancel(context.Background()) + go func() { + time.Sleep(200 * time.Millisecond) + cancel() + }() + start := time.Now() + _, err = e.RunOnce(ctx, logrus.WithField("repo", "cancel_test")) + asrt.NotNil(err) + asrt.Less(time.Since(start), 10*time.Second) } -func (i *limitReader) Read(p []byte) (int, error) { - if i.cnt > i.limit { - return 0, io.EOF - } - i.cnt += len(p) - for i := 0; i < len(p); i++ { - p[i] = 5 // shouldn't use zero here, because sometimes pages filled with zero are not allocated - } - return len(p), nil + +func TestShellScriptExecutorTelemetry(t *testing.T) { + asrt := assert.New(t) + e, err := newShellScriptExecutor(config.RepoConfig{ + "name": "telemetry_test", + "script": "true", + }) + asrt.Nil(err) + result, err := e.RunOnce(context.Background(), logrus.WithField("repo", "telemetry_test")) + asrt.Nil(err) + asrt.Greater(result.Duration, time.Duration(0)) + // PeakMemBytes > 0 only when cgroup v2 delegation is available; do not assert. } -func TestUtilityRlimit(t *testing.T) { +func TestTelemetryObserve(t *testing.T) { asrt := assert.New(t) - external_worker, ok := NewExternalWorker(config.RepoConfig{ - "name": "test_worker", - "rlimit_mem": "10M", + var tel Telemetry + tel.Observe(10*time.Second, 1000, false) + asrt.Equal(10*time.Second, tel.DurationEWMA) + asrt.Equal(uint64(1000), tel.PeakMemEWMA) + tel.Observe(20*time.Second, 2000, true) + asrt.Equal(uint64(1), tel.OOMKills) + asrt.Equal(uint64(2), tel.Runs) + // EWMA moves toward the new sample but not all the way + asrt.Greater(tel.DurationEWMA, 10*time.Second) + asrt.Less(tel.DurationEWMA, 20*time.Second) + asrt.Greater(tel.PeakMemEWMA, uint64(1000)) + asrt.Less(tel.PeakMemEWMA, uint64(2000)) + // Zero samples must not disturb the EWMAs + prev := tel + tel.Observe(0, 0, false) + asrt.Equal(prev.DurationEWMA, tel.DurationEWMA) + asrt.Equal(prev.PeakMemEWMA, tel.PeakMemEWMA) +} + +func TestRlimitMemRejected(t *testing.T) { + asrt := assert.New(t) + _, err := newShellScriptExecutor(config.RepoConfig{ + "name": "bad_rlimit", + "script": "true", + "rlimit_mem": "not-a-size", }) - asrt.Nil(ok) - - rlimitUtility := newRlimit(external_worker) - - cmd := exec.Command("rev") - cmd.Stdin = newLimitReader(20000000) // > 10M = 10485760 - asrt.NoError(rlimitUtility.preHook()) - err1 := cmd.Start() - asrt.NoError(rlimitUtility.postHook()) - var err2 error - if err1 == nil { - err2 = cmd.Wait() - } - asrt.True(err1 != nil || err2 != nil) + asrt.NotNil(err) } func TestShellScriptWorkerArgParse(t *testing.T) { From 536183de3541095dc6f5b212512c2eea4fee10c2 Mon Sep 17 00:00:00 2001 From: comonad Date: Mon, 14 Sep 2026 00:52:40 +0800 Subject: [PATCH 06/15] feat(api): add endpoint to abort in-flight sync Expose POST /lug/v1/admin/worker/:name/abort. Cancellation propagates through the worker context and kills the complete cgroup process tree. --- pkg/manager/json_rest.go | 10 ++++++++++ pkg/manager/manager.go | 22 ++++++++++++++++++++++ 2 files changed, 32 insertions(+) diff --git a/pkg/manager/json_rest.go b/pkg/manager/json_rest.go index 9996e48..626179d 100644 --- a/pkg/manager/json_rest.go +++ b/pkg/manager/json_rest.go @@ -29,6 +29,7 @@ func (r *RestfulAPI) GetAPIHandler() http.Handler { rest.Get("/lug/v1/manager/summary", r.getManagerStatusSummary), rest.Post("/lug/v1/admin/manager/start", r.startManager), rest.Post("/lug/v1/admin/manager/stop", r.stopManager), + rest.Post("/lug/v1/admin/worker/:name/abort", r.abortWorker), rest.Delete("/lug/v1/admin/manager", r.exitManager), ) if err != nil { @@ -96,3 +97,12 @@ func (r *RestfulAPI) stopManager(w rest.ResponseWriter, req *rest.Request) { func (r *RestfulAPI) exitManager(w rest.ResponseWriter, req *rest.Request) { r.manager.Exit() } + +func (r *RestfulAPI) abortWorker(w rest.ResponseWriter, req *rest.Request) { + name := req.PathParam("name") + if err := r.manager.AbortWorker(name); err != nil { + rest.Error(w, err.Error(), http.StatusNotFound) + return + } + w.WriteHeader(http.StatusAccepted) +} diff --git a/pkg/manager/manager.go b/pkg/manager/manager.go index 91dd2cf..8e348fa 100644 --- a/pkg/manager/manager.go +++ b/pkg/manager/manager.go @@ -366,6 +366,28 @@ func (m *Manager) Exit() { m.expectChanVal(m.finishChan, ExitFinish) } +// AbortWorker cancels the in-flight sync of the named worker, if any. +// The cancellation propagates through the executor context and kills the +// whole job process tree when cgroup isolation is active. +func (m *Manager) AbortWorker(name string) error { + for _, w := range m.workers { + if w.GetConfig()["name"] != name { + continue + } + aborter, ok := w.(worker.Aborter) + if !ok { + return fmt.Errorf("worker %s does not support aborting", name) + } + m.logger.WithFields(logrus.Fields{ + "event": "abort_worker", + "target_worker_name": name, + }).Infof("aborting sync of worker %s", name) + aborter.AbortSync() + return nil + } + return fmt.Errorf("no worker named %s", name) +} + // GetStatus gets status of Manager func (m *Manager) GetStatus() *Status { status := Status{ From 3bdb88c356824e5a11f664f6c575d058de87d7d0 Mon Sep 17 00:00:00 2001 From: comonad Date: Mon, 14 Sep 2026 00:52:58 +0800 Subject: [PATCH 07/15] feat(admission): add memory-pressure job scheduling Gate queued launches on memory PSI and learned per-repo peak-memory EWMA, while retaining concurrent_limit as a hard cap. Grants reserve estimated memory until workers return idle --- config.example.yaml | 13 +++- pkg/admission/controller.go | 128 +++++++++++++++++++++++++++++++ pkg/admission/controller_test.go | 69 +++++++++++++++++ pkg/admission/probe.go | 77 +++++++++++++++++++ pkg/admission/probe_test.go | 40 ++++++++++ pkg/config/config.go | 13 ++++ pkg/manager/manager.go | 69 +++++++++++++---- 7 files changed, 394 insertions(+), 15 deletions(-) create mode 100644 pkg/admission/controller.go create mode 100644 pkg/admission/controller_test.go create mode 100644 pkg/admission/probe.go create mode 100644 pkg/admission/probe_test.go diff --git a/config.example.yaml b/config.example.yaml index facc4c4..94d75c6 100644 --- a/config.example.yaml +++ b/config.example.yaml @@ -1,10 +1,21 @@ interval: 3 # Interval between pollings loglevel: 5 # 1-5 -concurrent_limit: 1 # Maximum worker that can run at the same time +concurrent_limit: 1 # Hard cap on workers running at the same time # Prometheus metrics are exposed at http://exporter_address/metrics exporter_address: :8081 checkpoint: checkpoint.json +# Elastic admission control: gates job launches on host memory pressure and +# learned per-repo memory budgets (peak-memory EWMA from past runs). +# IO is deliberately not an admission signal: a mirror host is IO-saturated +# by design, and gating on it would starve the queue. +# Requires cgroup v2 with a delegated subtree for full effect; concurrent_limit +# above always applies as a hard cap on top. +admission: + enabled: false # false: only concurrent_limit applies + mem_reserve_fraction: 0.2 # keep this fraction of available memory free + mem_psi_threshold: 15 # defer launches when memory some/avg10 exceeds this + #logstash: # address: listener.logz.io:5050 # logstash sink. Lug will send all logs to this address # additional_fields: diff --git a/pkg/admission/controller.go b/pkg/admission/controller.go new file mode 100644 index 0000000..dbea5be --- /dev/null +++ b/pkg/admission/controller.go @@ -0,0 +1,128 @@ +package admission + +import ( + "sync" + + log "github.com/sirupsen/logrus" +) + +// DefaultMemEstimate is assumed for jobs with no learned telemetry yet. +// Deliberately conservative: unknown jobs should not be admitted in bulk +// on a loaded host, and one successful run replaces the guess. +const DefaultMemEstimate = 512 << 20 + +// EstimateHeadroom scales the learned peak-memory EWMA into a budget, +// absorbing run-to-run variance and upstream growth. +const EstimateHeadroom = 1.5 + +// Config tunes the admission controller. Zero values select defaults. +type Config struct { + // MemReserveFraction of MemAvailable is kept free (default 0.2). + MemReserveFraction float64 + // MemPSIThreshold defers admission when memory "some avg10" exceeds it (default 15). + MemPSIThreshold float64 +} + +func (c Config) withDefaults() Config { + if c.MemReserveFraction <= 0 { + c.MemReserveFraction = 0.2 + } + if c.MemPSIThreshold <= 0 { + c.MemPSIThreshold = 15 + } + return c +} + +// Request describes one job asking to start. +type Request struct { + Name string + // MemEstimate is the expected peak memory in bytes (0 = unknown). + MemEstimate uint64 +} + +// Verdict is the outcome of an admission attempt. +type Verdict struct { + Admit bool + // Reason explains a deferral (empty when admitted). + Reason string +} + +// Controller gates job launches on host pressure and learned memory budgets. +// It complements (does not replace) the manager's static concurrency cap: +// when host signals are unavailable, everything is admitted and only the +// static cap applies. Callers must pair every admitted TryAdmit with a Release. +type Controller struct { + cfg Config + probe func() Signals + + mu sync.Mutex + granted map[string]uint64 + logger *log.Entry +} + +func NewController(cfg Config) *Controller { + return &Controller{ + cfg: cfg.withDefaults(), + probe: Probe, + granted: make(map[string]uint64), + logger: log.WithField("component", "admission"), + } +} + +// Estimate derives a job's memory budget from learned telemetry. +func Estimate(peakMemEWMA uint64) uint64 { + if peakMemEWMA == 0 { + return DefaultMemEstimate + } + return uint64(float64(peakMemEWMA) * EstimateHeadroom) +} + +// TryAdmit decides whether the job may start now. An admitted job holds its +// budget until Release(name) is called. +func (c *Controller) TryAdmit(req Request) Verdict { + c.mu.Lock() + defer c.mu.Unlock() + + if req.MemEstimate == 0 { + req.MemEstimate = DefaultMemEstimate + } + verdict := c.decide(req) + if verdict.Admit { + c.granted[req.Name] = req.MemEstimate + } + c.logger.WithFields(log.Fields{ + "event": "admission_verdict", + "job": req.Name, + "estimate": req.MemEstimate, + "admit": verdict.Admit, + "reason": verdict.Reason, + "running": len(c.granted), + }).Info("admission verdict") + return verdict +} + +func (c *Controller) decide(req Request) Verdict { + signals := c.probe() + if !signals.Valid { + return Verdict{Admit: true} + } + if signals.MemPSI > c.cfg.MemPSIThreshold { + return Verdict{Reason: "memory pressure too high"} + } + var grantedSum uint64 + for _, g := range c.granted { + grantedSum += g + } + reserve := uint64(float64(signals.MemAvailableBytes) * c.cfg.MemReserveFraction) + if grantedSum+req.MemEstimate+reserve > signals.MemAvailableBytes { + return Verdict{Reason: "insufficient memory headroom"} + } + return Verdict{Admit: true} +} + +// Release returns a job's granted budget to the pool. +func (c *Controller) Release(name string) { + c.mu.Lock() + defer c.mu.Unlock() + delete(c.granted, name) +} diff --git a/pkg/admission/controller_test.go b/pkg/admission/controller_test.go new file mode 100644 index 0000000..7c39e25 --- /dev/null +++ b/pkg/admission/controller_test.go @@ -0,0 +1,69 @@ +package admission + +import ( + "testing" + + "github.com/stretchr/testify/assert" +) + +func fixedProbe(s Signals) func() Signals { + return func() Signals { return s } +} + +func newTestController(cfg Config, s Signals) *Controller { + c := NewController(cfg) + c.probe = fixedProbe(s) + return c +} + +func TestAdmitWithHeadroom(t *testing.T) { + asrt := assert.New(t) + c := newTestController(Config{}, Signals{ + MemAvailableBytes: 8 << 30, + Valid: true, + }) + v := c.TryAdmit(Request{Name: "a", MemEstimate: 1 << 30}) + asrt.True(v.Admit) +} + +func TestDeferOnMemoryPressure(t *testing.T) { + asrt := assert.New(t) + c := newTestController(Config{}, Signals{ + MemAvailableBytes: 8 << 30, + MemPSI: 50, + Valid: true, + }) + v := c.TryAdmit(Request{Name: "a", MemEstimate: 1 << 30}) + asrt.False(v.Admit) + asrt.Contains(v.Reason, "memory pressure") +} + +func TestDeferOnInsufficientMemoryAndRelease(t *testing.T) { + asrt := assert.New(t) + c := newTestController(Config{}, Signals{ + MemAvailableBytes: 4 << 30, + Valid: true, + }) + // reserve = 0.2*4G; first job takes 2G, second 2G won't fit + v1 := c.TryAdmit(Request{Name: "a", MemEstimate: 2 << 30}) + asrt.True(v1.Admit) + v2 := c.TryAdmit(Request{Name: "b", MemEstimate: 2 << 30}) + asrt.False(v2.Admit) + asrt.Contains(v2.Reason, "memory headroom") + + c.Release("a") + v3 := c.TryAdmit(Request{Name: "b", MemEstimate: 2 << 30}) + asrt.True(v3.Admit) +} + +func TestInvalidSignalsAdmit(t *testing.T) { + asrt := assert.New(t) + c := newTestController(Config{}, Signals{}) + asrt.True(c.TryAdmit(Request{Name: "a"}).Admit) +} + +func TestEstimate(t *testing.T) { + asrt := assert.New(t) + asrt.Equal(uint64(DefaultMemEstimate), Estimate(0)) + asrt.Equal(uint64(1500), Estimate(1000)) +} diff --git a/pkg/admission/probe.go b/pkg/admission/probe.go new file mode 100644 index 0000000..51fedba --- /dev/null +++ b/pkg/admission/probe.go @@ -0,0 +1,77 @@ +package admission + +import ( + "os" + "strconv" + "strings" +) + +// Signals is a snapshot of host resource pressure used for admission decisions. +type Signals struct { + // MemAvailableBytes is MemAvailable from /proc/meminfo. + MemAvailableBytes uint64 + // MemPSI is the "some avg10" percentage from /proc/pressure/memory (0-100). + MemPSI float64 + // Valid is false when host signals could not be read; admission then + // falls back to the static concurrency gate. + Valid bool +} + +// Probe reads current host signals. PSI files may be absent (kernel without +// CONFIG_PSI); missing PSI reads as zero pressure while memory info is required. +func Probe() Signals { + s := Signals{} + mem, ok := readMemAvailable("/proc/meminfo") + if !ok { + return s + } + s.MemAvailableBytes = mem + s.MemPSI = readPSISomeAvg10("/proc/pressure/memory") + s.Valid = true + return s +} + +func readMemAvailable(path string) (uint64, bool) { + data, err := os.ReadFile(path) + if err != nil { + return 0, false + } + for _, line := range strings.Split(string(data), "\n") { + rest, ok := strings.CutPrefix(line, "MemAvailable:") + if !ok { + continue + } + fields := strings.Fields(rest) + if len(fields) < 1 { + return 0, false + } + kb, err := strconv.ParseUint(fields[0], 10, 64) + if err != nil { + return 0, false + } + return kb * 1024, true + } + return 0, false +} + +func readPSISomeAvg10(path string) float64 { + data, err := os.ReadFile(path) + if err != nil { + return 0 + } + for _, line := range strings.Split(string(data), "\n") { + if !strings.HasPrefix(line, "some ") { + continue + } + for _, field := range strings.Fields(line) { + if val, ok := strings.CutPrefix(field, "avg10="); ok { + f, err := strconv.ParseFloat(val, 64) + if err != nil { + return 0 + } + return f + } + } + } + return 0 +} diff --git a/pkg/admission/probe_test.go b/pkg/admission/probe_test.go new file mode 100644 index 0000000..b6ccf5b --- /dev/null +++ b/pkg/admission/probe_test.go @@ -0,0 +1,40 @@ +package admission + +import ( + "os" + "path/filepath" + "testing" + + "github.com/stretchr/testify/assert" +) + +func TestReadMemAvailable(t *testing.T) { + asrt := assert.New(t) + path := filepath.Join(t.TempDir(), "meminfo") + asrt.NoError(os.WriteFile(path, []byte( + "MemTotal: 32663396 kB\nMemFree: 1109108 kB\nMemAvailable: 17972072 kB\n"), 0o644)) + mem, ok := readMemAvailable(path) + asrt.True(ok) + asrt.Equal(uint64(17972072*1024), mem) + + _, ok = readMemAvailable(filepath.Join(t.TempDir(), "missing")) + asrt.False(ok) +} + +func TestReadPSISomeAvg10(t *testing.T) { + asrt := assert.New(t) + path := filepath.Join(t.TempDir(), "psi") + asrt.NoError(os.WriteFile(path, []byte( + "some avg10=75.83 avg60=85.71 avg300=91.47 total=13174650506\n"+ + "full avg10=72.90 avg60=83.80 avg300=90.37 total=12876717463\n"), 0o644)) + asrt.InDelta(75.83, readPSISomeAvg10(path), 0.001) + asrt.Zero(readPSISomeAvg10(filepath.Join(t.TempDir(), "missing"))) +} + +func TestProbeOnRealHost(t *testing.T) { + s := Probe() + if !s.Valid { + t.Skip("/proc/meminfo unavailable") + } + assert.Greater(t, s.MemAvailableBytes, uint64(0)) +} diff --git a/pkg/config/config.go b/pkg/config/config.go index a6c4a2b..7477651 100644 --- a/pkg/config/config.go +++ b/pkg/config/config.go @@ -27,6 +27,17 @@ type LogStashConfig struct { AdditionalFields map[string]interface{} `mapstructure:"additional_fields"` } +// AdmissionConfig tunes the elastic admission controller. +type AdmissionConfig struct { + // Enabled switches pressure-based admission on. When false, only the + // static concurrent_limit gate applies. + Enabled bool + // MemReserveFraction of available memory kept free (default 0.2). + MemReserveFraction float64 `mapstructure:"mem_reserve_fraction"` + // MemPSIThreshold defers launches when memory some/avg10 exceeds it (default 15). + MemPSIThreshold float64 `mapstructure:"mem_psi_threshold"` +} + // Config stores all configuration of lug type Config struct { // Interval between pollings in manager @@ -43,6 +54,8 @@ type Config struct { JsonAPIConfig JsonAPIConfig `mapstructure:"json_api"` // Worker sync checkpoint path Checkpoint string `mapstructure:"checkpoint"` + // Admission tunes the elastic admission controller + Admission AdmissionConfig `mapstructure:"admission"` // Config for each repo is represented as an array of RepoConfig. Nested structure is disallowed Repos []RepoConfig // A dummy section that will not be used in our program. diff --git a/pkg/manager/manager.go b/pkg/manager/manager.go index 8e348fa..c203b36 100644 --- a/pkg/manager/manager.go +++ b/pkg/manager/manager.go @@ -12,7 +12,9 @@ import ( "github.com/sirupsen/logrus" + "github.com/sjtug/lug/pkg/admission" "github.com/sjtug/lug/pkg/config" + "github.com/sjtug/lug/pkg/exporter" "github.com/sjtug/lug/pkg/worker" ) @@ -41,7 +43,12 @@ type Manager struct { running bool // storing index of worker to launch pendingQueue []int - logger *logrus.Entry + // admission gates launches on host pressure; nil when disabled + admission *admission.Controller + // hasGrant tracks workers holding an admission grant, so the poll loop + // can release budgets once a worker returns to idle + hasGrant map[string]bool + logger *logrus.Entry } // Status holds the status of a manager and its workers @@ -123,6 +130,13 @@ func NewManager(config *config.Config) (*Manager, error) { workersLastInvokeTime[name] = info.LastInvokeTime } } + var admissionCtl *admission.Controller + if config.Admission.Enabled { + admissionCtl = admission.NewController(admission.Config{ + MemReserveFraction: config.Admission.MemReserveFraction, + MemPSIThreshold: config.Admission.MemPSIThreshold, + }) + } newManager := Manager{ config: config, workers: []worker.Worker{}, @@ -130,6 +144,8 @@ func NewManager(config *config.Config) (*Manager, error) { controlChan: make(chan int), finishChan: make(chan int), running: true, + admission: admissionCtl, + hasGrant: make(map[string]bool), logger: logger, } for _, repoConfig := range config.Repos { @@ -197,34 +213,53 @@ func (m *Manager) isAlreadyInPendingQueue(workerIdx int) bool { return false } +// launchWorkerFromPendingQueue starts queued workers in FIFO order. +// max_allowed is the static concurrent_limit gate; when admission control is +// enabled, each launch must also pass the controller, which checks host +// pressure and learned memory budgets. On the first deferral the drain stops +// (FIFO, no overtaking), and the remaining workers retry next poll tick. func (m *Manager) launchWorkerFromPendingQueue(max_allowed int) { if max_allowed <= 0 { return } - var new_idx int - if max_allowed > len(m.pendingQueue) { - new_idx = len(m.pendingQueue) - } else { - new_idx = max_allowed - } m.logger.WithFields(logrus.Fields{ "event": "launch_worker_from_pending_queue", "max_allowed": max_allowed, - "new_idx": new_idx, "pending_queue": spew.Sprint(m.pendingQueue), }).Debug("launch worker from pending queue") - to_launch := m.pendingQueue[:new_idx] - m.pendingQueue = m.pendingQueue[new_idx:] - for _, w_idx := range to_launch { + launched := 0 + for launched < max_allowed && len(m.pendingQueue) > 0 { + w_idx := m.pendingQueue[0] w := m.workers[w_idx] wConfig := w.GetConfig() + name := wConfig["name"].(string) + + if m.admission != nil { + verdict := m.admission.TryAdmit(admission.Request{ + Name: name, + MemEstimate: admission.Estimate(w.GetStatus().Telemetry.PeakMemEWMA), + }) + exporter.GetInstance().AdmissionVerdict(verdict.Admit) + if !verdict.Admit { + m.logger.WithFields(logrus.Fields{ + "event": "admission_deferred", + "target_worker_name": name, + "reason": verdict.Reason, + }).Infof("admission deferred for worker %s: %s", name, verdict.Reason) + break + } + m.hasGrant[name] = true + } + + m.pendingQueue = m.pendingQueue[1:] m.logger.WithFields(logrus.Fields{ "event": "trigger_sync", - "target_worker_name": wConfig["name"], - }).Infof("trigger sync for worker %s from pendingQueue", wConfig["name"]) - m.workersLastInvokeTime[wConfig["name"].(string)] = time.Now() + "target_worker_name": name, + }).Infof("trigger sync for worker %s from pendingQueue", name) + m.workersLastInvokeTime[name] = time.Now() w.TriggerSync() + launched++ } } @@ -268,6 +303,12 @@ func (m *Manager) Run() { continue } wConfig := w.GetConfig() + if name, ok := wConfig["name"].(string); ok && m.hasGrant[name] { + if m.admission != nil { + m.admission.Release(name) + } + delete(m.hasGrant, name) + } elapsed := time.Since(m.workersLastInvokeTime[wConfig["name"].(string)]) sec2sync, ok := wConfig["interval"].(int) if !ok { From 5e2375afee5a21735eae388cc2673987f8dd9756 Mon Sep 17 00:00:00 2001 From: comonad Date: Mon, 14 Sep 2026 01:00:54 +0800 Subject: [PATCH 08/15] docs: update README with observability, cgroup v2 setup, and API reference Document all REST endpoints including the new abort endpoint, Prometheus metrics (sync counters, resource telemetry gauges, admission verdicts), structured log events, cgroup v2 requirements for rootful Docker / rootless Podman / bare systemd, elastic admission control, per-repo resource controls, development workflow with cgroup-aware test execution, and the manual sync triggering workaround. --- README.md | 300 +++++++++++++++++++++++++++++++++++++++++++++++------- 1 file changed, 261 insertions(+), 39 deletions(-) diff --git a/README.md b/README.md index 27c419f..3a66f2e 100644 --- a/README.md +++ b/README.md @@ -1,57 +1,279 @@ -# lug +# lug + [![release](https://img.shields.io/github/release/sjtug/lug.svg)](https://github.com/sjtug/lug/releases) [![Go Report Card](https://goreportcard.com/badge/github.com/sjtug/lug)](https://goreportcard.com/report/github.com/sjtug/lug) -[![Build Status](https://travis-ci.org/sjtug/lug.svg)](https://travis-ci.org/sjtug/lug) -[![Docker pulls](https://img.shields.io/docker/pulls/htfy96/lug.svg)](https://hub.docker.com/r/htfy96/lug/) -[![Apache License](https://img.shields.io/github/license/sjtug/lug.svg)](https://github.com/sjtug/lug/blob/master/LICENSE) -Extensible backend of software mirror. Read our [Wiki](https://github.com/sjtug/lug/wiki) for usage and guides for developmenet. +Extensible backend of software mirror. + +## Quick start + +```sh +docker run -d \ + -v /srv/mirror:/data \ + -v $(pwd)/config.yaml:/app/config.yaml \ + -p 8081:8081 -p 7001:7001 \ + sjtug/lug -c /app/config.yaml +``` + +Refer to [`config.example.yaml`](config.example.yaml) for available options. + +### Ports + +| Port | Purpose | +|------|---------| +| 8081 | Prometheus metrics (`/metrics`) | +| 7001 | JSON control API | + +## Configuration + +See [`config.example.yaml`](config.example.yaml) for a full annotated example. +Key settings: + +| Key | Description | +|-----|-------------| +| `interval` | Seconds between scheduler poll ticks | +| `concurrent_limit` | Hard cap on simultaneously running sync jobs | +| `checkpoint` | Path to persist worker state across restarts | +| `exporter_address` | Bind address for Prometheus metrics | +| `json_api.address` | Bind address for the JSON control API | + +### Per-repo options + +Each entry under `repos` configures one mirror worker. Shell-script workers +accept these additional fields: + +| Key | Description | +|-----|-------------| +| `rlimit_mem` | Memory budget per job (e.g. `1G`). Enforced via cgroup `memory.max`; the entire job tree is OOM-killed when exceeded. Ignored gracefully when cgroup isolation is unavailable. | +| `timeout` | Wall-clock budget in seconds. The job tree is killed on expiry. | +| `interval` | Seconds between syncs for this repo. If unset, the repo is synced only at startup. | + +### Elastic admission control + +When `admission.enabled` is `true`, lug gates queued job launches on: + +- **Memory PSI** — defers when memory `some/avg10` exceeds the threshold. +- **Memory headroom** — reserves a fraction of available memory and checks + that the sum of granted budgets plus the new estimate fits. + +Estimates are derived from per-repo peak-memory EWMAs learned over past runs. +Unknown repos assume a conservative default (512 MiB) until measured. The +static `concurrent_limit` always applies as a hard cap on top. + +IO pressure is deliberately **not** an admission signal: a mirror host is +IO-saturated by design, and gating on it would permanently starve the queue. + +## Cgroup v2 setup + +Lug uses cgroup v2 to enforce per-job memory limits, kill entire process trees +on cancellation or timeout, and collect resource telemetry (peak memory, OOM +kills, duration). **This is optional**: when cgroup isolation is unavailable, +lug falls back to plain subprocess execution with a logged warning, and +`rlimit_mem` / admission memory estimates are simply not enforced. + +### Requirements + +1. **Linux host with cgroup v2 unified hierarchy** (`/sys/fs/cgroup` mounted + as `cgroup2`). This is the default on all modern distros (Debian 11+, + Ubuntu 22.04+, Fedora 31+, RHEL 9+, NixOS 20.09+). +2. **Delegated cgroup subtree** — lug must be able to create child cgroups and + write controller knobs (`memory.max`, `cgroup.kill`, etc.) under its own + cgroup. How to achieve this depends on the runtime. + +### Rootful Docker (Docker Engine with root daemon) + +Docker ≥ 20.10 supports `--cgroupns=private` (the default since Docker 23), +which gives each container its own cgroup namespace and a writable cgroup root. +Run lug with: + +```sh +docker run -d \ + --cgroupns=private \ + -v /srv/mirror:/data \ + -v $(pwd)/config.yaml:/app/config.yaml \ + sjtug/lug -c /app/config.yaml +``` + +This is sufficient — no `--privileged`, no extra capabilities, no bind-mounts +of `/sys/fs/cgroup`. The container sees its own cgroup subtree as `/` and can +write to it freely. + +> **Verify**: inside the container, `cat /sys/fs/cgroup/cgroup.controllers` +> should list `memory` (and ideally `pids`). If it shows an empty file or +> the path does not exist, the host likely uses cgroup v1. + +### Rootless Podman + +Rootless Podman ≥ 4.0 delegates a user-owned cgroup subtree by default when +the host runs systemd ≥ 247 and cgroup v2. No extra flags are needed: + +```sh +podman run -d \ + -v /srv/mirror:/data \ + -v $(pwd)/config.yaml:/app/config.yaml \ + docker.io/sjtug/lug -c /app/config.yaml +``` + +### systemd service (bare metal / VM) + +If running lug as a systemd service without containers: + +```ini +[Service] +ExecStart=/usr/bin/lug -c /etc/lug/config.yaml +Delegate=yes +``` + +`Delegate=yes` permits lug to manage cgroups under its service scope. + +### Disabling cgroup isolation -## Use it in docker +If cgroup v2 is unavailable (cgroup v1 host, restricted container, etc.), lug +detects this at startup and logs a warning: + +``` +level=warning msg="cgroup v2 job isolation unavailable; falling back to plain process execution (no memory.max enforcement, rusage-only telemetry)" ``` -docker run -d -v {{host_path}}:{{docker_path}} -v {{absolute_path_of_config.yaml}}:/go/src/github.com/sjtug/lug/config.yaml htfy96/lug {other args...} + +No configuration change is needed — the fallback is automatic. `rlimit_mem` +and `timeout` settings in the config are silently ignored when enforcement is +unavailable; admission control will operate without memory estimates. + +## Observability + +### Prometheus metrics + +Exposed at `http:///metrics` (default `:8081`). + +**Sync counters** (labels: `worker`): + +| Metric | Description | +|--------|-------------| +| `success_sync` | Successful sync runs (counter) | +| `fail_sync` | Failed sync runs (counter) | + +**Resource telemetry** (labels: `worker`) — populated when cgroup isolation is +active: + +| Metric | Description | +|--------|-------------| +| `lug_job_peak_mem_bytes` | Peak memory (bytes) of the last sync run | +| `lug_job_peak_mem_ewma_bytes` | EWMA of peak memory across runs | +| `lug_job_duration_seconds` | Wall time (seconds) of the last sync run | +| `lug_job_duration_ewma_seconds` | EWMA of sync run wall time | +| `lug_job_oom_kills_total` | Cumulative OOM kills observed | + +**Admission verdicts** (labels: `outcome = admitted | deferred`): + +| Metric | Description | +|--------|-------------| +| `lug_admission_verdicts_total` | Admission decisions (counter) | + +**Disk usage** (labels: `worker`): + +| Metric | Description | +|--------|-------------| +| `lug_disk_usage` | Disk usage in bytes | + +### Structured logs + +All log lines are JSON with `repo` and `sync_id` fields. Key events: + +| Event | Description | +|-------|-------------| +| `attempt_telemetry` | Emitted after every sync attempt with `duration_sec`, `peak_mem`, `oom_killed`, `cgroup_scoped` | +| `trigger_sync` | A queued worker is launched | +| `admission_deferred` | Admission controller deferred a launch (`reason` field explains why) | +| `job_cgroup_kill` | Cancellation is reaping the job's cgroup tree | +| `abort_worker` | An operator aborted a sync via the REST API | + +## JSON control API + +Served at `json_api.address` (default `:7001`). All responses are JSON. + +### Endpoints + +#### `GET /lug/v1/manager/summary` + +Worker status overview (public, safe for dashboards). + +```json +{ + "Running": true, + "WorkerStatus": { + "putty": { "Result": true, "LastFinished": "2025-01-01T00:00:00Z", "Idle": true } + } +} ``` -### config.yaml +#### `GET /lug/v1/admin/manager/detail` + +Full status including per-worker stdout/stderr history and telemetry. + +#### `POST /lug/v1/admin/manager/start` + +Resume the scheduler after a stop. Workers whose interval has elapsed are +queued on the next poll tick. + +#### `POST /lug/v1/admin/manager/stop` + +Pause the scheduler. Already-running syncs finish, but no new syncs are +launched. + +#### `POST /lug/v1/admin/worker/:name/abort` -The below configuration may be outdated. Refer to [config.example.yaml](https://github.com/sjtug/lug/blob/master/config.example.yaml) -and [Wiki](https://github.com/sjtug/lug/wiki/Configuration) for the latest version. +Abort the in-flight sync of the named worker. The cancellation propagates +through the worker's context and (with cgroup isolation) kills the entire +process tree. +- **202 Accepted** — cancellation initiated. +- **404 Not Found** — no worker by that name, or the worker is idle. + +```sh +# Example: abort a stuck rsync job +curl -X POST http://localhost:7001/lug/v1/admin/worker/putty/abort ``` -interval: 3 # Interval between pollings -loglevel: 5 # 0-5. 0 for ERROR and 5 for DEBUG -logstashaddr: "172.0.0.4:6000" # TCP Address of logstash. empty means no logstash support -dummy: # place your anchor here! - common_interval: &common_interval - interval: 3600 - common_retry: &commone_retry - retry: 3 -repos: - - type: shell_script - script: rsync -av rsync://rsync.chiark.greenend.org.uk/ftp/users/sgtatham/putty-website-mirror/ /tmp/putty - name: putty - rlimit: 300M - <<: *common_interval # interval: 3600 will be inserted here - - type: external - name: ubuntu - proxy_to: http://ftp.sjtu.edu.cn/ubuntu/ - <<: [*common_interval, *common_retry] # use array for importing multiple anchors -# You can add more repos here, different repos may have different worker types, -# refer to Worker Types section for detailed explanation + +#### `DELETE /lug/v1/admin/manager` + +Gracefully shut down the manager (stop scheduler, then exit the run loop). + +### Manual sync triggering + +Lug does not currently expose a REST endpoint to trigger an immediate sync for +a specific worker — syncs are launched by the scheduler when each worker's +`interval` elapses. To force a sync to start earlier, restart lug or use the +stop / start cycle: + +```sh +# Force all eligible workers to re-enter the pending queue: +curl -X POST http://localhost:7001/lug/v1/admin/manager/stop +curl -X POST http://localhost:7001/lug/v1/admin/manager/start ``` +On start, every worker whose interval has elapsed (including those that were +just stopped) is queued for sync. + ## Development -Contributors should push to their own branch. Reviewed code will be merged to `master` branch. +This project requires **Go ≥ 1.23** and uses Nix for the development +environment. -Currently this project assumes Go >= 1.23. +```sh +# Enter the dev shell (includes Go toolchain, golangci-lint, treefmt) +nix develop -1. set your `GOPATH` to a directory: `export GOPATH=/home/go`. Set `$GOPATH/bin` to your `$PATH`: `export PATH=$PATH:$GOPATH/bin` -2. `go get github.com/sjtug/lug` -3. Install dep by `curl https://raw.githubusercontent.com/golang/dep/master/install.sh | sh` -3. `cd $GOPATH/src/github.com/sjtug/lug && dep ensure` -4. Modify code, then use `go build .` to build binary, or test with `go test $(go list ./... | grep -v /vendor/)` -5. Run `scripts/gen_license.sh` before committing your code +# Build +go build ./... -NOTICE: Please attach test files when contributing to your module +# Run tests (memory-capped to avoid host OOM from cgroup enforcement tests) +go test -c -o /tmp/worker.test ./pkg/worker/ +systemd-run --user --scope -p MemoryMax=2G -p Delegate=yes /tmp/worker.test -test.v + +# Or without cgroup delegation (enforcement tests auto-skip): +go test ./... +``` +Tests that exercise cgroup enforcement (`TestCgroupEnforcement`, +`TestCgroupTelemetryHappy`) require a delegated cgroup subtree and +automatically skip otherwise. From db8fb31779564b0bcdf879afe5557a8037b89254 Mon Sep 17 00:00:00 2001 From: comonad Date: Mon, 14 Sep 2026 01:13:43 +0800 Subject: [PATCH 09/15] feat(api): task queue inspection, manual sync trigger, and live job details MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit New REST endpoints: GET /lug/v1/admin/queue — running workers (with cgroup/PID info) and the pending launch queue in FIFO order. POST /lug/v1/admin/worker/:name/sync — insert/lift the named worker to the head of the pending queue and launch immediately if capacity allows. Returns 409 if the worker is already syncing (abort first). GET /lug/v1/admin/worker/:name/job — live job detail: cgroup path, main PID, start time, and real-time cgroup stats (memory.current, memory.peak, memory.max, pids.current). Includes an nsenter attach hint for interactive debugging. Implementation: queue inspection and manual-sync requests are routed through dedicated channels into the manager Run() goroutine, so pending-queue access is race-free without adding a mutex. Active job tracking is published from shellScriptExecutor (mutex-protected) via the new jobInspector interface and surfaced in worker.Status.ActiveJob. Non-Linux: ReadCgroupStats and jobCgroup.Path stubs compile on all platforms; ActiveJob.CgroupPath is empty when cgroup isolation is unavailable. --- README.md | 82 +++++++++++++++--- pkg/manager/json_rest.go | 62 ++++++++++++++ pkg/manager/manager.go | 122 ++++++++++++++++++++++++++- pkg/worker/cgroup_runner.go | 24 ++++++ pkg/worker/cgroup_runner_stub.go | 4 + pkg/worker/executor.go | 23 +++++ pkg/worker/executor_invoke_worker.go | 6 +- pkg/worker/shell_script_executor.go | 32 +++++++ pkg/worker/worker.go | 3 + 9 files changed, 343 insertions(+), 15 deletions(-) diff --git a/README.md b/README.md index 3a66f2e..2d097a2 100644 --- a/README.md +++ b/README.md @@ -234,25 +234,81 @@ process tree. curl -X POST http://localhost:7001/lug/v1/admin/worker/putty/abort ``` -#### `DELETE /lug/v1/admin/manager` - -Gracefully shut down the manager (stop scheduler, then exit the run loop). +#### `POST /lug/v1/admin/worker/:name/sync` -### Manual sync triggering +Trigger an immediate sync for the named worker. The worker is inserted at the +head of the pending queue (or lifted to the head if already queued) and +launched immediately if capacity allows. -Lug does not currently expose a REST endpoint to trigger an immediate sync for -a specific worker — syncs are launched by the scheduler when each worker's -`interval` elapses. To force a sync to start earlier, restart lug or use the -stop / start cycle: +- **202 Accepted** — sync queued / launched. +- **404 Not Found** — no worker by that name. +- **409 Conflict** — the worker is already syncing. Abort first if you need + to restart it. ```sh -# Force all eligible workers to re-enter the pending queue: -curl -X POST http://localhost:7001/lug/v1/admin/manager/stop -curl -X POST http://localhost:7001/lug/v1/admin/manager/start +# Trigger putty sync right now +curl -X POST http://localhost:7001/lug/v1/admin/worker/putty/sync + +# Force a restart: abort then re-trigger +curl -X POST http://localhost:7001/lug/v1/admin/worker/putty/abort +sleep 2 +curl -X POST http://localhost:7001/lug/v1/admin/worker/putty/sync +``` + +#### `GET /lug/v1/admin/queue` + +Snapshot of the task queue: which workers are currently syncing (with live +cgroup/PID info) and which are pending launch. + +```json +{ + "running": [ + { + "name": "putty", + "active_job": { + "cgroup_path": "/sys/fs/cgroup/system.slice/.../jobs/putty-1234567890", + "main_pid": 12345, + "started_at": "2025-01-01T00:00:00Z" + } + } + ], + "pending": ["ubuntu", "debian"] +} ``` -On start, every worker whose interval has elapsed (including those that were -just stopped) is queued for sync. +#### `GET /lug/v1/admin/worker/:name/job` + +Live details of the named worker's active sync job: cgroup path, PID, start +time, and — when cgroup isolation is active — live memory and PID counters +read directly from the kernel. + +- **200 OK** — job details (see below). +- **404 Not Found** — no such worker, or the worker is idle. + +```json +{ + "name": "putty", + "active_job": { + "cgroup_path": "/sys/fs/cgroup/.../jobs/putty-1234567890", + "main_pid": 12345, + "started_at": "2025-01-01T00:00:00Z" + }, + "cgroup_stats": { + "memory_current_bytes": 104857600, + "memory_peak_bytes": 209715200, + "memory_limit_bytes": 1073741824, + "pids": 5 + }, + "attach_hint": "nsenter --cgroup=/sys/fs/cgroup/.../jobs/putty-1234567890 --fork " +} +``` + +The `attach_hint` field shows the `nsenter` invocation to enter the job's +cgroup for interactive debugging (run a shell, inspect `/proc`, etc.). + +#### `DELETE /lug/v1/admin/manager` + +Gracefully shut down the manager (stop scheduler, then exit the run loop). ## Development diff --git a/pkg/manager/json_rest.go b/pkg/manager/json_rest.go index 626179d..6762fb2 100644 --- a/pkg/manager/json_rest.go +++ b/pkg/manager/json_rest.go @@ -1,11 +1,13 @@ package manager import ( + "fmt" "net/http" "time" "github.com/ant0ine/go-json-rest/rest" log "github.com/sirupsen/logrus" + "github.com/sjtug/lug/pkg/worker" ) // RestfulAPI is a JSON-like API of given manager @@ -30,6 +32,9 @@ func (r *RestfulAPI) GetAPIHandler() http.Handler { rest.Post("/lug/v1/admin/manager/start", r.startManager), rest.Post("/lug/v1/admin/manager/stop", r.stopManager), rest.Post("/lug/v1/admin/worker/:name/abort", r.abortWorker), + rest.Post("/lug/v1/admin/worker/:name/sync", r.triggerSync), + rest.Get("/lug/v1/admin/worker/:name/job", r.getWorkerJob), + rest.Get("/lug/v1/admin/queue", r.getQueueStatus), rest.Delete("/lug/v1/admin/manager", r.exitManager), ) if err != nil { @@ -106,3 +111,60 @@ func (r *RestfulAPI) abortWorker(w rest.ResponseWriter, req *rest.Request) { } w.WriteHeader(http.StatusAccepted) } + +func (r *RestfulAPI) triggerSync(w rest.ResponseWriter, req *rest.Request) { + name := req.PathParam("name") + if err := r.manager.TriggerWorkerSync(name); err != nil { + code := http.StatusNotFound + if err.Error() != fmt.Sprintf("no worker named %s", name) { + code = http.StatusConflict + } + rest.Error(w, err.Error(), code) + return + } + w.WriteHeader(http.StatusAccepted) +} + +func (r *RestfulAPI) getQueueStatus(w rest.ResponseWriter, req *rest.Request) { + qs := r.manager.GetQueueStatus() + if err := w.WriteJson(qs); err != nil { + log.Error(err) + } +} + +// jobDetail is the response from the per-worker job inspection endpoint. +type jobDetail struct { + Name string `json:"name"` + ActiveJob *worker.ActiveJobInfo `json:"active_job"` + CgroupStats *worker.CgroupStats `json:"cgroup_stats,omitempty"` + AttachHint string `json:"attach_hint,omitempty"` +} + +func (r *RestfulAPI) getWorkerJob(w rest.ResponseWriter, req *rest.Request) { + name := req.PathParam("name") + status := r.manager.GetStatus() + ws, ok := status.WorkerStatus[name] + if !ok { + rest.Error(w, "no worker named "+name, http.StatusNotFound) + return + } + if ws.ActiveJob == nil { + rest.Error(w, "worker "+name+" is idle", http.StatusNotFound) + return + } + resp := jobDetail{ + Name: name, + ActiveJob: ws.ActiveJob, + } + if ws.ActiveJob.CgroupPath != "" { + stats := worker.ReadCgroupStats(ws.ActiveJob.CgroupPath) + resp.CgroupStats = &stats + resp.AttachHint = fmt.Sprintf( + "nsenter --cgroup=%s --fork ", + ws.ActiveJob.CgroupPath, + ) + } + if err := w.WriteJson(resp); err != nil { + log.Error(err) + } +} diff --git a/pkg/manager/manager.go b/pkg/manager/manager.go index c203b36..59d7bde 100644 --- a/pkg/manager/manager.go +++ b/pkg/manager/manager.go @@ -33,6 +33,29 @@ const ( StartFinish ) +// syncRequest asks the Run() goroutine to enqueue a manual sync. +type syncRequest struct { + name string + resultCh chan error +} + +// queueRequest asks for a snapshot of the task queue. +type queueRequest struct { + resultCh chan QueueStatus +} + +// QueueStatus is the response from GetQueueStatus. +type QueueStatus struct { + Running []RunningEntry `json:"running"` + Pending []string `json:"pending"` +} + +// RunningEntry describes a currently syncing worker. +type RunningEntry struct { + Name string `json:"name"` + ActiveJob *worker.ActiveJobInfo `json:"active_job,omitempty"` +} + // Manager holds worker instances type Manager struct { config *config.Config @@ -48,7 +71,11 @@ type Manager struct { // hasGrant tracks workers holding an admission grant, so the poll loop // can release budgets once a worker returns to idle hasGrant map[string]bool - logger *logrus.Entry + // syncReqChan carries manual-sync requests into the Run() goroutine. + syncReqChan chan syncRequest + // queueReqChan carries queue-inspection requests into the Run() goroutine. + queueReqChan chan queueRequest + logger *logrus.Entry } // Status holds the status of a manager and its workers @@ -146,6 +173,8 @@ func NewManager(config *config.Config) (*Manager, error) { running: true, admission: admissionCtl, hasGrant: make(map[string]bool), + syncReqChan: make(chan syncRequest), + queueReqChan: make(chan queueRequest), logger: logger, } for _, repoConfig := range config.Repos { @@ -339,6 +368,10 @@ func (m *Manager) Run() { } } } + case req := <-m.syncReqChan: + req.resultCh <- m.handleSyncRequest(req.name) + case req := <-m.queueReqChan: + req.resultCh <- m.buildQueueStatus() case sig, ok := <-m.controlChan: if ok { switch sig { @@ -429,6 +462,93 @@ func (m *Manager) AbortWorker(name string) error { return fmt.Errorf("no worker named %s", name) } +// TriggerWorkerSync inserts the named worker at the head of the pending +// queue (or lifts it there if already queued). The scheduler launches it +// immediately if capacity allows. +func (m *Manager) TriggerWorkerSync(name string) error { + req := syncRequest{name: name, resultCh: make(chan error, 1)} + m.syncReqChan <- req + return <-req.resultCh +} + +// GetQueueStatus returns a snapshot of running and pending workers. +func (m *Manager) GetQueueStatus() QueueStatus { + req := queueRequest{resultCh: make(chan QueueStatus, 1)} + m.queueReqChan <- req + return <-req.resultCh +} + +// handleSyncRequest runs inside the Run() goroutine with exclusive queue +// access. It validates, enqueues, and immediately tries to launch. +func (m *Manager) handleSyncRequest(name string) error { + idx := m.findWorkerIdx(name) + if idx < 0 { + return fmt.Errorf("no worker named %s", name) + } + w := m.workers[idx] + if !w.GetStatus().Idle { + return fmt.Errorf("worker %s is currently syncing", name) + } + m.removeFromPendingQueue(idx) + m.pendingQueue = append([]int{idx}, m.pendingQueue...) + m.logger.WithFields(logrus.Fields{ + "event": "manual_sync", + "target_worker_name": name, + }).Infof("manual sync requested for worker %s", name) + // Try to launch immediately instead of waiting for the next tick. + running := m.countRunning() + m.launchWorkerFromPendingQueue(m.config.ConcurrentLimit - running) + return nil +} + +// buildQueueStatus runs inside the Run() goroutine. +func (m *Manager) buildQueueStatus() QueueStatus { + qs := QueueStatus{Running: []RunningEntry{}, Pending: []string{}} + for _, w := range m.workers { + status := w.GetStatus() + name := w.GetConfig()["name"].(string) + if !status.Idle { + qs.Running = append(qs.Running, RunningEntry{ + Name: name, + ActiveJob: status.ActiveJob, + }) + } + } + for _, idx := range m.pendingQueue { + name := m.workers[idx].GetConfig()["name"].(string) + qs.Pending = append(qs.Pending, name) + } + return qs +} + +func (m *Manager) findWorkerIdx(name string) int { + for i, w := range m.workers { + if w.GetConfig()["name"] == name { + return i + } + } + return -1 +} + +func (m *Manager) removeFromPendingQueue(idx int) { + for i, qIdx := range m.pendingQueue { + if qIdx == idx { + m.pendingQueue = append(m.pendingQueue[:i], m.pendingQueue[i+1:]...) + return + } + } +} + +func (m *Manager) countRunning() int { + count := 0 + for _, w := range m.workers { + if !w.GetStatus().Idle { + count++ + } + } + return count +} + // GetStatus gets status of Manager func (m *Manager) GetStatus() *Status { status := Status{ diff --git a/pkg/worker/cgroup_runner.go b/pkg/worker/cgroup_runner.go index da86e06..1b18820 100644 --- a/pkg/worker/cgroup_runner.go +++ b/pkg/worker/cgroup_runner.go @@ -178,6 +178,9 @@ func newJobCgroup(name string, memMax uint64) (*jobCgroup, error) { return &jobCgroup{dir: dir, fd: fd}, nil } +// Path returns the absolute cgroup directory for this job. +func (j *jobCgroup) Path() string { return j.dir } + // Kill terminates every process in the job cgroup (cgroup.kill, Linux >= 5.14). func (j *jobCgroup) Kill() error { return os.WriteFile(filepath.Join(j.dir, "cgroup.kill"), []byte("1"), 0) @@ -213,6 +216,27 @@ func (j *jobCgroup) Close() { Warn("failed to remove job cgroup; it will leak until manual cleanup") } +// ReadCgroupStats reads live resource counters from a cgroup v2 directory. +// Returns zero values for any counter that cannot be read. +func ReadCgroupStats(dir string) CgroupStats { + var s CgroupStats + if data, err := os.ReadFile(filepath.Join(dir, "memory.current")); err == nil { + s.MemoryCurrentBytes, _ = strconv.ParseUint(strings.TrimSpace(string(data)), 10, 64) + } + if data, err := os.ReadFile(filepath.Join(dir, "memory.peak")); err == nil { + s.MemoryPeakBytes, _ = strconv.ParseUint(strings.TrimSpace(string(data)), 10, 64) + } + if data, err := os.ReadFile(filepath.Join(dir, "memory.max")); err == nil { + if v := strings.TrimSpace(string(data)); v != "max" { + s.MemoryLimitBytes, _ = strconv.ParseUint(v, 10, 64) + } + } + if data, err := os.ReadFile(filepath.Join(dir, "pids.current")); err == nil { + s.PIDs, _ = strconv.Atoi(strings.TrimSpace(string(data))) + } + return s +} + // attachToCgroup makes cmd's child enter the job cgroup atomically at // clone3 time (CLONE_INTO_CGROUP), so not a single instruction runs outside // the resource-limited scope. diff --git a/pkg/worker/cgroup_runner_stub.go b/pkg/worker/cgroup_runner_stub.go index 1f3ea82..b6162d3 100644 --- a/pkg/worker/cgroup_runner_stub.go +++ b/pkg/worker/cgroup_runner_stub.go @@ -11,7 +11,11 @@ type jobCgroup struct{} func newJobCgroup(name string, memMax uint64) (*jobCgroup, error) { return nil, nil } +func (j *jobCgroup) Path() string { return "" } func (j *jobCgroup) Kill() error { return nil } func (j *jobCgroup) Collect() (peakMem uint64, oom bool) { return 0, false } func (j *jobCgroup) Close() {} func attachToCgroup(cmd *exec.Cmd, cg *jobCgroup) {} + +// ReadCgroupStats is a no-op on non-Linux platforms. +func ReadCgroupStats(dir string) CgroupStats { return CgroupStats{} } diff --git a/pkg/worker/executor.go b/pkg/worker/executor.go index a756184..74939ec 100644 --- a/pkg/worker/executor.go +++ b/pkg/worker/executor.go @@ -17,9 +17,32 @@ type execResult struct { OOMKilled bool } +// ActiveJobInfo describes the currently executing sync job. Exposed through +// worker.Status so operators can inspect live resource usage and attach to +// the job's cgroup for debugging. +type ActiveJobInfo struct { + CgroupPath string `json:"cgroup_path,omitempty"` + MainPID int `json:"main_pid,omitempty"` + StartedAt time.Time `json:"started_at"` +} + +// CgroupStats holds live resource counters read from a cgroup v2 directory. +type CgroupStats struct { + MemoryCurrentBytes uint64 `json:"memory_current_bytes"` + MemoryPeakBytes uint64 `json:"memory_peak_bytes,omitempty"` + MemoryLimitBytes uint64 `json:"memory_limit_bytes,omitempty"` + PIDs int `json:"pids"` +} + // executor is a layer beneath worker, called by executorInvokeWorker. // ctx cancellation must terminate the whole job (including descendants). type executor interface { // When called, the executor performs sync for one time RunOnce(ctx context.Context, logger *logrus.Entry) (execResult, error) } + +// jobInspector is optionally implemented by executors that expose live +// information about the currently executing job. +type jobInspector interface { + ActiveJob() *ActiveJobInfo +} diff --git a/pkg/worker/executor_invoke_worker.go b/pkg/worker/executor_invoke_worker.go index f323eb3..905e90e 100644 --- a/pkg/worker/executor_invoke_worker.go +++ b/pkg/worker/executor_invoke_worker.go @@ -97,7 +97,7 @@ func (eiw *executorInvokeWorker) AbortSync() { func (eiw *executorInvokeWorker) GetStatus() Status { eiw.rwmutex.RLock() defer eiw.rwmutex.RUnlock() - return Status{ + status := Status{ Idle: eiw.idle, Result: eiw.result, LastFinished: eiw.lastFinished, @@ -105,6 +105,10 @@ func (eiw *executorInvokeWorker) GetStatus() Status { Stdout: eiw.stdout.GetAll(), Stderr: eiw.stderr.GetAll(), } + if inspector, ok := eiw.executor.(jobInspector); ok { + status.ActiveJob = inspector.ActiveJob() + } + return status } func (eiw *executorInvokeWorker) GetConfig() config.RepoConfig { diff --git a/pkg/worker/shell_script_executor.go b/pkg/worker/shell_script_executor.go index 7442479..d3eed4a 100644 --- a/pkg/worker/shell_script_executor.go +++ b/pkg/worker/shell_script_executor.go @@ -9,6 +9,7 @@ import ( "os" "os/exec" "strings" + "sync" "time" "github.com/davecgh/go-spew/spew" @@ -27,6 +28,10 @@ type shellScriptExecutor struct { // timeout is the per-attempt wall clock budget (0 = no timeout), parsed // from `timeout` (seconds). timeout time.Duration + + // mu guards activeJob, which is set while RunOnce is executing. + mu sync.Mutex + activeJob *ActiveJobInfo } func newShellScriptExecutor(cfg config.RepoConfig) (*shellScriptExecutor, error) { @@ -159,6 +164,21 @@ func (w *shellScriptExecutor) RunOnce(ctx context.Context, logger *logrus.Entry) if err != nil { return execResult{}, fmt.Errorf("execution cannot start: %w", err) } + + // Publish the active job so operators can inspect live resource usage. + info := &ActiveJobInfo{MainPID: cmd.Process.Pid, StartedAt: start} + if cg != nil { + info.CgroupPath = cg.Path() + } + w.mu.Lock() + w.activeJob = info + w.mu.Unlock() + defer func() { + w.mu.Lock() + w.activeJob = nil + w.mu.Unlock() + }() + err = cmd.Wait() result := execResult{ @@ -192,3 +212,15 @@ func (w *shellScriptExecutor) RunOnce(ctx context.Context, logger *logrus.Entry) } return result, nil } + +// ActiveJob returns information about the currently running job, or nil when +// the executor is idle. Implements jobInspector. +func (w *shellScriptExecutor) ActiveJob() *ActiveJobInfo { + w.mu.Lock() + defer w.mu.Unlock() + if w.activeJob == nil { + return nil + } + copy := *w.activeJob + return © +} diff --git a/pkg/worker/worker.go b/pkg/worker/worker.go index 8c145a7..a2beb80 100644 --- a/pkg/worker/worker.go +++ b/pkg/worker/worker.go @@ -35,6 +35,9 @@ type Status struct { Telemetry Telemetry // Idle stands for whether worker is idle, false if syncing Idle bool + // ActiveJob describes the running job (cgroup path, PID, start time). + // Nil when the worker is idle. + ActiveJob *ActiveJobInfo `json:"active_job,omitempty"` // Last stdout(s) for admin. Internal implementation may vary to provide it in Status() Stdout []string // Last stderr(s) for admin. Internal implementation may vary to provide it in Status() From 94972fb98213421e56dfa59a8536b642ec006f7c Mon Sep 17 00:00:00 2001 From: comonad Date: Mon, 14 Sep 2026 01:31:25 +0800 Subject: [PATCH 10/15] chore(nix): add gomod2nix pre-commit hooks --- flake.nix | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/flake.nix b/flake.nix index 282b3a1..20276ce 100644 --- a/flake.nix +++ b/flake.nix @@ -83,6 +83,14 @@ commitizen.enable = true; golangci-lint.enable = true; treefmt.enable = true; + gomod2nix = { + enable = true; + name = "Check if gomod2nix.toml is up-to-date"; + entry = "${pkgs.gomod2nix}/bin/gomod2nix generate"; + files = "(gomod2nix\\.toml|go\\.(mod|sum))$"; + pass_filenames = false; + stages = [ "pre-commit" ]; + }; }; packages.default = pkgs.callPackage ./package.nix { From d67ef1ae1cb63a61a6bb7188a83a930570b697bb Mon Sep 17 00:00:00 2001 From: comonad Date: Mon, 14 Sep 2026 03:00:33 +0800 Subject: [PATCH 11/15] fix(worker): emit start event for every retry --- pkg/worker/executor_invoke_worker.go | 5 +++- pkg/worker/worker_test.go | 44 ++++++++++++++++++++++++++++ 2 files changed, 48 insertions(+), 1 deletion(-) diff --git a/pkg/worker/executor_invoke_worker.go b/pkg/worker/executor_invoke_worker.go index 905e90e..616eda0 100644 --- a/pkg/worker/executor_invoke_worker.go +++ b/pkg/worker/executor_invoke_worker.go @@ -145,11 +145,14 @@ func (w *executorInvokeWorker) RunSync() { w.cancelRun = cancel w.cancelMu.Unlock() - logger.WithField("event", "start_execution").Info("sync started") retry_limit := w.retry var result execResult var err error for retry_cnt := 1; retry_cnt <= retry_limit; retry_cnt++ { + logger.WithFields(log.Fields{ + "event": "start_execution", + "try_cnt": retry_cnt, + }).Info("sync attempt started") logger.WithField("event", "invoke_executor").WithField( "try_cnt", retry_cnt).Debugf("Invoke executor for the %v time", retry_cnt) result, err = w.executor.RunOnce(ctx, logger) diff --git a/pkg/worker/worker_test.go b/pkg/worker/worker_test.go index de86016..a5dbe67 100644 --- a/pkg/worker/worker_test.go +++ b/pkg/worker/worker_test.go @@ -14,6 +14,7 @@ import ( "sync/atomic" "github.com/sirupsen/logrus" + logrustest "github.com/sirupsen/logrus/hooks/test" "github.com/sjtug/lug/pkg/config" ) @@ -104,6 +105,49 @@ func (d *dummyExecutor) RunOnce(ctx context.Context, logger *logrus.Entry) (exec return execResult{}, errors.New("dummy error") } +type failOnceExecutor struct { + RunCnt int32 +} + +func (d *failOnceExecutor) RunOnce(ctx context.Context, logger *logrus.Entry) (execResult, error) { + if atomic.AddInt32(&d.RunCnt, 1) == 1 { + return execResult{}, errors.New("first attempt failed") + } + return execResult{}, nil +} + +func TestExecutorInvokeWorkerLogsEveryAttemptStart(t *testing.T) { + asrt := assert.New(t) + d := &failOnceExecutor{} + cfg := config.RepoConfig{ + "retry": 2, + "retry_interval": 0, + "name": "retry-log-test", + } + control := make(chan int, 1) + w, err := NewExecutorInvokeWorker(d, Status{Idle: true}, cfg, control) + asrt.NoError(err) + + logger, hook := logrustest.NewNullLogger() + w.logger = logger.WithField("repo", "retry-log-test") + go w.RunSync() + w.TriggerSync() + + asrt.Eventually(func() bool { + return atomic.LoadInt32(&d.RunCnt) == 2 && w.GetStatus().Idle + }, time.Second, 10*time.Millisecond) + + var attempts []int + for _, entry := range hook.AllEntries() { + if entry.Data["event"] == "start_execution" { + attempt, ok := entry.Data["try_cnt"].(int) + asrt.True(ok) + attempts = append(attempts, attempt) + } + } + asrt.Equal([]int{1, 2}, attempts) +} + func TestExecutorInvokeWorker(t *testing.T) { asrt := assert.New(t) d := &dummyExecutor{} From 99d0bd729fdc80e0e7695b85deba478abb1ce0ba Mon Sep 17 00:00:00 2001 From: comonad Date: Mon, 14 Sep 2026 13:16:54 +0800 Subject: [PATCH 12/15] fix(worker): reap descendants without cgroups --- README.md | 62 ++++++++++++++------------- pkg/worker/cgroup_runner.go | 63 ++++++++++++++++++++++++++-- pkg/worker/cgroup_test.go | 44 ++++++++++++++++--- pkg/worker/process_group_linux.go | 35 ++++++++++++++++ pkg/worker/process_group_stub.go | 17 ++++++++ pkg/worker/shell_script_executor.go | 16 +++++-- pkg/worker/worker_test.go | 65 ++++++++++++++++++++++++++--- 7 files changed, 254 insertions(+), 48 deletions(-) create mode 100644 pkg/worker/process_group_linux.go create mode 100644 pkg/worker/process_group_stub.go diff --git a/README.md b/README.md index 2d097a2..8f74622 100644 --- a/README.md +++ b/README.md @@ -68,8 +68,8 @@ IO-saturated by design, and gating on it would permanently starve the queue. Lug uses cgroup v2 to enforce per-job memory limits, kill entire process trees on cancellation or timeout, and collect resource telemetry (peak memory, OOM kills, duration). **This is optional**: when cgroup isolation is unavailable, -lug falls back to plain subprocess execution with a logged warning, and -`rlimit_mem` / admission memory estimates are simply not enforced. +lug still kills the job's whole process group on cancellation or timeout, but +`rlimit_mem` and peak-memory telemetry are unavailable. ### Requirements @@ -82,25 +82,26 @@ lug falls back to plain subprocess execution with a logged warning, and ### Rootful Docker (Docker Engine with root daemon) -Docker ≥ 20.10 supports `--cgroupns=private` (the default since Docker 23), -which gives each container its own cgroup namespace and a writable cgroup root. -Run lug with: +`--cgroupns=private` gives the container its own cgroup namespace, but Docker +mounts `/sys/fs/cgroup` read-only for an unprivileged container. The namespace +flag alone therefore does **not** delegate a writable subtree. -```sh -docker run -d \ - --cgroupns=private \ - -v /srv/mirror:/data \ - -v $(pwd)/config.yaml:/app/config.yaml \ - sjtug/lug -c /app/config.yaml -``` +A container launcher can prepare a narrow delegation as follows: -This is sufficient — no `--privileged`, no extra capabilities, no bind-mounts -of `/sys/fs/cgroup`. The container sees its own cgroup subtree as `/` and can -write to it freely. +1. Start in a private cgroup namespace with `CAP_SYS_ADMIN`. +2. Remount the namespace's cgroup filesystem writable. +3. Move the supervisor into a leaf cgroup, enable the `memory` and `pids` + controllers, and create a delegated `jobs` subtree. +4. Set `LUG_CGROUP_JOBS_ROOT` to that subtree. +5. Drop `CAP_SYS_ADMIN` before executing lug. -> **Verify**: inside the container, `cat /sys/fs/cgroup/cgroup.controllers` -> should list `memory` (and ideally `pids`). If it shows an empty file or -> the path does not exist, the host likely uses cgroup v1. +The mirror deployment's LUG image implements this sequence in its entrypoint. +Do not bind-mount the host's complete `/sys/fs/cgroup` into the container: that +exposes cgroups outside the container's own subtree. + +> **Verify**: `GET /lug/v1/admin/worker/:name/job` should include a +> `cgroup_path` while a job is active, and `attempt_telemetry` should report +> `cgroup_scoped: true`. ### Rootless Podman @@ -135,9 +136,10 @@ detects this at startup and logs a warning: level=warning msg="cgroup v2 job isolation unavailable; falling back to plain process execution (no memory.max enforcement, rusage-only telemetry)" ``` -No configuration change is needed — the fallback is automatic. `rlimit_mem` -and `timeout` settings in the config are silently ignored when enforcement is -unavailable; admission control will operate without memory estimates. +No configuration change is needed — the fallback is automatic. Timeouts and +manual cancellation still terminate the complete process group. `rlimit_mem` +and peak-memory telemetry are unavailable, so admission control uses its +conservative estimate until cgroup telemetry becomes available. ## Observability @@ -322,14 +324,16 @@ nix develop # Build go build ./... -# Run tests (memory-capped to avoid host OOM from cgroup enforcement tests) -go test -c -o /tmp/worker.test ./pkg/worker/ -systemd-run --user --scope -p MemoryMax=2G -p Delegate=yes /tmp/worker.test -test.v - -# Or without cgroup delegation (enforcement tests auto-skip): +# Normal tests never invoke the kernel OOM killer. go test ./... + +# Optional destructive OOM integration test, isolated in a bounded scope: +go test -c -o /tmp/worker.test ./pkg/worker/ +systemd-run --user --scope -p MemoryMax=2G -p Delegate=yes \ + env LUG_RUN_CGROUP_OOM_TEST=1 \ + /tmp/worker.test -test.run TestCgroupOOMEnforcement -test.v ``` -Tests that exercise cgroup enforcement (`TestCgroupEnforcement`, -`TestCgroupTelemetryHappy`) require a delegated cgroup subtree and -automatically skip otherwise. +Non-destructive cgroup tests require a delegated subtree and automatically skip +otherwise. `TestCgroupOOMEnforcement` additionally requires the explicit +`LUG_RUN_CGROUP_OOM_TEST=1` opt-in shown above. diff --git a/pkg/worker/cgroup_runner.go b/pkg/worker/cgroup_runner.go index 1b18820..f11d1f1 100644 --- a/pkg/worker/cgroup_runner.go +++ b/pkg/worker/cgroup_runner.go @@ -32,7 +32,10 @@ import ( // cgroup namespace). When unavailable, executors fall back to plain process // execution with rusage-based telemetry; see fallback paths in the executor. -const cgroupMountpoint = "/sys/fs/cgroup" +const ( + cgroupMountpoint = "/sys/fs/cgroup" + cgroupJobsRootEnv = "LUG_CGROUP_JOBS_ROOT" +) var ( cgroupRootOnce sync.Once @@ -46,11 +49,28 @@ var ( // Returns "" if the environment does not support it. func jobsCgroupRoot() string { cgroupRootOnce.Do(func() { + if configured := os.Getenv(cgroupJobsRootEnv); configured != "" { + root, err := validateDelegatedJobsRoot(configured) + if err == nil { + log.WithFields(log.Fields{ + "event": "cgroup_jobs_root", + "path": root, + "source": cgroupJobsRootEnv, + }).Info("cgroup v2 job isolation enabled") + cgroupJobsRoot = root + return + } + log.WithFields(log.Fields{ + "event": "cgroup_configured_root_invalid", + "path": configured, + }).WithError(err).Warn("configured cgroup jobs root is unusable") + } + root, err := setupJobsCgroup() if err != nil { log.WithField("event", "cgroup_unavailable").WithError(err). - Warn("cgroup v2 job isolation unavailable; falling back to plain process execution " + - "(no memory.max enforcement, rusage-only telemetry)") + Warn("cgroup v2 job isolation unavailable; falling back to process-group execution " + + "(no memory.max enforcement or peak-memory telemetry)") return } log.WithField("event", "cgroup_jobs_root").WithField("path", root). @@ -60,6 +80,43 @@ func jobsCgroupRoot() string { return cgroupJobsRoot } +func validateDelegatedJobsRoot(configured string) (string, error) { + root := filepath.Clean(configured) + if !filepath.IsAbs(root) || root == cgroupMountpoint || + !strings.HasPrefix(root, cgroupMountpoint+string(filepath.Separator)) { + return "", fmt.Errorf("must be an absolute child of %s", cgroupMountpoint) + } + info, err := os.Stat(root) + if err != nil { + return "", err + } + if !info.IsDir() { + return "", fmt.Errorf("not a directory") + } + if err := unixAccessWritable(root); err != nil { + return "", fmt.Errorf("not writable: %w", err) + } + controllers, err := os.ReadFile(filepath.Join(root, "cgroup.subtree_control")) + if err != nil { + return "", err + } + for _, required := range []string{"memory", "pids"} { + if !containsWord(string(controllers), required) { + return "", fmt.Errorf("%s controller is not delegated", required) + } + } + return root, nil +} + +func containsWord(words, target string) bool { + for _, word := range strings.Fields(words) { + if word == target { + return true + } + } + return false +} + func setupJobsCgroup() (string, error) { own, err := ownCgroupPath() if err != nil { diff --git a/pkg/worker/cgroup_test.go b/pkg/worker/cgroup_test.go index 5f77549..5fcf145 100644 --- a/pkg/worker/cgroup_test.go +++ b/pkg/worker/cgroup_test.go @@ -2,6 +2,9 @@ package worker import ( "context" + "os" + "strconv" + "strings" "testing" "time" @@ -9,21 +12,50 @@ import ( "github.com/sjtug/lug/pkg/config" ) -// TestCgroupEnforcement verifies memory.max enforcement and OOM telemetry. -// It skips unless run under a delegated cgroup subtree, e.g.: +func TestCgroupMemoryLimitConfiguration(t *testing.T) { + const limit = 20 * 1024 * 1024 + cg, err := newJobCgroup("cg_limit_test", limit) + if err != nil { + t.Fatal(err) + } + if cg == nil { + t.Skip("cgroup delegation unavailable in this environment") + } + defer cg.Close() + + data, err := os.ReadFile(cg.Path() + "/memory.max") + if err != nil { + t.Fatal(err) + } + got, err := strconv.ParseUint(strings.TrimSpace(string(data)), 10, 64) + if err != nil { + t.Fatal(err) + } + if got != limit { + t.Fatalf("memory.max = %d, want %d", got, limit) + } +} + +// TestCgroupOOMEnforcement verifies kernel OOM enforcement and telemetry. It +// intentionally invokes the kernel OOM killer, so it is excluded from normal +// test runs. Run it explicitly in a disposable delegated scope with: // -// systemd-run --user --scope -p Delegate=yes go test -run TestCgroup ./pkg/worker/ -func TestCgroupEnforcement(t *testing.T) { +// LUG_RUN_CGROUP_OOM_TEST=1 go test -run TestCgroupOOMEnforcement ./pkg/worker/ +func TestCgroupOOMEnforcement(t *testing.T) { + if os.Getenv("LUG_RUN_CGROUP_OOM_TEST") != "1" { + t.Skip("set LUG_RUN_CGROUP_OOM_TEST=1 to run the destructive OOM integration test") + } + logrus.SetLevel(logrus.DebugLevel) e, err := newShellScriptExecutor(config.RepoConfig{ - "name": "cg_mem_test", + "name": "cg_oom_test", "script": `python3 -c "x = bytearray(100*1024*1024); import time; time.sleep(1)"`, "rlimit_mem": "20M", }) if err != nil { t.Fatal(err) } - result, err := e.RunOnce(context.Background(), logrus.WithField("repo", "cg_mem_test")) + result, err := e.RunOnce(context.Background(), logrus.WithField("repo", "cg_oom_test")) t.Logf("err=%v peak=%d oom=%v dur=%v", err, result.PeakMemBytes, result.OOMKilled, result.Duration) if result.PeakMemBytes == 0 { t.Skip("cgroup delegation unavailable in this environment") diff --git a/pkg/worker/process_group_linux.go b/pkg/worker/process_group_linux.go new file mode 100644 index 0000000..ee6d9f1 --- /dev/null +++ b/pkg/worker/process_group_linux.go @@ -0,0 +1,35 @@ +//go:build linux + +package worker + +import ( + "errors" + "os" + "os/exec" + "syscall" +) + +// configureProcessGroup starts the command as a process-group leader so the +// fallback cancellation path can terminate wrappers and all descendants. +func configureProcessGroup(cmd *exec.Cmd) { + if cmd.SysProcAttr == nil { + cmd.SysProcAttr = &syscall.SysProcAttr{} + } + cmd.SysProcAttr.Setpgid = true +} + +func killProcessGroup(cmd *exec.Cmd) error { + if cmd.Process == nil { + return os.ErrProcessDone + } + // Descendants retain the leader's process-group ID even if the wrapper has + // already exited. Address the group directly instead of killing only the + // process represented by cmd.Process. + if err := syscall.Kill(-cmd.Process.Pid, syscall.SIGKILL); err != nil { + if errors.Is(err, syscall.ESRCH) { + return os.ErrProcessDone + } + return err + } + return nil +} diff --git a/pkg/worker/process_group_stub.go b/pkg/worker/process_group_stub.go new file mode 100644 index 0000000..2cd9345 --- /dev/null +++ b/pkg/worker/process_group_stub.go @@ -0,0 +1,17 @@ +//go:build !linux + +package worker + +import ( + "os" + "os/exec" +) + +func configureProcessGroup(cmd *exec.Cmd) {} + +func killProcessGroup(cmd *exec.Cmd) error { + if cmd.Process == nil { + return os.ErrProcessDone + } + return cmd.Process.Kill() +} diff --git a/pkg/worker/shell_script_executor.go b/pkg/worker/shell_script_executor.go index d3eed4a..8ffb654 100644 --- a/pkg/worker/shell_script_executor.go +++ b/pkg/worker/shell_script_executor.go @@ -29,13 +29,16 @@ type shellScriptExecutor struct { // from `timeout` (seconds). timeout time.Duration + // newJobCgroup is an internal seam used to exercise the non-cgroup fallback. + newJobCgroup func(name string, memMax uint64) (*jobCgroup, error) + // mu guards activeJob, which is set while RunOnce is executing. mu sync.Mutex activeJob *ActiveJobInfo } func newShellScriptExecutor(cfg config.RepoConfig) (*shellScriptExecutor, error) { - e := &shellScriptExecutor{cfg: cfg} + e := &shellScriptExecutor{cfg: cfg, newJobCgroup: newJobCgroup} if raw, ok := cfg["rlimit_mem"]; ok { s, ok := raw.(string) if !ok { @@ -120,9 +123,14 @@ func (w *shellScriptExecutor) RunOnce(ctx context.Context, logger *logrus.Entry) defer cancel() } - logger.Debug("Invoking command:", fields[0], "with args:", fields[1:]) + logger.WithField("command", fields[0]).WithField("args", fields[1:]).Debug("invoking command") cmd := exec.CommandContext(ctx, fields[0], fields[1:]...) - cmd.WaitDelay = 30 * time.Second // don't hang forever on inherited pipes + configureProcessGroup(cmd) + cmd.Cancel = func() error { + logger.WithField("event", "job_process_group_kill").Info("killing job process group") + return killProcessGroup(cmd) + } + cmd.WaitDelay = 30 * time.Second // last-resort bound for inherited pipes // Forwarding config items to shell script as environmental variables // Adds a LUG_ prefix to their key @@ -143,7 +151,7 @@ func (w *shellScriptExecutor) RunOnce(ctx context.Context, logger *logrus.Entry) // Ephemeral per-job cgroup; nil when the environment lacks cgroup v2 // delegation, in which case the job runs unconfined (plain subprocess). jobName := fmt.Sprintf("%v-%d", w.cfg["name"], time.Now().UnixNano()) - cg, err := newJobCgroup(jobName, w.memMax) + cg, err := w.newJobCgroup(jobName, w.memMax) if err != nil { logger.WithField("event", "job_cgroup_failed").WithError(err). Warn("failed to create job cgroup; running unconfined") diff --git a/pkg/worker/worker_test.go b/pkg/worker/worker_test.go index a5dbe67..14c2e68 100644 --- a/pkg/worker/worker_test.go +++ b/pkg/worker/worker_test.go @@ -2,20 +2,22 @@ package worker import ( "context" + "errors" + "fmt" + "os" + "strconv" "strings" + "sync/atomic" + "syscall" "testing" "time" "github.com/davecgh/go-spew/spew" - "github.com/spf13/viper" - "github.com/stretchr/testify/assert" - - "errors" - "sync/atomic" - "github.com/sirupsen/logrus" logrustest "github.com/sirupsen/logrus/hooks/test" "github.com/sjtug/lug/pkg/config" + "github.com/spf13/viper" + "github.com/stretchr/testify/assert" ) func TestNewExternalWorker(t *testing.T) { @@ -214,6 +216,57 @@ func TestShellScriptExecutorCancel(t *testing.T) { asrt.Less(time.Since(start), 10*time.Second) } +func TestShellScriptExecutorCancelKillsDescendants(t *testing.T) { + asrt := assert.New(t) + pidFile := t.TempDir() + "/child.pid" + e, err := newShellScriptExecutor(config.RepoConfig{ + "name": "cancel_descendants_test", + "script": fmt.Sprintf( + `sh -c 'sleep 30 & echo $! > %s; wait'`, pidFile, + ), + }) + asrt.NoError(err) + // Force the production fallback used by restricted containers where cgroup + // delegation is unavailable. + e.newJobCgroup = func(string, uint64) (*jobCgroup, error) { return nil, nil } + + ctx, cancel := context.WithCancel(context.Background()) + errCh := make(chan error, 1) + go func() { + _, runErr := e.RunOnce(ctx, logrus.WithField("repo", "cancel_descendants_test")) + errCh <- runErr + }() + + var childPID int + asrt.Eventually(func() bool { + data, readErr := os.ReadFile(pidFile) + if readErr != nil { + return false + } + childPID, readErr = strconv.Atoi(strings.TrimSpace(string(data))) + return readErr == nil + }, 2*time.Second, 10*time.Millisecond) + if childPID == 0 { + cancel() + t.Fatal("child process did not publish its PID") + } + + start := time.Now() + cancel() + select { + case runErr := <-errCh: + asrt.Error(runErr) + asrt.Contains(runErr.Error(), "canceled") + case <-time.After(5 * time.Second): + t.Fatal("executor did not return promptly after cancellation") + } + asrt.Less(time.Since(start), 5*time.Second) + asrt.Eventually(func() bool { + err := syscall.Kill(childPID, 0) + return errors.Is(err, syscall.ESRCH) + }, 2*time.Second, 10*time.Millisecond, "descendant process %d survived cancellation", childPID) +} + func TestShellScriptExecutorTelemetry(t *testing.T) { asrt := assert.New(t) e, err := newShellScriptExecutor(config.RepoConfig{ From 0803ca27465f0d96864ccec5295b4cf4b4267b88 Mon Sep 17 00:00:00 2001 From: comonad Date: Mon, 14 Sep 2026 14:01:48 +0800 Subject: [PATCH 13/15] fix(api): emit working cgroup attach hint --- README.md | 15 ++++++++++++--- pkg/manager/json_rest.go | 16 ++++++++++++---- pkg/manager/json_rest_test.go | 19 +++++++++++++++++++ 3 files changed, 43 insertions(+), 7 deletions(-) create mode 100644 pkg/manager/json_rest_test.go diff --git a/README.md b/README.md index 8f74622..0d4bcf8 100644 --- a/README.md +++ b/README.md @@ -301,12 +301,21 @@ read directly from the kernel. "memory_limit_bytes": 1073741824, "pids": 5 }, - "attach_hint": "nsenter --cgroup=/sys/fs/cgroup/.../jobs/putty-1234567890 --fork " + "attach_hint": "nsenter --target=12345 --cgroup --join-cgroup -- " } ``` -The `attach_hint` field shows the `nsenter` invocation to enter the job's -cgroup for interactive debugging (run a shell, inspect `/proc`, etc.). +The `attach_hint` field shows the `nsenter` invocation to join the job's +cgroup for interactive debugging. Run it in the same PID and cgroup namespaces +as lug. For Docker, prepend `docker exec -it `; for example: + +```sh +docker exec -it siyuan-lug \ + nsenter --target=12345 --cgroup --join-cgroup -- sh +``` + +`--cgroup` enters the target process's cgroup namespace, while +`--join-cgroup` moves the new command into that process's cgroup. #### `DELETE /lug/v1/admin/manager` diff --git a/pkg/manager/json_rest.go b/pkg/manager/json_rest.go index 6762fb2..8010b20 100644 --- a/pkg/manager/json_rest.go +++ b/pkg/manager/json_rest.go @@ -159,12 +159,20 @@ func (r *RestfulAPI) getWorkerJob(w rest.ResponseWriter, req *rest.Request) { if ws.ActiveJob.CgroupPath != "" { stats := worker.ReadCgroupStats(ws.ActiveJob.CgroupPath) resp.CgroupStats = &stats - resp.AttachHint = fmt.Sprintf( - "nsenter --cgroup=%s --fork ", - ws.ActiveJob.CgroupPath, - ) + resp.AttachHint = jobAttachHint(ws.ActiveJob.MainPID) } if err := w.WriteJson(resp); err != nil { log.Error(err) } } + +func jobAttachHint(mainPID int) string { + // --cgroup selects the target process's cgroup namespace; --join-cgroup + // then moves the command into the target process's actual cgroup. Passing + // the cgroup directory to --cgroup is incorrect: that option expects a + // namespace file such as /proc//ns/cgroup. + return fmt.Sprintf( + "nsenter --target=%d --cgroup --join-cgroup -- ", + mainPID, + ) +} diff --git a/pkg/manager/json_rest_test.go b/pkg/manager/json_rest_test.go new file mode 100644 index 0000000..837d68f --- /dev/null +++ b/pkg/manager/json_rest_test.go @@ -0,0 +1,19 @@ +package manager + +import ( + "testing" + + "github.com/stretchr/testify/assert" +) + +func TestJobAttachHintTargetsProcessAndJoinsItsCgroup(t *testing.T) { + hint := jobAttachHint(12345) + + assert.Equal( + t, + "nsenter --target=12345 --cgroup --join-cgroup -- ", + hint, + ) + assert.NotContains(t, hint, "--fork") + assert.NotContains(t, hint, "/sys/fs/cgroup") +} From 5ab70fcee2a8da9442cfa175f382fa16c60e0cf0 Mon Sep 17 00:00:00 2001 From: comonad Date: Mon, 14 Sep 2026 14:09:30 +0800 Subject: [PATCH 14/15] fix(api): align job controls with documentation --- README.md | 42 ++++++++++++++++------------ pkg/manager/json_rest.go | 2 +- pkg/manager/json_rest_test.go | 27 ++++++++++++++++++ pkg/manager/manager.go | 5 +++- pkg/manager/manager_test.go | 15 ++++++++++ pkg/worker/cgroup_runner.go | 5 ++-- pkg/worker/executor_invoke_worker.go | 5 ++-- 7 files changed, 76 insertions(+), 25 deletions(-) diff --git a/README.md b/README.md index 0d4bcf8..9f7b4b9 100644 --- a/README.md +++ b/README.md @@ -21,7 +21,7 @@ Refer to [`config.example.yaml`](config.example.yaml) for available options. | Port | Purpose | |------|---------| -| 8081 | Prometheus metrics (`/metrics`) | +| 8080 by default (`8081` in `config.example.yaml`) | Prometheus metrics (`/metrics`) | | 7001 | JSON control API | ## Configuration @@ -67,9 +67,9 @@ IO-saturated by design, and gating on it would permanently starve the queue. Lug uses cgroup v2 to enforce per-job memory limits, kill entire process trees on cancellation or timeout, and collect resource telemetry (peak memory, OOM -kills, duration). **This is optional**: when cgroup isolation is unavailable, -lug still kills the job's whole process group on cancellation or timeout, but -`rlimit_mem` and peak-memory telemetry are unavailable. +kills, duration). **This is optional**: on Linux, when cgroup isolation is +unavailable, lug still kills the job's whole process group on cancellation or +timeout, but `rlimit_mem` and peak-memory telemetry are unavailable. ### Requirements @@ -133,7 +133,7 @@ If cgroup v2 is unavailable (cgroup v1 host, restricted container, etc.), lug detects this at startup and logs a warning: ``` -level=warning msg="cgroup v2 job isolation unavailable; falling back to plain process execution (no memory.max enforcement, rusage-only telemetry)" +level=warning msg="cgroup v2 job isolation unavailable; falling back to process-group execution (no memory.max enforcement or peak-memory telemetry)" ``` No configuration change is needed — the fallback is automatic. Timeouts and @@ -145,7 +145,8 @@ conservative estimate until cgroup telemetry becomes available. ### Prometheus metrics -Exposed at `http:///metrics` (default `:8081`). +Exposed at `http:///metrics` (built-in default `:8080`; +`config.example.yaml` selects `:8081`). **Sync counters** (labels: `worker`): @@ -154,8 +155,8 @@ Exposed at `http:///metrics` (default `:8081`). | `success_sync` | Successful sync runs (counter) | | `fail_sync` | Failed sync runs (counter) | -**Resource telemetry** (labels: `worker`) — populated when cgroup isolation is -active: +**Resource telemetry** (labels: `worker`) — duration is always populated; +peak-memory and OOM values require cgroup isolation: | Metric | Description | |--------|-------------| @@ -179,14 +180,18 @@ active: ### Structured logs -All log lines are JSON with `repo` and `sync_id` fields. Key events: +Log lines are JSON. Worker-run events carry `repo` and `sync_id`; manager +lifecycle and queue events instead carry their relevant manager/worker fields. +Key events: | Event | Description | |-------|-------------| +| `start_execution` | A sync attempt started (`try_cnt` identifies retries) | | `attempt_telemetry` | Emitted after every sync attempt with `duration_sec`, `peak_mem`, `oom_killed`, `cgroup_scoped` | | `trigger_sync` | A queued worker is launched | | `admission_deferred` | Admission controller deferred a launch (`reason` field explains why) | | `job_cgroup_kill` | Cancellation is reaping the job's cgroup tree | +| `job_process_group_kill` | Linux fallback cancellation is reaping the job's process group | | `abort_worker` | An operator aborted a sync via the REST API | ## JSON control API @@ -224,9 +229,9 @@ launched. #### `POST /lug/v1/admin/worker/:name/abort` -Abort the in-flight sync of the named worker. The cancellation propagates -through the worker's context and (with cgroup isolation) kills the entire -process tree. +Abort the in-flight sync of the named worker. Cancellation propagates through +the worker's context and kills the entire cgroup tree, or the Linux process +group when cgroup isolation is unavailable. - **202 Accepted** — cancellation initiated. - **404 Not Found** — no worker by that name, or the worker is idle. @@ -280,12 +285,13 @@ cgroup/PID info) and which are pending launch. #### `GET /lug/v1/admin/worker/:name/job` -Live details of the named worker's active sync job: cgroup path, PID, start -time, and — when cgroup isolation is active — live memory and PID counters -read directly from the kernel. +Live details of the named worker's active process: cgroup path, PID, start time, +and — when cgroup isolation is active — live memory and PID counters read +directly from the kernel. -- **200 OK** — job details (see below). -- **404 Not Found** — no such worker, or the worker is idle. +- **200 OK** — active process details (see below). +- **404 Not Found** — no such worker, or there is currently no active process + (including while queued or between retries). ```json { @@ -323,7 +329,7 @@ Gracefully shut down the manager (stop scheduler, then exit the run loop). ## Development -This project requires **Go ≥ 1.23** and uses Nix for the development +This project requires **Go ≥ 1.26** and uses Nix for the development environment. ```sh diff --git a/pkg/manager/json_rest.go b/pkg/manager/json_rest.go index 8010b20..5bf4f0c 100644 --- a/pkg/manager/json_rest.go +++ b/pkg/manager/json_rest.go @@ -149,7 +149,7 @@ func (r *RestfulAPI) getWorkerJob(w rest.ResponseWriter, req *rest.Request) { return } if ws.ActiveJob == nil { - rest.Error(w, "worker "+name+" is idle", http.StatusNotFound) + rest.Error(w, "worker "+name+" has no active job", http.StatusNotFound) return } resp := jobDetail{ diff --git a/pkg/manager/json_rest_test.go b/pkg/manager/json_rest_test.go index 837d68f..23085cc 100644 --- a/pkg/manager/json_rest_test.go +++ b/pkg/manager/json_rest_test.go @@ -1,11 +1,38 @@ package manager import ( + "net/http" + "net/http/httptest" "testing" + "github.com/sjtug/lug/pkg/config" "github.com/stretchr/testify/assert" ) +func TestWorkerJobReportsNoActiveProcess(t *testing.T) { + manager, err := NewManager(&config.Config{ + Repos: []config.RepoConfig{ + { + "type": "shell_script", + "name": "idle", + "script": "true", + }, + }, + }) + assert.NoError(t, err) + + request := httptest.NewRequest( + http.MethodGet, + "/lug/v1/admin/worker/idle/job", + nil, + ) + response := httptest.NewRecorder() + NewRestfulAPI(manager).GetAPIHandler().ServeHTTP(response, request) + + assert.Equal(t, http.StatusNotFound, response.Code) + assert.Contains(t, response.Body.String(), "worker idle has no active job") +} + func TestJobAttachHintTargetsProcessAndJoinsItsCgroup(t *testing.T) { hint := jobAttachHint(12345) diff --git a/pkg/manager/manager.go b/pkg/manager/manager.go index 59d7bde..f5f73cc 100644 --- a/pkg/manager/manager.go +++ b/pkg/manager/manager.go @@ -442,12 +442,15 @@ func (m *Manager) Exit() { // AbortWorker cancels the in-flight sync of the named worker, if any. // The cancellation propagates through the executor context and kills the -// whole job process tree when cgroup isolation is active. +// whole cgroup tree, or the Linux process group when cgroups are unavailable. func (m *Manager) AbortWorker(name string) error { for _, w := range m.workers { if w.GetConfig()["name"] != name { continue } + if w.GetStatus().Idle { + return fmt.Errorf("worker %s is idle", name) + } aborter, ok := w.(worker.Aborter) if !ok { return fmt.Errorf("worker %s does not support aborting", name) diff --git a/pkg/manager/manager_test.go b/pkg/manager/manager_test.go index 70a0a06..0bbda71 100644 --- a/pkg/manager/manager_test.go +++ b/pkg/manager/manager_test.go @@ -11,6 +11,21 @@ import ( "github.com/sjtug/lug/pkg/config" ) +func TestAbortWorkerRejectsIdleWorker(t *testing.T) { + manager, err := NewManager(&config.Config{ + Repos: []config.RepoConfig{ + { + "type": "shell_script", + "name": "idle", + "script": "true", + }, + }, + }) + assert.NoError(t, err) + + assert.EqualError(t, manager.AbortWorker("idle"), "worker idle is idle") +} + func TestManagerStartUp(t *testing.T) { manager, err := NewManager(&config.Config{ Interval: 3, diff --git a/pkg/worker/cgroup_runner.go b/pkg/worker/cgroup_runner.go index f11d1f1..b1ce5ce 100644 --- a/pkg/worker/cgroup_runner.go +++ b/pkg/worker/cgroup_runner.go @@ -29,8 +29,9 @@ import ( // // Requirements: cgroup v2 unified hierarchy and a delegated subtree (systemd // service with Delegate=yes, or a container started with a private, writable -// cgroup namespace). When unavailable, executors fall back to plain process -// execution with rusage-based telemetry; see fallback paths in the executor. +// cgroup namespace). When unavailable on Linux, executors retain whole-tree +// cancellation through process groups, but memory enforcement and peak-memory +// telemetry are unavailable; see fallback paths in the executor. const ( cgroupMountpoint = "/sys/fs/cgroup" diff --git a/pkg/worker/executor_invoke_worker.go b/pkg/worker/executor_invoke_worker.go index 616eda0..02305e8 100644 --- a/pkg/worker/executor_invoke_worker.go +++ b/pkg/worker/executor_invoke_worker.go @@ -83,9 +83,8 @@ func (eiw *executorInvokeWorker) TriggerSync() { eiw.signal <- 1 } -// AbortSync cancels the in-flight sync run, if any. The cancellation -// propagates through the executor's context and (with cgroup isolation) -// kills the entire job process tree. +// AbortSync cancels the in-flight sync run, if any. The executor kills the +// whole cgroup tree, or the Linux process group when cgroups are unavailable. func (eiw *executorInvokeWorker) AbortSync() { eiw.cancelMu.Lock() defer eiw.cancelMu.Unlock() From 1bdad3d9a78ebffcc22860322441856abd39423f Mon Sep 17 00:00:00 2001 From: comonad Date: Mon, 14 Sep 2026 16:50:28 +0800 Subject: [PATCH 15/15] fix(worker): make retry delays cancelable --- README.md | 1 + pkg/worker/executor_invoke_worker.go | 25 +++++++++++++++- pkg/worker/worker_test.go | 44 ++++++++++++++++++++++++++++ 3 files changed, 69 insertions(+), 1 deletion(-) diff --git a/README.md b/README.md index 9f7b4b9..d197ae8 100644 --- a/README.md +++ b/README.md @@ -187,6 +187,7 @@ Key events: | Event | Description | |-------|-------------| | `start_execution` | A sync attempt started (`try_cnt` identifies retries) | +| `retry_wait` | A failed attempt is in its cancelable retry delay (`duration_sec`) | | `attempt_telemetry` | Emitted after every sync attempt with `duration_sec`, `peak_mem`, `oom_killed`, `cgroup_scoped` | | `trigger_sync` | A queued worker is launched | | `admission_deferred` | Admission controller deferred a launch (`reason` field explains why) | diff --git a/pkg/worker/executor_invoke_worker.go b/pkg/worker/executor_invoke_worker.go index 02305e8..97a82c2 100644 --- a/pkg/worker/executor_invoke_worker.go +++ b/pkg/worker/executor_invoke_worker.go @@ -5,6 +5,7 @@ import ( "crypto/rand" "encoding/hex" "errors" + "fmt" "sync" "time" @@ -147,6 +148,7 @@ func (w *executorInvokeWorker) RunSync() { retry_limit := w.retry var result execResult var err error + retryLoop: for retry_cnt := 1; retry_cnt <= retry_limit; retry_cnt++ { logger.WithFields(log.Fields{ "event": "start_execution", @@ -163,7 +165,28 @@ func (w *executorInvokeWorker) RunSync() { "try_cnt", retry_cnt).Infof( "Failed on the %v-th executor. Error: %v", retry_cnt, err.Error()) logger.Debug("Stderr: ", result.Stderr) - time.Sleep(w.retry_interval) + if retry_cnt == retry_limit { + break + } + + logger.WithFields(log.Fields{ + "event": "retry_wait", + "try_cnt": retry_cnt, + "duration_sec": w.retry_interval.Seconds(), + }).Info("waiting before retry") + timer := time.NewTimer(w.retry_interval) + select { + case <-timer.C: + case <-ctx.Done(): + if !timer.Stop() { + select { + case <-timer.C: + default: + } + } + err = fmt.Errorf("execution canceled during retry wait: %w", ctx.Err()) + break retryLoop + } } w.cancelMu.Lock() diff --git a/pkg/worker/worker_test.go b/pkg/worker/worker_test.go index 14c2e68..3de54cc 100644 --- a/pkg/worker/worker_test.go +++ b/pkg/worker/worker_test.go @@ -150,6 +150,50 @@ func TestExecutorInvokeWorkerLogsEveryAttemptStart(t *testing.T) { asrt.Equal([]int{1, 2}, attempts) } +func TestExecutorInvokeWorkerDoesNotWaitAfterFinalAttempt(t *testing.T) { + asrt := assert.New(t) + d := &dummyExecutor{} + control := make(chan int, 1) + w, err := NewExecutorInvokeWorker(d, Status{Idle: true}, config.RepoConfig{ + "retry": 1, + "retry_interval": 3600, + "name": "no-final-wait-test", + }, control) + asrt.NoError(err) + + go w.RunSync() + w.TriggerSync() + + asrt.Eventually(func() bool { + return atomic.LoadInt32(&d.RunCnt) == 1 && w.GetStatus().Idle + }, time.Second, 10*time.Millisecond) +} + +func TestExecutorInvokeWorkerAbortInterruptsRetryWait(t *testing.T) { + asrt := assert.New(t) + d := &dummyExecutor{} + control := make(chan int, 1) + w, err := NewExecutorInvokeWorker(d, Status{Idle: true}, config.RepoConfig{ + "retry": 2, + "retry_interval": 3600, + "name": "cancel-retry-wait-test", + }, control) + asrt.NoError(err) + + go w.RunSync() + w.TriggerSync() + asrt.Eventually(func() bool { + return atomic.LoadInt32(&d.RunCnt) == 1 && !w.GetStatus().Idle + }, time.Second, 10*time.Millisecond) + + w.AbortSync() + + asrt.Eventually(func() bool { + return w.GetStatus().Idle + }, time.Second, 10*time.Millisecond) + asrt.Equal(int32(1), atomic.LoadInt32(&d.RunCnt)) +} + func TestExecutorInvokeWorker(t *testing.T) { asrt := assert.New(t) d := &dummyExecutor{}