diff --git a/backend/Cargo.lock b/backend/Cargo.lock index 811920560..84ec881d8 100644 --- a/backend/Cargo.lock +++ b/backend/Cargo.lock @@ -2003,7 +2003,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "39cab71617ae0d63f51a36d69f866391735b51691dbda63cf6f96d042b63efeb" dependencies = [ "libc", - "windows-sys 0.61.2", + "windows-sys 0.52.0", ] [[package]] @@ -3378,7 +3378,7 @@ dependencies = [ [[package]] name = "harmonia-daemon" version = "3.3.0" -source = "git+https://github.com/nix-community/harmonia.git#acf86e8a667266a7cfe54714db503dffafd14d60" +source = "git+https://github.com/DerDennisOP/harmonia.git?branch=build-resource-usage#ef33c7a4d6416d3d656fb4319f958522b8fc0bbb" dependencies = [ "futures-core", "futures-util", @@ -3401,7 +3401,7 @@ dependencies = [ [[package]] name = "harmonia-file-core" version = "3.3.0" -source = "git+https://github.com/nix-community/harmonia.git#acf86e8a667266a7cfe54714db503dffafd14d60" +source = "git+https://github.com/DerDennisOP/harmonia.git?branch=build-resource-usage#ef33c7a4d6416d3d656fb4319f958522b8fc0bbb" dependencies = [ "serde", "serde_json", @@ -3411,7 +3411,7 @@ dependencies = [ [[package]] name = "harmonia-file-nar" version = "3.3.0" -source = "git+https://github.com/nix-community/harmonia.git#acf86e8a667266a7cfe54714db503dffafd14d60" +source = "git+https://github.com/DerDennisOP/harmonia.git?branch=build-resource-usage#ef33c7a4d6416d3d656fb4319f958522b8fc0bbb" dependencies = [ "bstr", "bytes", @@ -3435,7 +3435,7 @@ dependencies = [ [[package]] name = "harmonia-protocol" version = "3.3.0" -source = "git+https://github.com/nix-community/harmonia.git#acf86e8a667266a7cfe54714db503dffafd14d60" +source = "git+https://github.com/DerDennisOP/harmonia.git?branch=build-resource-usage#ef33c7a4d6416d3d656fb4319f958522b8fc0bbb" dependencies = [ "async-stream", "bstr", @@ -3469,7 +3469,7 @@ dependencies = [ [[package]] name = "harmonia-protocol-derive" version = "3.3.0" -source = "git+https://github.com/nix-community/harmonia.git#acf86e8a667266a7cfe54714db503dffafd14d60" +source = "git+https://github.com/DerDennisOP/harmonia.git?branch=build-resource-usage#ef33c7a4d6416d3d656fb4319f958522b8fc0bbb" dependencies = [ "proc-macro2", "quote", @@ -3479,7 +3479,7 @@ dependencies = [ [[package]] name = "harmonia-store-aterm" version = "0.0.0-alpha.0" -source = "git+https://github.com/nix-community/harmonia.git#acf86e8a667266a7cfe54714db503dffafd14d60" +source = "git+https://github.com/DerDennisOP/harmonia.git?branch=build-resource-usage#ef33c7a4d6416d3d656fb4319f958522b8fc0bbb" dependencies = [ "bytes", "harmonia-store-content-address", @@ -3495,7 +3495,7 @@ dependencies = [ [[package]] name = "harmonia-store-build-result" version = "3.3.0" -source = "git+https://github.com/nix-community/harmonia.git#acf86e8a667266a7cfe54714db503dffafd14d60" +source = "git+https://github.com/DerDennisOP/harmonia.git?branch=build-resource-usage#ef33c7a4d6416d3d656fb4319f958522b8fc0bbb" dependencies = [ "harmonia-store-derivation", "num_enum", @@ -3506,7 +3506,7 @@ dependencies = [ [[package]] name = "harmonia-store-content-address" version = "3.3.0" -source = "git+https://github.com/nix-community/harmonia.git#acf86e8a667266a7cfe54714db503dffafd14d60" +source = "git+https://github.com/DerDennisOP/harmonia.git?branch=build-resource-usage#ef33c7a4d6416d3d656fb4319f958522b8fc0bbb" dependencies = [ "derive_more", "harmonia-store-path", @@ -3518,7 +3518,7 @@ dependencies = [ [[package]] name = "harmonia-store-db" version = "3.3.0" -source = "git+https://github.com/nix-community/harmonia.git#acf86e8a667266a7cfe54714db503dffafd14d60" +source = "git+https://github.com/DerDennisOP/harmonia.git?branch=build-resource-usage#ef33c7a4d6416d3d656fb4319f958522b8fc0bbb" dependencies = [ "foldhash", "harmonia-store-content-address", @@ -3535,7 +3535,7 @@ dependencies = [ [[package]] name = "harmonia-store-derivation" version = "0.0.0-alpha.0" -source = "git+https://github.com/nix-community/harmonia.git#acf86e8a667266a7cfe54714db503dffafd14d60" +source = "git+https://github.com/DerDennisOP/harmonia.git?branch=build-resource-usage#ef33c7a4d6416d3d656fb4319f958522b8fc0bbb" dependencies = [ "bytes", "data-encoding", @@ -3554,7 +3554,7 @@ dependencies = [ [[package]] name = "harmonia-store-path" version = "3.3.0" -source = "git+https://github.com/nix-community/harmonia.git#acf86e8a667266a7cfe54714db503dffafd14d60" +source = "git+https://github.com/DerDennisOP/harmonia.git?branch=build-resource-usage#ef33c7a4d6416d3d656fb4319f958522b8fc0bbb" dependencies = [ "derive_more", "harmonia-utils-base-encoding", @@ -3567,7 +3567,7 @@ dependencies = [ [[package]] name = "harmonia-store-path-info" version = "3.3.0" -source = "git+https://github.com/nix-community/harmonia.git#acf86e8a667266a7cfe54714db503dffafd14d60" +source = "git+https://github.com/DerDennisOP/harmonia.git?branch=build-resource-usage#ef33c7a4d6416d3d656fb4319f958522b8fc0bbb" dependencies = [ "harmonia-store-content-address", "harmonia-store-path", @@ -3579,7 +3579,7 @@ dependencies = [ [[package]] name = "harmonia-store-remote" version = "3.3.0" -source = "git+https://github.com/nix-community/harmonia.git#acf86e8a667266a7cfe54714db503dffafd14d60" +source = "git+https://github.com/DerDennisOP/harmonia.git?branch=build-resource-usage#ef33c7a4d6416d3d656fb4319f958522b8fc0bbb" dependencies = [ "async-stream", "futures-core", @@ -3599,7 +3599,7 @@ dependencies = [ [[package]] name = "harmonia-utils-base-encoding" version = "0.0.0-alpha.0" -source = "git+https://github.com/nix-community/harmonia.git#acf86e8a667266a7cfe54714db503dffafd14d60" +source = "git+https://github.com/DerDennisOP/harmonia.git?branch=build-resource-usage#ef33c7a4d6416d3d656fb4319f958522b8fc0bbb" dependencies = [ "data-encoding", "derive_more", @@ -3609,7 +3609,7 @@ dependencies = [ [[package]] name = "harmonia-utils-hash" version = "0.0.0-alpha.0" -source = "git+https://github.com/nix-community/harmonia.git#acf86e8a667266a7cfe54714db503dffafd14d60" +source = "git+https://github.com/DerDennisOP/harmonia.git?branch=build-resource-usage#ef33c7a4d6416d3d656fb4319f958522b8fc0bbb" dependencies = [ "blake3", "data-encoding", @@ -3626,7 +3626,7 @@ dependencies = [ [[package]] name = "harmonia-utils-io" version = "0.0.0-alpha.0" -source = "git+https://github.com/nix-community/harmonia.git#acf86e8a667266a7cfe54714db503dffafd14d60" +source = "git+https://github.com/DerDennisOP/harmonia.git?branch=build-resource-usage#ef33c7a4d6416d3d656fb4319f958522b8fc0bbb" dependencies = [ "bytes", "futures-util", @@ -3640,7 +3640,7 @@ dependencies = [ [[package]] name = "harmonia-utils-signature" version = "3.3.0" -source = "git+https://github.com/nix-community/harmonia.git#acf86e8a667266a7cfe54714db503dffafd14d60" +source = "git+https://github.com/DerDennisOP/harmonia.git?branch=build-resource-usage#ef33c7a4d6416d3d656fb4319f958522b8fc0bbb" dependencies = [ "data-encoding", "ed25519-dalek 3.0.0", @@ -5968,7 +5968,7 @@ dependencies = [ "once_cell", "socket2", "tracing", - "windows-sys 0.61.2", + "windows-sys 0.52.0", ] [[package]] @@ -6485,7 +6485,7 @@ dependencies = [ "errno", "libc", "linux-raw-sys 0.4.15", - "windows-sys 0.59.0", + "windows-sys 0.52.0", ] [[package]] @@ -6498,7 +6498,7 @@ dependencies = [ "errno", "libc", "linux-raw-sys 0.12.1", - "windows-sys 0.61.2", + "windows-sys 0.52.0", ] [[package]] @@ -6557,7 +6557,7 @@ dependencies = [ "security-framework", "security-framework-sys", "webpki-root-certs", - "windows-sys 0.61.2", + "windows-sys 0.52.0", ] [[package]] @@ -7722,10 +7722,10 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "32497e9a4c7b38532efcdebeef879707aa9f794296a4f0244f6f69e9bc8574bd" dependencies = [ "fastrand", - "getrandom 0.4.3", + "getrandom 0.3.4", "once_cell", "rustix 1.1.5", - "windows-sys 0.61.2", + "windows-sys 0.52.0", ] [[package]] @@ -8610,7 +8610,7 @@ version = "0.1.11" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "c2a7b1c03c876122aa43f3020e6c3c3ee5c05081c9a00739faf7503aeba10d22" dependencies = [ - "windows-sys 0.61.2", + "windows-sys 0.52.0", ] [[package]] @@ -8729,15 +8729,6 @@ dependencies = [ "windows-targets", ] -[[package]] -name = "windows-sys" -version = "0.59.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1e38bc4d79ed67fd075bcc251a1c39b32a1776bbe92e5bef1f0bf1f8c531853b" -dependencies = [ - "windows-targets", -] - [[package]] name = "windows-sys" version = "0.61.2" diff --git a/backend/Cargo.toml b/backend/Cargo.toml index de66bd31a..6e0160015 100644 --- a/backend/Cargo.toml +++ b/backend/Cargo.toml @@ -157,20 +157,20 @@ russh = { version = "0.63", default-features = false, features = ["aws-lc- shell-words = { version = "1.1", default-features = false, features = ["std"] } socket2 = { version = "0.6", default-features = false } -harmonia-daemon = { git = "https://github.com/nix-community/harmonia.git", default-features = false } -harmonia-file-nar = { git = "https://github.com/nix-community/harmonia.git", default-features = false } -harmonia-protocol = { git = "https://github.com/nix-community/harmonia.git", default-features = false } -harmonia-store-aterm = { git = "https://github.com/nix-community/harmonia.git", default-features = false } -harmonia-store-content-address = { git = "https://github.com/nix-community/harmonia.git", default-features = false } -harmonia-store-derivation = { git = "https://github.com/nix-community/harmonia.git", default-features = false } -harmonia-store-db = { git = "https://github.com/nix-community/harmonia.git", default-features = false } -harmonia-store-path = { git = "https://github.com/nix-community/harmonia.git", default-features = false } -harmonia-store-path-info = { git = "https://github.com/nix-community/harmonia.git", default-features = false } -harmonia-store-remote = { git = "https://github.com/nix-community/harmonia.git", default-features = false } -harmonia-utils-base-encoding = { git = "https://github.com/nix-community/harmonia.git", default-features = false } -harmonia-utils-hash = { git = "https://github.com/nix-community/harmonia.git", default-features = false } -harmonia-utils-io = { git = "https://github.com/nix-community/harmonia.git", default-features = false } -harmonia-utils-signature = { git = "https://github.com/nix-community/harmonia.git", default-features = false } +harmonia-daemon = { git = "https://github.com/DerDennisOP/harmonia.git", branch = "build-resource-usage", default-features = false } +harmonia-file-nar = { git = "https://github.com/DerDennisOP/harmonia.git", branch = "build-resource-usage", default-features = false } +harmonia-protocol = { git = "https://github.com/DerDennisOP/harmonia.git", branch = "build-resource-usage", default-features = false } +harmonia-store-aterm = { git = "https://github.com/DerDennisOP/harmonia.git", branch = "build-resource-usage", default-features = false } +harmonia-store-content-address = { git = "https://github.com/DerDennisOP/harmonia.git", branch = "build-resource-usage", default-features = false } +harmonia-store-derivation = { git = "https://github.com/DerDennisOP/harmonia.git", branch = "build-resource-usage", default-features = false } +harmonia-store-db = { git = "https://github.com/DerDennisOP/harmonia.git", branch = "build-resource-usage", default-features = false } +harmonia-store-path = { git = "https://github.com/DerDennisOP/harmonia.git", branch = "build-resource-usage", default-features = false } +harmonia-store-path-info = { git = "https://github.com/DerDennisOP/harmonia.git", branch = "build-resource-usage", default-features = false } +harmonia-store-remote = { git = "https://github.com/DerDennisOP/harmonia.git", branch = "build-resource-usage", default-features = false } +harmonia-utils-base-encoding = { git = "https://github.com/DerDennisOP/harmonia.git", branch = "build-resource-usage", default-features = false } +harmonia-utils-hash = { git = "https://github.com/DerDennisOP/harmonia.git", branch = "build-resource-usage", default-features = false } +harmonia-utils-io = { git = "https://github.com/DerDennisOP/harmonia.git", branch = "build-resource-usage", default-features = false } +harmonia-utils-signature = { git = "https://github.com/DerDennisOP/harmonia.git", branch = "build-resource-usage", default-features = false } # Nix bindings. The version tracks the FORK below, not upstream: the patch # replaces the source, so only the requirement is resolved against the index. diff --git a/backend/deny.toml b/backend/deny.toml index ba3f5222b..96220742b 100644 --- a/backend/deny.toml +++ b/backend/deny.toml @@ -47,7 +47,7 @@ ignore = [] unknown-registry = "deny" unknown-git = "deny" allow-git = [ - "https://github.com/nix-community/harmonia", + "https://github.com/DerDennisOP/harmonia", "https://github.com/DerDennisOP/nix-bindings", "https://github.com/launchbadge/sqlx", ] diff --git a/backend/gradient-cache/src/cacher/test_support.rs b/backend/gradient-cache/src/cacher/test_support.rs index 4f92748f8..7a84bf756 100644 --- a/backend/gradient-cache/src/cacher/test_support.rs +++ b/backend/gradient-cache/src/cacher/test_support.rs @@ -55,6 +55,7 @@ pub(crate) fn test_server_state_with_log( git_host: gradient_git_host::GitHostRegistry::with_builtin(), github_app_install_url: Default::default(), upstream_query: Arc::new(tokio::sync::Semaphore::new(32)), + nar_downloads: std::sync::Arc::new(tokio::sync::Semaphore::new(16)), upload_admission: gradient_storage::admission::UploadAdmission::new( gradient_storage::admission::Limits { concurrency: 16, diff --git a/backend/gradient-core/src/lib.rs b/backend/gradient-core/src/lib.rs index 889d5336f..0c0b50ed3 100644 --- a/backend/gradient-core/src/lib.rs +++ b/backend/gradient-core/src/lib.rs @@ -197,6 +197,7 @@ pub async fn init_state(cli: Cli) -> Result, InitError> { }; let upstream_query_concurrency = config.cache.upstream_query_concurrency; + let nar_downloads = config.nar.max_concurrent_downloads.max(1); let upload_limits = gradient_storage::admission::Limits { concurrency: config.upload.concurrency.max(1), bytes: config.upload.bytes_budget.max(1), @@ -216,6 +217,7 @@ pub async fn init_state(cli: Cli) -> Result, InitError> { upstream_query: Arc::new(tokio::sync::Semaphore::new( upstream_query_concurrency.max(1), )), + nar_downloads: Arc::new(tokio::sync::Semaphore::new(nar_downloads)), upload_admission: gradient_storage::admission::UploadAdmission::new(upload_limits), git_host: gradient_git_host::GitHostRegistry::with_builtin(), github_app_install_url: Default::default(), diff --git a/backend/gradient-core/src/state_root.rs b/backend/gradient-core/src/state_root.rs index 6ed9b8c55..f1e43d134 100644 --- a/backend/gradient-core/src/state_root.rs +++ b/backend/gradient-core/src/state_root.rs @@ -46,6 +46,7 @@ pub struct AppState { pub nar_storage: NarStore, pub http: reqwest::Client, pub upstream_query: Arc, + pub nar_downloads: Arc, pub upload_admission: Arc, pub git_host: GitHostRegistry, pub github_app_install_url: Arc>, diff --git a/backend/gradient-daemon/src/mock/build.rs b/backend/gradient-daemon/src/mock/build.rs index d22f444b5..538f427df 100644 --- a/backend/gradient-daemon/src/mock/build.rs +++ b/backend/gradient-daemon/src/mock/build.rs @@ -249,6 +249,10 @@ fn success(built_outputs: BuiltOutputs, started: i64) -> BuildResult { stop_time: now_secs(), cpu_user: None, cpu_system: None, + memory_peak: None, + io_read_bytes: None, + io_write_bytes: None, + oom_kills: None, } } @@ -264,6 +268,10 @@ fn failure(status: FailureStatus, msg: &str) -> BuildResult { stop_time: 0, cpu_user: None, cpu_system: None, + memory_peak: None, + io_read_bytes: None, + io_write_bytes: None, + oom_kills: None, } } diff --git a/backend/gradient-daemon/src/mock/conn.rs b/backend/gradient-daemon/src/mock/conn.rs index 9d0d0395e..f46e0cdc4 100644 --- a/backend/gradient-daemon/src/mock/conn.rs +++ b/backend/gradient-daemon/src/mock/conn.rs @@ -282,6 +282,10 @@ fn already_valid(path: &DerivedPath) -> KeyedBuildResult { stop_time: 0, cpu_user: None, cpu_system: None, + memory_peak: None, + io_read_bytes: None, + io_write_bytes: None, + oom_kills: None, }, } } diff --git a/backend/gradient-db/src/scheduling/priority.rs b/backend/gradient-db/src/scheduling/priority.rs index 62aec3efa..02fa56ffd 100644 --- a/backend/gradient-db/src/scheduling/priority.rs +++ b/backend/gradient-db/src/scheduling/priority.rs @@ -4,11 +4,13 @@ * SPDX-License-Identifier: AGPL-3.0-only */ -use crate::{DbContext, graph::closure::transitive_closure_reachable}; +use crate::build_request_task::BUILD_REQUEST_TASK_NAME; +use crate::{DbContext, fetch_in_chunks, graph::closure::transitive_closure_reachable}; use gradient_entity::build::BuildStatus; use gradient_entity::evaluation::EvaluationStatus; use gradient_types::*; use sea_orm::{ConnectionTrait, DbErr, FromQueryResult}; +use std::collections::HashSet; pub const PRIORITIZABLE: [BuildStatus; 4] = [ BuildStatus::Created, @@ -18,7 +20,7 @@ pub const PRIORITIZABLE: [BuildStatus; 4] = [ ]; #[derive(FromQueryResult)] -struct SharedBuildRow { +struct IdRow { id: uuid::Uuid, } @@ -26,11 +28,7 @@ fn prioritize_evaluation_sql() -> String { format!( "UPDATE evaluation SET prioritized = true \ WHERE id = $1 AND status NOT IN ({}) RETURNING id", - crate::sql::status::eval_in(&[ - EvaluationStatus::Completed, - EvaluationStatus::Failed, - EvaluationStatus::Aborted, - ]) + crate::sql::status::eval_in(&EvaluationStatus::TERMINAL) ) } @@ -70,6 +68,45 @@ crate::sql_fn! { tier = Bulk; } +/// Mirrors the scheduler's QoS lift: a running evaluation that a user prioritized or that runs a +/// build request. +fn evaluation_has_qos_sql(evaluation: &str) -> String { + format!( + "{evaluation}.status NOT IN ({}) AND ({evaluation}.prioritized OR EXISTS (\ + SELECT 1 FROM task t WHERE t.id = {evaluation}.task AND t.managed AND t.name = '{}'))", + crate::sql::status::eval_in(&EvaluationStatus::TERMINAL), + BUILD_REQUEST_TASK_NAME, + ) +} + +fn evaluations_with_qos_sql() -> String { + format!( + "SELECT e.id AS id FROM evaluation e WHERE e.id = ANY($1) AND {}", + evaluation_has_qos_sql("e") + ) +} + +crate::sql_fn! { + EVALUATIONS_WITH_QOS = evaluations_with_qos_sql, + params = [EvaluationIds(64)]; +} + +fn shared_builds_with_qos_sql() -> String { + format!( + "SELECT db.id AS id FROM derivation_build db \ + WHERE db.id = ANY($1) AND (db.prioritized OR EXISTS (\ + SELECT 1 FROM build_job bj JOIN evaluation e ON e.id = bj.evaluation \ + WHERE bj.derivation_build = db.id AND {}))", + evaluation_has_qos_sql("e") + ) +} + +crate::sql_fn! { + SHARED_BUILDS_WITH_QOS = shared_builds_with_qos_sql, + params = [SharedBuildIds(64)], + tier = Bulk; +} + pub async fn prioritize_evaluation( ctx: &DbContext, evaluation: EvaluationId, @@ -83,7 +120,7 @@ pub async fn prioritize_evaluation( return Ok(Vec::new()); } - let rows = SharedBuildRow::find_by_statement(EVALUATION_OPEN_SHARED_BUILDS.bind([id])) + let rows = IdRow::find_by_statement(EVALUATION_OPEN_SHARED_BUILDS.bind([id])) .all(&ctx.worker_db) .await?; Ok(rows @@ -104,7 +141,7 @@ pub async fn prioritize_build_closure( .collect(); closure.sort_unstable(); - let rows = SharedBuildRow::find_by_statement(PRIORITIZE_SHARED_BUILDS.bind([closure.into()])) + let rows = IdRow::find_by_statement(PRIORITIZE_SHARED_BUILDS.bind([closure.into()])) .all(&ctx.worker_db) .await?; Ok(rows @@ -113,6 +150,37 @@ pub async fn prioritize_build_closure( .collect()) } +pub async fn evaluations_with_qos( + db: &C, + evaluations: &[EvaluationId], +) -> Result, DbErr> { + let rows = fetch_in_chunks(evaluations, |chunk| async move { + let ids: Vec = chunk.iter().map(|id| id.into_inner()).collect(); + IdRow::find_by_statement(EVALUATIONS_WITH_QOS.bind([ids.into()])) + .all(db) + .await + }) + .await?; + Ok(rows.into_iter().map(|r| EvaluationId::from(r.id)).collect()) +} + +pub async fn shared_builds_with_qos( + db: &C, + shared_builds: &[DerivationBuildId], +) -> Result, DbErr> { + let rows = fetch_in_chunks(shared_builds, |chunk| async move { + let ids: Vec = chunk.iter().map(|id| id.into_inner()).collect(); + IdRow::find_by_statement(SHARED_BUILDS_WITH_QOS.bind([ids.into()])) + .all(db) + .await + }) + .await?; + Ok(rows + .into_iter() + .map(|r| DerivationBuildId::from(r.id)) + .collect()) +} + #[cfg(test)] mod tests { use super::*; diff --git a/backend/gradient-entity/src/derivation_metric.rs b/backend/gradient-entity/src/derivation_metric.rs index 295ca76b3..35621a8ef 100644 --- a/backend/gradient-entity/src/derivation_metric.rs +++ b/backend/gradient-entity/src/derivation_metric.rs @@ -24,9 +24,11 @@ pub struct Model { pub avg_cpu_pct: Option, pub disk_read_bytes: Option, pub disk_write_bytes: Option, - pub peak_network_mbps: Option, pub oom_killed: bool, pub build_time_ms: Option, + pub concurrent_builds: Option, + pub build_cores: Option, + pub cpu_core_score: Option, pub worker_id: String, pub created_at: NaiveDateTime, } diff --git a/backend/gradient-entity/src/worker_sample.rs b/backend/gradient-entity/src/worker_sample.rs index cdda9ffe6..9defce40e 100644 --- a/backend/gradient-entity/src/worker_sample.rs +++ b/backend/gradient-entity/src/worker_sample.rs @@ -57,7 +57,8 @@ pub struct Model { pub ram_free_mb: Option, pub ram_total_mb: Option, pub disk_speed_mbps: Option, - pub network_speed_mbps: Option, + pub upload_speed_mbps: Option, + pub download_speed_mbps: Option, pub assigned_jobs: i32, pub max_concurrent_builds: i32, pub state: WorkerSampleState, diff --git a/backend/gradient-graph/src/messages.rs b/backend/gradient-graph/src/messages.rs index 9097e1845..94a806c12 100644 --- a/backend/gradient-graph/src/messages.rs +++ b/backend/gradient-graph/src/messages.rs @@ -107,6 +107,7 @@ pub enum Transition { log_banner: String, kind: BuildFailureKind, missing_paths: Vec, + metrics: Option, }, Assigned { evaluation: EvaluationId, diff --git a/backend/gradient-graph/src/policy.rs b/backend/gradient-graph/src/policy.rs index a16d7d107..025846759 100644 --- a/backend/gradient-graph/src/policy.rs +++ b/backend/gradient-graph/src/policy.rs @@ -6,7 +6,7 @@ use gradient_entity::build::BuildStatus; use gradient_entity::build_attempt::{AttemptFailureReason, AttemptOutcome}; -use gradient_wire::types::BuildFailureKind; +use gradient_wire::types::{BuildFailureKind, BuildMetrics}; #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub(crate) struct Substitution { @@ -122,6 +122,23 @@ pub(crate) fn attempt_outcome(kind: BuildFailureKind) -> AttemptOutcome { } } +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(crate) enum BuildEnd { + Built, + Substituted, + Failed, +} + +/// A failed build is only joining the history when it ran out of memory. Its time and CPU are +/// partial, but its peak memory is what the scheduler must not repeat. +pub(crate) fn history_sample(metrics: Option, end: BuildEnd) -> Option { + metrics.filter(|m| match end { + BuildEnd::Built => true, + BuildEnd::Substituted => false, + BuildEnd::Failed => m.oom_killed, + }) +} + pub(crate) fn inputs_unavailable_circuit_open(prior_failures: i64, max_loops: u32) -> bool { prior_failures >= max_loops as i64 } @@ -164,14 +181,42 @@ mod tests { } } use super::{ - FailureOutcome, Substitution, attempt_outcome, attempt_reason, attempt_reason_for, - decide_failure_outcome, inputs_unavailable_circuit_open, retry_backoff_elapsed, - retry_failed_eval, spends_substitute_budget, terminal_success_outcome, - terminal_success_status, truncate_failure_message, + BuildEnd, FailureOutcome, Substitution, attempt_outcome, attempt_reason, + attempt_reason_for, decide_failure_outcome, history_sample, + inputs_unavailable_circuit_open, retry_backoff_elapsed, retry_failed_eval, + spends_substitute_budget, terminal_success_outcome, terminal_success_status, + truncate_failure_message, }; use gradient_entity::build::BuildStatus; use gradient_entity::build_attempt::{AttemptFailureReason, AttemptOutcome}; - use gradient_wire::types::BuildFailureKind; + use gradient_wire::types::{BuildFailureKind, BuildMetrics}; + + #[test] + fn history_keeps_real_builds_and_out_of_memory_failures_only() { + let measured = BuildMetrics { + peak_ram_mb: Some(512), + ..Default::default() + }; + let killed = BuildMetrics { + oom_killed: true, + ..measured.clone() + }; + + assert_eq!( + history_sample(Some(measured.clone()), BuildEnd::Built), + Some(measured.clone()) + ); + assert_eq!( + history_sample(Some(measured.clone()), BuildEnd::Substituted), + None + ); + assert_eq!(history_sample(Some(measured), BuildEnd::Failed), None); + assert_eq!( + history_sample(Some(killed.clone()), BuildEnd::Failed), + Some(killed) + ); + assert_eq!(history_sample(None, BuildEnd::Built), None); + } #[test] fn abort_is_not_a_deterministic_build_failure() { diff --git a/backend/gradient-graph/src/transition.rs b/backend/gradient-graph/src/transition.rs index 813d3a19e..98d26dd19 100644 --- a/backend/gradient-graph/src/transition.rs +++ b/backend/gradient-graph/src/transition.rs @@ -82,8 +82,18 @@ pub(crate) async fn apply(ctx: &DbContext, transition: Transition) -> Result { - build_failed(ctx, shared_build, &error, &log_banner, kind, &missing_paths).await?; + build_failed( + ctx, + shared_build, + &error, + &log_banner, + kind, + &missing_paths, + metrics, + ) + .await?; Ok(TransitionReport::default()) } Transition::Assigned { @@ -316,7 +326,12 @@ async fn build_output( let build_id = shared_build.id; let derivation_id = shared_build.derivation; - if let Some(metrics) = metrics { + let end = if substituted { + policy::BuildEnd::Substituted + } else { + policy::BuildEnd::Built + }; + if let Some(metrics) = policy::history_sample(metrics, end) { record_metrics(ctx, &shared_build, derivation_id, &metrics).await; } @@ -497,6 +512,7 @@ async fn build_failed( log_banner: &str, kind: BuildFailureKind, missing_paths: &[String], + metrics: Option, ) -> Result<()> { let Some(shared_build) = EDerivationBuild::find_by_id(derivation_build) .one(&ctx.worker_db) @@ -506,6 +522,10 @@ async fn build_failed( return Ok(()); }; + if let Some(metrics) = policy::history_sample(metrics, policy::BuildEnd::Failed) { + record_metrics(ctx, &shared_build, shared_build.derivation, &metrics).await; + } + if let Some(attempt_id) = gradient_db::scheduling::build_attempt::latest_attempt_id(&ctx.worker_db, shared_build.id) .await @@ -763,9 +783,11 @@ async fn record_metrics( avg_cpu_pct: metrics.avg_cpu_pct.map(|v| v as f64), disk_read_bytes: metrics.disk_read_bytes.map(|v| v as i64), disk_write_bytes: metrics.disk_write_bytes.map(|v| v as i64), - peak_network_mbps: metrics.peak_network_mbps.map(|v| v as f64), oom_killed: metrics.oom_killed, build_time_ms: metrics.build_time_ms.map(|v| v as i64), + concurrent_builds: metrics.concurrent_builds.map(|v| v as i32), + build_cores: metrics.build_cores.map(|v| v as i32), + cpu_core_score: metrics.cpu_core_score.map(|v| v as i32), worker_id: gradient_db::scheduling::build_attempt::latest_attempt_worker( &ctx.worker_db, shared_build.id, diff --git a/backend/gradient-migration/src/lib.rs b/backend/gradient-migration/src/lib.rs index 3cb95586d..3878bbd73 100644 --- a/backend/gradient-migration/src/lib.rs +++ b/backend/gradient-migration/src/lib.rs @@ -108,6 +108,8 @@ mod m20261003_000000_adopt_cached_nar_references; mod m20261003_000001_unique_task_name; mod m20261004_000000_teams; mod m20261004_000001_drop_base_workers; +mod m20261004_000002_transfer_speeds; +mod m20261004_000003_build_conditions; pub struct Migrator; @@ -212,6 +214,8 @@ impl MigratorTrait for Migrator { Box::new(m20261003_000001_unique_task_name::Migration), Box::new(m20261004_000000_teams::Migration), Box::new(m20261004_000001_drop_base_workers::Migration), + Box::new(m20261004_000002_transfer_speeds::Migration), + Box::new(m20261004_000003_build_conditions::Migration), ] } } diff --git a/backend/gradient-migration/src/m20261004_000002_transfer_speeds.rs b/backend/gradient-migration/src/m20261004_000002_transfer_speeds.rs new file mode 100644 index 000000000..acdd905e3 --- /dev/null +++ b/backend/gradient-migration/src/m20261004_000002_transfer_speeds.rs @@ -0,0 +1,48 @@ +/* + * SPDX-FileCopyrightText: 2026 Wavelens GmbH + * + * SPDX-License-Identifier: AGPL-3.0-only + */ + +use sea_orm_migration::prelude::*; +use sea_orm_migration::sea_orm::ConnectionTrait; + +const UP: [&str; 2] = [ + "ALTER TABLE derivation_metric DROP COLUMN peak_network_mbps", + "ALTER TABLE worker_sample DROP COLUMN network_speed_mbps, \ + ADD COLUMN upload_speed_mbps real, ADD COLUMN download_speed_mbps real", +]; + +const DOWN: [&str; 2] = [ + "ALTER TABLE worker_sample DROP COLUMN upload_speed_mbps, \ + DROP COLUMN download_speed_mbps, ADD COLUMN network_speed_mbps real", + "ALTER TABLE derivation_metric ADD COLUMN peak_network_mbps double precision", +]; + +#[derive(DeriveMigrationName)] +pub struct Migration; + +#[async_trait::async_trait] +impl MigrationTrait for Migration { + async fn up(&self, manager: &SchemaManager) -> Result<(), DbErr> { + for statement in UP { + manager + .get_connection() + .execute_unprepared(statement) + .await?; + } + + Ok(()) + } + + async fn down(&self, manager: &SchemaManager) -> Result<(), DbErr> { + for statement in DOWN { + manager + .get_connection() + .execute_unprepared(statement) + .await?; + } + + Ok(()) + } +} diff --git a/backend/gradient-migration/src/m20261004_000003_build_conditions.rs b/backend/gradient-migration/src/m20261004_000003_build_conditions.rs new file mode 100644 index 000000000..f03f4990d --- /dev/null +++ b/backend/gradient-migration/src/m20261004_000003_build_conditions.rs @@ -0,0 +1,34 @@ +/* + * SPDX-FileCopyrightText: 2026 Wavelens GmbH + * + * SPDX-License-Identifier: AGPL-3.0-only + */ + +use sea_orm_migration::prelude::*; +use sea_orm_migration::sea_orm::ConnectionTrait; + +const UP: &str = "ALTER TABLE derivation_metric \ + ADD COLUMN concurrent_builds integer, \ + ADD COLUMN build_cores integer, \ + ADD COLUMN cpu_core_score integer"; + +const DOWN: &str = "ALTER TABLE derivation_metric \ + DROP COLUMN concurrent_builds, \ + DROP COLUMN build_cores, \ + DROP COLUMN cpu_core_score"; + +#[derive(DeriveMigrationName)] +pub struct Migration; + +#[async_trait::async_trait] +impl MigrationTrait for Migration { + async fn up(&self, manager: &SchemaManager) -> Result<(), DbErr> { + manager.get_connection().execute_unprepared(UP).await?; + Ok(()) + } + + async fn down(&self, manager: &SchemaManager) -> Result<(), DbErr> { + manager.get_connection().execute_unprepared(DOWN).await?; + Ok(()) + } +} diff --git a/backend/gradient-pool/src/score/context.rs b/backend/gradient-pool/src/score/context.rs index 5367dbfe6..110e2cb3c 100644 --- a/backend/gradient-pool/src/score/context.rs +++ b/backend/gradient-pool/src/score/context.rs @@ -31,7 +31,6 @@ pub struct InstanceContext { pub cpu_time_ms: Windowed, pub avg_cpu_pct: Windowed, pub disk_bytes: Windowed, - pub network_mbps: Windowed, pub oom_rate: Windowed, pub closure_size: Windowed, pub nar_size_mb: Windowed, @@ -43,14 +42,27 @@ pub struct InstanceContext { pub total_workers: u32, pub idle_workers: u32, pub cpu_core_score_mean: Option, + pub upload_speed_mean_mbps: Option, + pub download_speed_mean_mbps: Option, + pub downloads_in_flight: u32, + pub uploads_in_flight: u32, + pub download_slots: u32, + pub upload_slots: u32, + pub storage_read_mbps: Option, + pub storage_write_mbps: Option, + pub compression_ratio: Option, + pub per_path_secs: Option, } #[derive(Clone, Copy, Debug, Default, PartialEq)] pub struct HistoryPrediction { - pub predicted_peak_ram_mb: u64, - pub avg_cpu_time_ms: u64, - pub build_time_ms: u64, - pub avg_disk_bytes: u64, + pub predicted_peak_ram_mb: Option, + pub avg_cpu_time_ms: Option, + pub build_time_ms: Option, + pub uncontended_build_time_ms: Option, + pub build_core_score: Option, + pub avg_disk_bytes: Option, + pub output_nar_size: Option, pub oom_rate: f32, pub samples: u32, } @@ -63,7 +75,9 @@ pub struct WorkerMetricsView { pub ram_free_mb: Option, pub cpu_usage_pct: Option, pub disk_speed_mbps: Option, - pub network_speed_mbps: Option, + pub upload_speed_mbps: Option, + pub download_speed_mbps: Option, + pub running_builds: u32, } #[derive(Clone, Debug, PartialEq, serde::Serialize, serde::Deserialize)] @@ -101,19 +115,22 @@ impl BuildContext { } else { None }; - out.history.predicted_peak_ram_mb = items - .iter() - .map(|i| i.history.predicted_peak_ram_mb) - .max() - .unwrap_or(0); + let known = |value: fn(&HistoryPrediction) -> Option| { + items.iter().filter_map(move |i| value(&i.history)) + }; + out.history.predicted_peak_ram_mb = known(|h| h.predicted_peak_ram_mb).max(); out.history.oom_rate = items.iter().map(|i| i.history.oom_rate).fold(0.0, f32::max); - out.history.avg_cpu_time_ms = items.iter().map(|i| i.history.avg_cpu_time_ms).sum(); - out.history.build_time_ms = items + out.history.avg_cpu_time_ms = known(|h| h.avg_cpu_time_ms).reduce(u64::saturating_add); + out.history.build_time_ms = known(|h| h.build_time_ms).max(); + let longest = items .iter() - .map(|i| i.history.build_time_ms) - .max() - .unwrap_or(0); - out.history.avg_disk_bytes = items.iter().map(|i| i.history.avg_disk_bytes).sum(); + .filter(|i| i.history.uncontended_build_time_ms.is_some()) + .max_by_key(|i| i.history.uncontended_build_time_ms); + out.history.uncontended_build_time_ms = + longest.and_then(|i| i.history.uncontended_build_time_ms); + out.history.build_core_score = longest.and_then(|i| i.history.build_core_score); + out.history.avg_disk_bytes = known(|h| h.avg_disk_bytes).reduce(u64::saturating_add); + out.history.output_nar_size = known(|h| h.output_nar_size).reduce(u64::saturating_add); out.history.samples = items.iter().map(|i| i.history.samples).min().unwrap_or(0); out.derivations = items.iter().flat_map(|i| i.derivations.clone()).collect(); out @@ -272,10 +289,13 @@ mod tests { pname: Some("curl".into()), closure_size: Some(100), history: HistoryPrediction { - predicted_peak_ram_mb: 500, - avg_cpu_time_ms: 1000, - build_time_ms: 0, - avg_disk_bytes: 10, + predicted_peak_ram_mb: Some(500), + avg_cpu_time_ms: Some(1000), + build_time_ms: None, + uncontended_build_time_ms: None, + build_core_score: None, + avg_disk_bytes: Some(10), + output_nar_size: Some(7), oom_rate: 0.1, samples: 5, }, @@ -284,14 +304,19 @@ mod tests { assert_eq!(BuildContext::aggregate(std::slice::from_ref(&a)), a); let mut b = a.clone(); - b.history.predicted_peak_ram_mb = 900; - b.history.avg_cpu_time_ms = 4000; + b.history.predicted_peak_ram_mb = Some(900); + b.history.avg_cpu_time_ms = Some(4000); + b.history.avg_disk_bytes = None; + b.history.output_nar_size = Some(3); b.history.samples = 2; b.pname = Some("git".into()); b.prefer_local_build = true; let agg = BuildContext::aggregate(&[a.clone(), b]); - assert_eq!(agg.history.predicted_peak_ram_mb, 900); - assert_eq!(agg.history.avg_cpu_time_ms, 5000); + assert_eq!(agg.history.predicted_peak_ram_mb, Some(900)); + assert_eq!(agg.history.avg_cpu_time_ms, Some(5000)); + assert_eq!(agg.history.avg_disk_bytes, Some(10)); + assert_eq!(agg.history.output_nar_size, Some(10)); + assert_eq!(agg.history.build_time_ms, None); assert_eq!(agg.history.samples, 2); assert_eq!(agg.dependency_count, 4); assert!(agg.prefer_local_build); diff --git a/backend/gradient-pool/src/score/mod.rs b/backend/gradient-pool/src/score/mod.rs index c2790dd40..36d33091e 100644 --- a/backend/gradient-pool/src/score/mod.rs +++ b/backend/gradient-pool/src/score/mod.rs @@ -18,3 +18,4 @@ pub use context::{ }; pub use policy::{RulePolicy, ScoringPolicy, policy_by_name, rule_catalog}; pub use rule::{JobContext, ScoreRule, WorkerContext}; +pub use rules::estimated_time::contention_factor; diff --git a/backend/gradient-pool/src/score/policy.rs b/backend/gradient-pool/src/score/policy.rs index ead3d8bdb..548940cab 100644 --- a/backend/gradient-pool/src/score/policy.rs +++ b/backend/gradient-pool/src/score/policy.rs @@ -6,13 +6,9 @@ use crate::score::context::InstanceContext; use crate::score::rule::{JobContext, ScoreRule, WorkerContext}; -use crate::score::rules::builtin::{ - BuiltinDeprioritizeRule, DependencyCountRule, MissingNarSizeRule, MissingPathsRule, - RealisedOutputsRule, RescoreWaitRule, ReserveFetchWorkersRule, WaitTimeRule, -}; use crate::score::rules::{ - CpuAffinityRule, DiskAffinityRule, FairShareRule, NetworkAffinityRule, PreferLocalBuildRule, - QosRule, ResourceFitRule, ResourceSaturationRule, + EstimatedTimeRule, FairShareRule, QosRule, RescoreWaitRule, ReserveFetchWorkersRule, + ResourceSaturationRule, TransferLimitRule, WaitTimeRule, }; pub trait ScoringPolicy: Send + Sync + std::fmt::Debug { @@ -124,29 +120,21 @@ fn spec(enabled: bool, rule: Box) -> RuleSpec { fn simple_table() -> Vec { vec![ - spec(true, Box::new(MissingPathsRule::default())), - spec(true, Box::new(MissingNarSizeRule::default())), - spec(true, Box::new(RealisedOutputsRule::default())), + spec(true, Box::new(EstimatedTimeRule::default())), spec(true, Box::new(RescoreWaitRule::default())), - spec(true, Box::new(DependencyCountRule::default())), spec(true, Box::new(WaitTimeRule::default())), - spec(true, Box::new(BuiltinDeprioritizeRule::default())), spec(true, Box::new(ReserveFetchWorkersRule::default())), + spec(true, Box::new(TransferLimitRule::default())), spec(true, Box::new(QosRule::default())), ] } fn resource_aware_table() -> Vec { let mut rules = simple_table(); - rules.push(spec(true, Box::new(ResourceFitRule::default()))); rules.push(spec(true, Box::new(ResourceSaturationRule::default()))); - rules.push(spec(true, Box::new(PreferLocalBuildRule::default()))); // FairShareRule is disabled because its idle gate is counting zero occupancy, not spare // capacity. Re-enabling it is a scheduling-policy decision (#476). rules.push(spec(false, Box::new(FairShareRule::default()))); - rules.push(spec(true, Box::new(NetworkAffinityRule::default()))); - rules.push(spec(true, Box::new(DiskAffinityRule::default()))); - rules.push(spec(true, Box::new(CpuAffinityRule::default()))); rules } @@ -305,60 +293,6 @@ mod tests { ); } - #[test] - fn resource_aware_prefers_fast_net_for_fod() { - use crate::score::context::WorkerMetricsView; - let policy = policy_by_name("resource-aware"); - let archs = vec!["x86_64-linux".to_string()]; - let feats: Vec = vec![]; - let j = ScoredJob::new_build( - "j", - ProjectId::now_v7(), - "x86_64-linux", - false, - true, - None, - None, - HistoryPrediction::default(), - ); - let c = JobContext { - job: &j, - missing_count: Some(0), - missing_nar_size: Some(0), - outputs_present: false, - dependency_count: 0, - queued_at: now(), - ready_at: now(), - project_work_share: None, - prioritized: false, - build_request: false, - rescore_count: 0, - now: now(), - }; - let fast = WorkerContext { - architectures: &archs, - system_features: &feats, - fetch: false, - metrics: Some(WorkerMetricsView { - network_speed_mbps: Some(100.0), - ..Default::default() - }), - }; - let slow = WorkerContext { - architectures: &archs, - system_features: &feats, - fetch: false, - metrics: Some(WorkerMetricsView { - network_speed_mbps: Some(5.0), - ..Default::default() - }), - }; - assert!( - policy.score(&c, &fast, &InstanceContext::default()) - > policy.score(&c, &slow, &InstanceContext::default()) - ); - } - #[test] fn resource_aware_sends_heavy_build_to_fast_cold_worker_over_slow_warm_one() { use crate::score::context::WorkerMetricsView; @@ -374,7 +308,9 @@ mod tests { None, None, HistoryPrediction { - avg_cpu_time_ms: 30 * 60_000, + build_time_ms: Some(20 * 60_000), + uncontended_build_time_ms: Some(20 * 60_000), + build_core_score: Some(10_000), samples: 5, ..Default::default() }, @@ -430,8 +366,9 @@ mod tests { None, None, HistoryPrediction { - avg_cpu_time_ms: 30 * 60_000, - predicted_peak_ram_mb: 64_000, + build_time_ms: Some(30 * 60_000), + uncontended_build_time_ms: Some(30 * 60_000), + predicted_peak_ram_mb: Some(64_000), samples: 5, ..Default::default() }, @@ -549,8 +486,8 @@ mod tests { (breakdown.total - total).abs() < 1e-9, "total must match score()" ); - assert_eq!(breakdown.rules.len(), 9, "simple policy has 9 rules"); - assert!(breakdown.rules.contains_key("MissingPathsRule")); + assert_eq!(breakdown.rules.len(), 6, "simple policy has 6 rules"); + assert!(breakdown.rules.contains_key("EstimatedTimeRule")); assert!(breakdown.rules.contains_key("QosRule")); assert!(breakdown.rules.contains_key("WaitTimeRule")); let sum: f64 = breakdown.rules.values().sum(); @@ -565,20 +502,12 @@ mod tests { #[test] fn rule_names_are_pinned() { let expected = [ - "BuiltinDeprioritizeRule", - "CpuAffinityRule", - "DependencyCountRule", - "DiskAffinityRule", - "MissingNarSizeRule", - "MissingPathsRule", - "NetworkAffinityRule", - "PreferLocalBuildRule", + "EstimatedTimeRule", "QosRule", - "RealisedOutputsRule", "RescoreWaitRule", "ReserveFetchWorkersRule", - "ResourceFitRule", "ResourceSaturationRule", + "TransferLimitRule", "WaitTimeRule", ]; let mut got: Vec<&str> = resource_aware_rules().iter().map(|r| r.name()).collect(); @@ -587,6 +516,37 @@ mod tests { assert_eq!(FairShareRule::default().name(), "FairShareRule"); } + #[test] + fn a_held_transfer_drops_below_the_floor_until_its_wait_releases_it() { + let policy = policy_by_name("simple"); + let archs = vec!["x86_64-linux".to_string()]; + let feats: Vec = vec![]; + let w = worker_ctx(&archs, &feats); + let j = scored_job("x86_64-linux"); + let waited = |secs| JobContext { + job: &j, + missing_count: Some(0), + missing_nar_size: Some(4 << 30), + outputs_present: false, + dependency_count: 0, + queued_at: now() - chrono::Duration::seconds(secs), + ready_at: now() - chrono::Duration::seconds(secs), + project_work_share: None, + prioritized: false, + build_request: false, + rescore_count: 0, + now: now(), + }; + let full = InstanceContext { + downloads_in_flight: 16, + download_slots: 16, + ..Default::default() + }; + + assert!(policy.score(&waited(0), &w, &full) < crate::score::weights::ASSIGN_FLOOR); + assert!(policy.score(&waited(36_000), &w, &full) >= crate::score::weights::ASSIGN_FLOOR); + } + #[test] fn unmeasured_build_is_vetoed_not_penalized() { let policy = policy_by_name("simple"); diff --git a/backend/gradient-pool/src/score/rules/affinity.rs b/backend/gradient-pool/src/score/rules/affinity.rs deleted file mode 100644 index 6acea7493..000000000 --- a/backend/gradient-pool/src/score/rules/affinity.rs +++ /dev/null @@ -1,447 +0,0 @@ -/* - * SPDX-FileCopyrightText: 2026 Wavelens GmbH - * - * SPDX-License-Identifier: AGPL-3.0-only - */ - -use crate::score::context::InstanceContext; -use crate::score::rule::{JobContext, ScoreRule, WorkerContext}; - -#[derive(Debug)] -pub struct NetworkAffinityRule { - pub bonus: f64, - pub reference_mbps: f64, -} - -impl Default for NetworkAffinityRule { - fn default() -> Self { - Self { - bonus: crate::score::weights::NETWORK_AFFINITY_BONUS, - reference_mbps: crate::score::weights::NETWORK_REFERENCE_MBPS, - } - } -} - -impl ScoreRule for NetworkAffinityRule { - fn name(&self) -> &'static str { - "NetworkAffinityRule" - } - - fn score( - &self, - job: &JobContext<'_>, - worker: &WorkerContext<'_>, - instance: &InstanceContext, - ) -> f64 { - let Some(b) = job.job.build() else { return 0.0 }; - if !b.is_fixed_output { - return 0.0; - } - - let Some(net) = worker.metrics.and_then(|m| m.network_speed_mbps) else { - return 0.0; - }; - - let reference = instance.network_mbps.w24h_or(self.reference_mbps); - self.bonus * (net as f64 / reference).min(1.0) - } - - fn description(&self) -> &'static str { - "Steers fixed-output (network-fetching) derivations towards workers with faster measured network throughput." - } -} - -#[derive(Debug)] -pub struct DiskAffinityRule { - pub bonus: f64, - pub heavy_threshold_bytes: u64, - pub reference_mbps: f64, -} - -impl Default for DiskAffinityRule { - fn default() -> Self { - Self { - bonus: crate::score::weights::DISK_AFFINITY_BONUS, - heavy_threshold_bytes: crate::score::weights::DISK_HEAVY_THRESHOLD_BYTES, - reference_mbps: crate::score::weights::DISK_REFERENCE_MBPS, - } - } -} - -impl ScoreRule for DiskAffinityRule { - fn name(&self) -> &'static str { - "DiskAffinityRule" - } - - fn score( - &self, - job: &JobContext<'_>, - worker: &WorkerContext<'_>, - instance: &InstanceContext, - ) -> f64 { - if job.job.build().is_none() { - return 0.0; - } - let h = job.build_history(); - let heavy_threshold = instance - .disk_bytes - .w24h_or(self.heavy_threshold_bytes as f64); - if h.samples == 0 || (h.avg_disk_bytes as f64) < heavy_threshold { - return 0.0; - } - - let Some(disk) = worker.metrics.and_then(|m| m.disk_speed_mbps) else { - return 0.0; - }; - self.bonus * (disk as f64 / self.reference_mbps).min(1.0) - } - - fn description(&self) -> &'static str { - "Steers builds that have historically been disk-heavy towards workers with faster measured disk throughput." - } -} - -#[derive(Debug)] -pub struct CpuAffinityRule { - pub weight: f64, - pub heaviness_cap: f64, - pub heavy_threshold_ms: u64, -} - -impl Default for CpuAffinityRule { - fn default() -> Self { - Self { - weight: crate::score::weights::CPU_AFFINITY_WEIGHT, - heaviness_cap: crate::score::weights::CPU_HEAVINESS_CAP, - heavy_threshold_ms: crate::score::weights::CPU_HEAVY_THRESHOLD_MS, - } - } -} - -impl CpuAffinityRule { - fn heaviness(&self, work_ms: u64, instance: &InstanceContext) -> Option { - let threshold = instance - .cpu_time_ms - .w1h_or(self.heavy_threshold_ms as f64) - .max(1.0); - let ratio = work_ms as f64 / threshold; - (ratio >= 1.0).then(|| (ratio.log2() + 1.0).min(self.heaviness_cap)) - } -} - -impl ScoreRule for CpuAffinityRule { - fn name(&self) -> &'static str { - "CpuAffinityRule" - } - - fn score( - &self, - job: &JobContext<'_>, - worker: &WorkerContext<'_>, - instance: &InstanceContext, - ) -> f64 { - let h = job.build_history(); - let cores = worker.metrics.map_or(0, |m| m.cpu_core_score); - let Some(fleet) = instance.cpu_core_score_mean.filter(|f| *f > 0.0) else { - return 0.0; - }; - if h.samples == 0 || cores == 0 { - return 0.0; - } - - let work_ms = if h.avg_cpu_time_ms > 0 { - h.avg_cpu_time_ms - } else { - h.build_time_ms - }; - let Some(heaviness) = self.heaviness(work_ms, instance) else { - return 0.0; - }; - - let relative_speed = (cores as f64 / fleet - 1.0).clamp(-1.0, 1.0); - self.weight * heaviness * relative_speed - } - - fn description(&self) -> &'static str { - "Steers builds that have historically been CPU-heavy towards workers with faster cores than the fleet average and away from slower ones, outweighing cache warmth for long builds." - } -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::score::context::{HistoryPrediction, ScoredJob, WorkerMetricsView}; - use gradient_types::ids::ProjectId; - - fn job(is_fixed_output: bool, h: HistoryPrediction) -> ScoredJob<'static> { - ScoredJob::new_build( - "t", - ProjectId::now_v7(), - "x86_64-linux", - false, - is_fixed_output, - None, - None, - h, - ) - } - - fn ctx<'a>(job: &'a ScoredJob<'a>) -> JobContext<'a> { - JobContext { - job, - missing_count: None, - missing_nar_size: None, - outputs_present: false, - dependency_count: 0, - queued_at: gradient_types::now(), - ready_at: gradient_types::now(), - project_work_share: None, - prioritized: false, - build_request: false, - rescore_count: 0, - now: gradient_types::now(), - } - } - - fn worker_with(metrics: WorkerMetricsView) -> WorkerContext<'static> { - WorkerContext { - architectures: &[], - system_features: &[], - fetch: false, - metrics: Some(metrics), - } - } - - #[test] - fn network_rule_prefers_fast_net_for_fod() { - let rule = NetworkAffinityRule::default(); - let j = job(true, HistoryPrediction::default()); - let fast = worker_with(WorkerMetricsView { - network_speed_mbps: Some(100.0), - ..Default::default() - }); - let slow = worker_with(WorkerMetricsView { - network_speed_mbps: Some(10.0), - ..Default::default() - }); - assert!( - rule.score(&ctx(&j), &fast, &InstanceContext::default()) - > rule.score(&ctx(&j), &slow, &InstanceContext::default()) - ); - } - - #[test] - fn network_rule_zero_for_non_fod() { - let rule = NetworkAffinityRule::default(); - let j = job(false, HistoryPrediction::default()); - let fast = worker_with(WorkerMetricsView { - network_speed_mbps: Some(100.0), - ..Default::default() - }); - assert_eq!( - rule.score(&ctx(&j), &fast, &InstanceContext::default()), - 0.0 - ); - } - - #[test] - fn network_rule_zero_without_metric() { - let rule = NetworkAffinityRule::default(); - let j = job(true, HistoryPrediction::default()); - let w = worker_with(WorkerMetricsView { - network_speed_mbps: None, - ..Default::default() - }); - assert_eq!(rule.score(&ctx(&j), &w, &InstanceContext::default()), 0.0); - } - - #[test] - fn disk_rule_prefers_fast_disk_for_heavy_build() { - let rule = DiskAffinityRule::default(); - let heavy = HistoryPrediction { - avg_disk_bytes: 500 * 1_048_576, - samples: 5, - ..Default::default() - }; - let j = job(false, heavy); - let fast = worker_with(WorkerMetricsView { - disk_speed_mbps: Some(500.0), - ..Default::default() - }); - let slow = worker_with(WorkerMetricsView { - disk_speed_mbps: Some(50.0), - ..Default::default() - }); - assert!( - rule.score(&ctx(&j), &fast, &InstanceContext::default()) - > rule.score(&ctx(&j), &slow, &InstanceContext::default()) - ); - } - - #[test] - fn disk_rule_zero_for_light_build() { - let rule = DiskAffinityRule::default(); - let light = HistoryPrediction { - avg_disk_bytes: 1_048_576, - samples: 5, - ..Default::default() - }; - let j = job(false, light); - let fast = worker_with(WorkerMetricsView { - disk_speed_mbps: Some(500.0), - ..Default::default() - }); - assert_eq!( - rule.score(&ctx(&j), &fast, &InstanceContext::default()), - 0.0 - ); - } - - #[test] - fn disk_rule_zero_without_history() { - let rule = DiskAffinityRule::default(); - let j = job( - false, - HistoryPrediction { - avg_disk_bytes: 999 * 1_048_576, - samples: 0, - ..Default::default() - }, - ); - let fast = worker_with(WorkerMetricsView { - disk_speed_mbps: Some(500.0), - ..Default::default() - }); - assert_eq!( - rule.score(&ctx(&j), &fast, &InstanceContext::default()), - 0.0 - ); - } - - #[test] - fn disk_heavy_uses_instance_threshold() { - let rule = DiskAffinityRule::default(); - let j = job( - false, - HistoryPrediction { - avg_disk_bytes: 50 * 1_048_576, - samples: 5, - ..Default::default() - }, - ); - let fast = worker_with(WorkerMetricsView { - disk_speed_mbps: Some(500.0), - ..Default::default() - }); - - assert_eq!( - rule.score(&ctx(&j), &fast, &InstanceContext::default()), - 0.0 - ); - - let mut inst = InstanceContext::default(); - inst.disk_bytes.w24h = Some((10 * 1_048_576) as f64); - assert!(rule.score(&ctx(&j), &fast, &inst) > 0.0); - } - - fn cpu_history(avg_cpu_time_ms: u64) -> HistoryPrediction { - HistoryPrediction { - avg_cpu_time_ms, - samples: 5, - ..Default::default() - } - } - - fn cores(cpu_core_score: u32) -> WorkerContext<'static> { - worker_with(WorkerMetricsView { - cpu_core_score, - ..Default::default() - }) - } - - fn fleet(mean: f64) -> InstanceContext { - InstanceContext { - cpu_core_score_mean: Some(mean), - ..Default::default() - } - } - - #[test] - fn cpu_rule_attracts_heavy_builds_to_faster_than_fleet_workers() { - let rule = CpuAffinityRule::default(); - let j = job(false, cpu_history(10 * 60_000)); - let inst = fleet(10_000.0); - - assert!(rule.score(&ctx(&j), &cores(15_000), &inst) > 0.0); - assert!(rule.score(&ctx(&j), &cores(5_000), &inst) < 0.0); - assert_eq!(rule.score(&ctx(&j), &cores(10_000), &inst), 0.0); - } - - #[test] - fn cpu_rule_weighs_heavier_builds_more_up_to_a_bound() { - let rule = CpuAffinityRule::default(); - let inst = fleet(10_000.0); - let fast = cores(1_000_000); - let score = |ms| rule.score(&ctx(&job(false, cpu_history(ms))), &fast, &inst); - - assert!(score(4 * 60_000) > score(2 * 60_000)); - assert!(score(2 * 60_000) > score(60_000)); - assert_eq!(score(1_000 * 60_000), rule.weight * rule.heaviness_cap); - } - - #[test] - fn cpu_rule_ignores_light_builds() { - let rule = CpuAffinityRule::default(); - let j = job(false, cpu_history(1_000)); - assert_eq!(rule.score(&ctx(&j), &cores(50_000), &fleet(10_000.0)), 0.0); - } - - #[test] - fn cpu_rule_needs_history_and_a_fleet_reference() { - let rule = CpuAffinityRule::default(); - let unmeasured = job( - false, - HistoryPrediction { - avg_cpu_time_ms: 10 * 60_000, - samples: 0, - ..Default::default() - }, - ); - let heavy = job(false, cpu_history(10 * 60_000)); - - assert_eq!( - rule.score(&ctx(&unmeasured), &cores(50_000), &fleet(10_000.0)), - 0.0 - ); - assert_eq!( - rule.score(&ctx(&heavy), &cores(50_000), &InstanceContext::default()), - 0.0 - ); - assert_eq!(rule.score(&ctx(&heavy), &cores(0), &fleet(10_000.0)), 0.0); - } - - #[test] - fn cpu_rule_heavy_threshold_follows_the_instance_average() { - let rule = CpuAffinityRule::default(); - let j = job(false, cpu_history(30_000)); - let mut inst = fleet(10_000.0); - assert_eq!(rule.score(&ctx(&j), &cores(15_000), &inst), 0.0); - - inst.cpu_time_ms.w1h = Some(10_000.0); - assert!(rule.score(&ctx(&j), &cores(15_000), &inst) > 0.0); - } - - #[test] - fn cpu_rule_falls_back_to_build_time_without_cpu_samples() { - let rule = CpuAffinityRule::default(); - let j = job( - false, - HistoryPrediction { - build_time_ms: 10 * 60_000, - samples: 5, - ..Default::default() - }, - ); - assert!(rule.score(&ctx(&j), &cores(15_000), &fleet(10_000.0)) > 0.0); - } -} diff --git a/backend/gradient-pool/src/score/rules/builtin.rs b/backend/gradient-pool/src/score/rules/builtin.rs index c8584707d..7f9c71027 100644 --- a/backend/gradient-pool/src/score/rules/builtin.rs +++ b/backend/gradient-pool/src/score/rules/builtin.rs @@ -7,226 +7,6 @@ use crate::score::context::{InstanceContext, JobKindContext}; use crate::score::rule::{JobContext, ScoreRule, WorkerContext}; -#[derive(Debug)] -pub struct MissingPathsRule { - pub cap: f64, - pub k: f64, - pub fallback_avg: f64, -} - -impl Default for MissingPathsRule { - fn default() -> Self { - Self { - cap: crate::score::weights::MISSING_PATHS_CAP, - k: crate::score::weights::MISSING_PATHS_BASELINE_K, - fallback_avg: crate::score::weights::MISSING_PATHS_FALLBACK_AVG, - } - } -} - -impl ScoreRule for MissingPathsRule { - fn name(&self) -> &'static str { - "MissingPathsRule" - } - - fn score( - &self, - job: &JobContext<'_>, - _worker: &WorkerContext<'_>, - instance: &InstanceContext, - ) -> f64 { - match job.missing_count { - None => 0.0, - Some(n) => { - let base = self.k * instance.missing_paths.w1h_or(self.fallback_avg); - if base <= 0.0 { - // A measured zero fleet average is giving the full bonus only to a fully warm - // worker. - return if n == 0 { self.cap } else { 0.0 }; - } - - self.cap * (1.0 - (n as f64 / base).clamp(0.0, 1.0)) - } - } - } - - fn description(&self) -> &'static str { - "Rewards jobs whose dependencies are mostly already on the worker, so fewer store paths must be substituted before the build can start." - } -} - -#[derive(Debug)] -pub struct MissingNarSizeRule { - pub cap: f64, - pub k: f64, -} - -impl Default for MissingNarSizeRule { - fn default() -> Self { - Self { - cap: crate::score::weights::MISSING_NAR_SIZE_CAP, - k: crate::score::weights::MISSING_NAR_SIZE_BASELINE_K, - } - } -} - -impl ScoreRule for MissingNarSizeRule { - fn name(&self) -> &'static str { - "MissingNarSizeRule" - } - - fn score( - &self, - job: &JobContext<'_>, - _worker: &WorkerContext<'_>, - instance: &InstanceContext, - ) -> f64 { - match job.missing_nar_size { - None => 0.0, - Some(0) => self.cap, - Some(b) => { - let mb = b as f64 / 1_048_576.0; - let baseline = self.k * instance.nar_size_mb.w1h_or(1024.0); - - self.cap * (1.0 - (mb / baseline).clamp(0.0, 1.0)) - } - } - } - - fn description(&self) -> &'static str { - "Rewards jobs with little or no data left to download, favouring small substitution transfers over large ones." - } -} - -#[derive(Debug)] -pub struct RealisedOutputsRule { - pub bonus: f64, -} - -impl Default for RealisedOutputsRule { - fn default() -> Self { - Self { - bonus: crate::score::weights::REALISED_OUTPUTS_BONUS, - } - } -} - -impl ScoreRule for RealisedOutputsRule { - fn name(&self) -> &'static str { - "RealisedOutputsRule" - } - - fn score( - &self, - job: &JobContext<'_>, - _worker: &WorkerContext<'_>, - _instance: &InstanceContext, - ) -> f64 { - if job.outputs_present { self.bonus } else { 0.0 } - } - - fn description(&self) -> &'static str { - "Strongly prefers a worker that already holds every output of the job, since it only uploads them instead of building." - } -} - -#[derive(Debug)] -pub struct BuiltinDeprioritizeRule { - pub bonus: f64, - pub archless_bonus: f64, -} - -impl Default for BuiltinDeprioritizeRule { - fn default() -> Self { - Self { - bonus: crate::score::weights::REAL_BUILD_BONUS, - archless_bonus: crate::score::weights::ARCHLESS_BUILTIN_BONUS, - } - } -} - -impl ScoreRule for BuiltinDeprioritizeRule { - fn name(&self) -> &'static str { - "BuiltinDeprioritizeRule" - } - - fn score( - &self, - job: &JobContext<'_>, - worker: &WorkerContext<'_>, - _instance: &InstanceContext, - ) -> f64 { - let Some(b) = job.job.build() else { - return 0.0; - }; - - // An arch-less worker can only run builtins and fetches. A strong lift is keeping it from - // sitting idle. - if b.architecture == gradient_types::BUILTIN_ARCH { - return if worker.architectures.is_empty() { - self.archless_bonus - } else { - 0.0 - }; - } - - self.bonus - } - - fn description(&self) -> &'static str { - "Rewards real compilation jobs over Nix builtin derivations so builtins yield worker slots, and lifts jobs on architecture-less workers so those workers are not left idle." - } -} - -#[derive(Debug)] -pub struct DependencyCountRule { - pub cap: f64, - pub k: f64, - pub fallback_avg: f64, -} - -impl Default for DependencyCountRule { - fn default() -> Self { - Self { - cap: crate::score::weights::DEPENDENCY_COUNT_CAP, - k: crate::score::weights::DEPENDENCY_COUNT_BASELINE_K, - fallback_avg: crate::score::weights::DEPENDENCY_COUNT_FALLBACK_AVG, - } - } -} - -impl ScoreRule for DependencyCountRule { - fn name(&self) -> &'static str { - "DependencyCountRule" - } - - fn score( - &self, - job: &JobContext<'_>, - _worker: &WorkerContext<'_>, - instance: &InstanceContext, - ) -> f64 { - if job.job.build().is_none() { - return 0.0; - } - - let base = self.k * instance.dependency_cnt.w1h_or(self.fallback_avg); - if base <= 0.0 { - return if job.dependency_count > 0 { - self.cap - } else { - 0.0 - }; - } - - self.cap * (job.dependency_count as f64 / base).clamp(0.0, 1.0) - } - - fn description(&self) -> &'static str { - "Rewards builds with many direct inputs, relative to the instance's recent average input count." - } -} - #[derive(Debug)] pub struct WaitTimeRule { pub gain: f64, @@ -387,390 +167,6 @@ mod tests { } } - #[test] - fn missing_paths_scored_zero_wins_over_unscored() { - let rule = MissingPathsRule::default(); - let job = build_job("x86_64-linux"); - let archs = vec!["x86_64-linux".to_string()]; - let w = worker(&archs, false); - let now = gradient_types::now(); - let inst = crate::score::context::InstanceContext { - missing_paths: crate::score::context::Windowed { - w1h: Some(10.0), - ..Default::default() - }, - ..Default::default() - }; - - let scored = JobContext { - job: &job, - missing_count: Some(0), - missing_nar_size: None, - outputs_present: false, - dependency_count: 0, - queued_at: now, - ready_at: now, - project_work_share: None, - prioritized: false, - build_request: false, - rescore_count: 0, - now, - }; - let unscored = JobContext { - job: &job, - missing_count: None, - missing_nar_size: None, - outputs_present: false, - dependency_count: 0, - queued_at: now, - ready_at: now, - project_work_share: None, - prioritized: false, - build_request: false, - rescore_count: 0, - now, - }; - - assert!((rule.score(&scored, &w, &inst) - 200.0).abs() < 1e-9); - assert_eq!(rule.score(&unscored, &w, &inst), 0.0); - assert!(rule.score(&scored, &w, &inst) > rule.score(&unscored, &w, &inst)); - } - - #[test] - fn missing_paths_fewer_missing_wins() { - let rule = MissingPathsRule::default(); - let job = build_job("x86_64-linux"); - let archs = vec!["x86_64-linux".to_string()]; - let w = worker(&archs, false); - let now = gradient_types::now(); - let inst = crate::score::context::InstanceContext { - missing_paths: crate::score::context::Windowed { - w1h: Some(10.0), - ..Default::default() - }, - ..Default::default() - }; - - let c1 = JobContext { - job: &job, - missing_count: Some(2), - missing_nar_size: None, - outputs_present: false, - dependency_count: 0, - queued_at: now, - ready_at: now, - project_work_share: None, - prioritized: false, - build_request: false, - rescore_count: 0, - now, - }; - let c2 = JobContext { - job: &job, - missing_count: Some(10), - missing_nar_size: None, - outputs_present: false, - dependency_count: 0, - queued_at: now, - ready_at: now, - project_work_share: None, - prioritized: false, - build_request: false, - rescore_count: 0, - now, - }; - - assert!(rule.score(&c1, &w, &inst) > rule.score(&c2, &w, &inst)); - assert!(rule.score(&c1, &w, &inst) >= 0.0); - assert!(rule.score(&c2, &w, &inst) >= 0.0); - } - - #[test] - fn missing_nar_size_bounded_bonus() { - let rule = MissingNarSizeRule::default(); - let job = build_job("x86_64-linux"); - let archs = vec!["x86_64-linux".to_string()]; - let w = worker(&archs, false); - let now = gradient_types::now(); - let inst = crate::score::context::InstanceContext { - nar_size_mb: crate::score::context::Windowed { - w1h: Some(100.0), - ..Default::default() - }, - ..Default::default() - }; - - let c_none = JobContext { - job: &job, - missing_count: None, - missing_nar_size: None, - outputs_present: false, - dependency_count: 0, - queued_at: now, - ready_at: now, - project_work_share: None, - prioritized: false, - build_request: false, - rescore_count: 0, - now, - }; - let c_zero = JobContext { - job: &job, - missing_count: None, - missing_nar_size: Some(0), - outputs_present: false, - dependency_count: 0, - queued_at: now, - ready_at: now, - project_work_share: None, - prioritized: false, - build_request: false, - rescore_count: 0, - now, - }; - let c_huge = JobContext { - job: &job, - missing_count: None, - missing_nar_size: Some(100_000_000_000), - outputs_present: false, - dependency_count: 0, - queued_at: now, - ready_at: now, - project_work_share: None, - prioritized: false, - build_request: false, - rescore_count: 0, - now, - }; - - assert_eq!(rule.score(&c_none, &w, &inst), 0.0); - assert!((rule.score(&c_zero, &w, &inst) - 500.0).abs() < 1e-9); - assert!(rule.score(&c_huge, &w, &inst) >= 0.0); - assert!(rule.score(&c_zero, &w, &inst) > rule.score(&c_huge, &w, &inst)); - } - - #[test] - fn builtin_rule_rewards_real_zeroes_builtin_and_lifts_archless_worker() { - let rule = BuiltinDeprioritizeRule::default(); - let real = build_job("x86_64-linux"); - let builtin = build_job("builtin"); - let archs = vec!["x86_64-linux".to_string()]; - let w = worker(&archs, false); - let no_archs: Vec = vec![]; - let archless = worker(&no_archs, false); - let inst = InstanceContext::default(); - let now = gradient_types::now(); - - let c_real = JobContext { - job: &real, - missing_count: None, - missing_nar_size: None, - outputs_present: false, - dependency_count: 0, - queued_at: now, - ready_at: now, - project_work_share: None, - prioritized: false, - build_request: false, - rescore_count: 0, - now, - }; - let c_builtin = JobContext { - job: &builtin, - missing_count: None, - missing_nar_size: None, - outputs_present: false, - dependency_count: 0, - queued_at: now, - ready_at: now, - project_work_share: None, - prioritized: false, - build_request: false, - rescore_count: 0, - now, - }; - - assert_eq!( - rule.score(&c_real, &w, &inst), - 50.0, - "a real build gets the default bonus" - ); - assert_eq!( - rule.score(&c_real, &archless, &inst), - 50.0, - "the arch-less lift is builtin-only" - ); - assert_eq!( - rule.score(&c_builtin, &w, &inst), - 0.0, - "a builtin yields its slot to real work" - ); - assert_eq!( - rule.score(&c_builtin, &archless, &inst), - 100.0, - "a builtin on an arch-less worker is lifted so it isn't starved" - ); - } - - #[test] - fn builtin_rule_ignores_evals() { - let rule = BuiltinDeprioritizeRule::default(); - let eval = eval_job(false); - let no_archs: Vec = vec![]; - let archless = worker(&no_archs, false); - let now = gradient_types::now(); - - let c_eval = JobContext { - job: &eval, - missing_count: None, - missing_nar_size: None, - outputs_present: false, - dependency_count: 0, - queued_at: now, - ready_at: now, - project_work_share: None, - prioritized: false, - build_request: false, - rescore_count: 0, - now, - }; - - assert_eq!( - rule.score(&c_eval, &archless, &InstanceContext::default()), - 0.0 - ); - } - - #[test] - fn dependency_count_more_deps_wins() { - let rule = DependencyCountRule::default(); - let job = build_job("x86_64-linux"); - let archs = vec!["x86_64-linux".to_string()]; - let w = worker(&archs, false); - let now = gradient_types::now(); - let inst = crate::score::context::InstanceContext { - dependency_cnt: crate::score::context::Windowed { - w1h: Some(10.0), - ..Default::default() - }, - ..Default::default() - }; - - let c_few = JobContext { - job: &job, - missing_count: None, - missing_nar_size: None, - outputs_present: false, - dependency_count: 1, - queued_at: now, - ready_at: now, - project_work_share: None, - prioritized: false, - build_request: false, - rescore_count: 0, - now, - }; - let c_many = JobContext { - job: &job, - missing_count: None, - missing_nar_size: None, - outputs_present: false, - dependency_count: 15, - queued_at: now, - ready_at: now, - project_work_share: None, - prioritized: false, - build_request: false, - rescore_count: 0, - now, - }; - - assert!(rule.score(&c_many, &w, &inst) > rule.score(&c_few, &w, &inst)); - assert!(rule.score(&c_few, &w, &inst) > 0.0); - } - - #[test] - fn dependency_count_zero_deps_zero_score() { - let rule = DependencyCountRule::default(); - let build = build_job("x86_64-linux"); - let eval = eval_job(false); - let archs = vec!["x86_64-linux".to_string()]; - let w = worker(&archs, false); - let now = gradient_types::now(); - let inst = crate::score::context::InstanceContext { - dependency_cnt: crate::score::context::Windowed { - w1h: Some(10.0), - ..Default::default() - }, - ..Default::default() - }; - - let ctx_zero = JobContext { - job: &build, - missing_count: None, - missing_nar_size: None, - outputs_present: false, - dependency_count: 0, - queued_at: now, - ready_at: now, - project_work_share: None, - prioritized: false, - build_request: false, - rescore_count: 0, - now, - }; - let ctx_eval = JobContext { - job: &eval, - missing_count: None, - missing_nar_size: None, - outputs_present: false, - dependency_count: 5, - queued_at: now, - ready_at: now, - project_work_share: None, - prioritized: false, - build_request: false, - rescore_count: 0, - now, - }; - - assert_eq!(rule.score(&ctx_zero, &w, &inst), 0.0); - assert_eq!(rule.score(&ctx_eval, &w, &inst), 0.0); - } - - #[test] - fn dependency_count_capped_at_50() { - let rule = DependencyCountRule::default(); - let job = build_job("x86_64-linux"); - let archs = vec!["x86_64-linux".to_string()]; - let w = worker(&archs, false); - let now = gradient_types::now(); - let inst = crate::score::context::InstanceContext { - dependency_cnt: crate::score::context::Windowed { - w1h: Some(10.0), - ..Default::default() - }, - ..Default::default() - }; - - let ctx_huge = JobContext { - job: &job, - missing_count: None, - missing_nar_size: None, - outputs_present: false, - dependency_count: 100_000, - queued_at: now, - ready_at: now, - project_work_share: None, - prioritized: false, - build_request: false, - rescore_count: 0, - now, - }; - - assert!(rule.score(&ctx_huge, &w, &inst) <= 50.0); - } - #[test] fn wait_time_longer_wait_scores_higher_but_capped() { let rule = WaitTimeRule::default(); @@ -1023,31 +419,4 @@ mod tests { "build job not penalized" ); } - - #[test] - fn realised_outputs_reward_only_a_worker_holding_them() { - let rule = RealisedOutputsRule::default(); - let job = build_job("x86_64-linux"); - let archs = vec!["x86_64-linux".to_string()]; - let w = worker(&archs, false); - let now = gradient_types::now(); - let ctx = |outputs_present| JobContext { - job: &job, - missing_count: Some(0), - missing_nar_size: Some(0), - outputs_present, - dependency_count: 0, - queued_at: now, - ready_at: now, - project_work_share: None, - prioritized: false, - build_request: false, - rescore_count: 0, - now, - }; - let inst = crate::score::context::InstanceContext::default(); - - assert_eq!(rule.score(&ctx(true), &w, &inst), rule.bonus); - assert_eq!(rule.score(&ctx(false), &w, &inst), 0.0); - } } diff --git a/backend/gradient-pool/src/score/rules/estimated_time.rs b/backend/gradient-pool/src/score/rules/estimated_time.rs new file mode 100644 index 000000000..0e1ee5445 --- /dev/null +++ b/backend/gradient-pool/src/score/rules/estimated_time.rs @@ -0,0 +1,554 @@ +/* + * SPDX-FileCopyrightText: 2026 Wavelens GmbH + * + * SPDX-License-Identifier: AGPL-3.0-only + */ + +use crate::score::context::{HistoryPrediction, InstanceContext, WorkerMetricsView}; +use crate::score::rule::{JobContext, ScoreRule, WorkerContext}; +use crate::score::weights; + +const BYTES_PER_MIB: f64 = 1_048_576.0; +const BITS_PER_BYTE: f64 = 8.0; +const BITS_PER_MEGABIT: f64 = 1_000_000.0; + +pub fn contention_factor(other_builds: u32) -> f64 { + 1.0 + weights::BUILD_CONTENTION_PER_BUILD * f64::from(other_builds) +} + +pub struct Transfer { + pub nar_bytes: f64, + pub worker_mbps: Option, + pub storage_mbps: f64, + pub stored_per_nar_byte: f64, + pub in_flight: u32, +} + +pub fn transfer_secs(t: Transfer) -> f64 { + let storage_share = + t.storage_mbps / t.stored_per_nar_byte / f64::from(t.in_flight.saturating_add(1)); + let mbps = match t.worker_mbps.filter(|w| *w > 0.0) { + Some(worker) => worker.min(storage_share), + None => storage_share, + }; + if t.nar_bytes <= 0.0 || mbps <= 0.0 { + return 0.0; + } + + t.nar_bytes * BITS_PER_BYTE / BITS_PER_MEGABIT / mbps +} + +fn stored_per_nar_byte(instance: &InstanceContext) -> f64 { + instance + .compression_ratio + .filter(|r| *r > 0.0) + .unwrap_or(weights::STORED_PER_NAR_BYTE_FALLBACK) +} + +fn speed(own: Option, fleet_mean: Option) -> Option { + own.map(f64::from).or(fleet_mean) +} + +pub fn download_secs( + job: &JobContext<'_>, + worker: Option<&WorkerMetricsView>, + instance: &InstanceContext, +) -> f64 { + let nar_bytes = job + .missing_nar_size + .map(|b| b as f64) + .or_else(|| instance.nar_size_mb.w1h.map(|mb| mb * BYTES_PER_MIB)) + .unwrap_or(0.0); + + transfer_secs(Transfer { + nar_bytes, + worker_mbps: speed( + worker.and_then(|m| m.download_speed_mbps), + instance.download_speed_mean_mbps, + ), + storage_mbps: instance + .storage_read_mbps + .unwrap_or(weights::STORAGE_READ_FALLBACK_MBPS), + stored_per_nar_byte: stored_per_nar_byte(instance), + in_flight: instance.downloads_in_flight, + }) +} + +pub fn path_secs(job: &JobContext<'_>, instance: &InstanceContext) -> f64 { + let paths = job + .missing_count + .map(f64::from) + .or(instance.missing_paths.w1h) + .unwrap_or(0.0); + paths + * instance + .per_path_secs + .unwrap_or(weights::PER_PATH_FALLBACK_SECS) +} + +pub fn upload_secs( + history: &HistoryPrediction, + worker: Option<&WorkerMetricsView>, + instance: &InstanceContext, +) -> f64 { + transfer_secs(Transfer { + nar_bytes: history.output_nar_size.unwrap_or(0) as f64, + worker_mbps: speed( + worker.and_then(|m| m.upload_speed_mbps), + instance.upload_speed_mean_mbps, + ), + storage_mbps: instance + .storage_write_mbps + .unwrap_or(weights::STORAGE_WRITE_FALLBACK_MBPS), + stored_per_nar_byte: stored_per_nar_byte(instance), + in_flight: instance.uploads_in_flight, + }) +} + +fn core_ratio(built_on: Option, runs_on: Option) -> f64 { + match (built_on, runs_on) { + (Some(built), Some(runs)) if built > 0.0 && runs > 0.0 => { + (built / runs).clamp(weights::CORE_SCORE_RATIO_MIN, weights::CORE_SCORE_RATIO_MAX) + } + _ => 1.0, + } +} + +fn run_secs( + history: &HistoryPrediction, + worker: Option<&WorkerMetricsView>, + instance: &InstanceContext, + fallback_ms: Option, +) -> f64 { + let fleet = instance.cpu_core_score_mean; + let (uncontended_ms, built_on) = match history.uncontended_build_time_ms { + Some(ms) => (ms as f64, history.build_core_score.map(f64::from).or(fleet)), + None => match fallback_ms { + Some(ms) => (ms, fleet), + None => return 0.0, + }, + }; + let runs_on = worker + .map(|m| m.cpu_core_score) + .filter(|score| *score > 0) + .map(f64::from) + .or(fleet); + let running = worker.map_or(0, |m| m.running_builds); + + uncontended_ms / 1000.0 * core_ratio(built_on, runs_on) * contention_factor(running) +} + +pub fn build_secs( + history: &HistoryPrediction, + worker: Option<&WorkerMetricsView>, + instance: &InstanceContext, +) -> f64 { + let window = instance.build_time_ms.w1h.or(instance.build_time_ms.w24h); + run_secs(history, worker, instance, window) +} + +pub fn eval_secs( + history: &HistoryPrediction, + worker: Option<&WorkerMetricsView>, + instance: &InstanceContext, +) -> f64 { + run_secs(history, worker, instance, None) +} + +pub fn oom_retry_secs( + history: &HistoryPrediction, + worker: Option<&WorkerMetricsView>, + run_secs: f64, +) -> f64 { + let overshoot = match ( + history.predicted_peak_ram_mb, + worker.and_then(|m| m.ram_free_mb), + ) { + (Some(peak), Some(free)) if free > 0 && peak > free => { + ((peak - free) as f64 / free as f64).min(1.0) + } + _ => 0.0, + }; + + (f64::from(history.oom_rate) + overshoot).min(1.0) * run_secs +} + +pub fn estimated_secs( + job: &JobContext<'_>, + worker: &WorkerContext<'_>, + instance: &InstanceContext, +) -> f64 { + let metrics = worker.metrics.as_ref(); + if job.job.build().is_none() { + let history = job.job.history(); + let run = eval_secs(&history, metrics, instance); + return run + oom_retry_secs(&history, metrics, run); + } + + if job.outputs_present { + return upload_secs(&job.job.history(), metrics, instance); + } + + let history = job.build_history(); + let run = build_secs(&history, metrics, instance); + download_secs(job, metrics, instance) + + path_secs(job, instance) + + run + + oom_retry_secs(&history, metrics, run) + + upload_secs(&history, metrics, instance) +} + +#[derive(Debug)] +pub struct EstimatedTimeRule { + pub points_per_sec: f64, + pub cap_secs: f64, +} + +impl Default for EstimatedTimeRule { + fn default() -> Self { + Self { + points_per_sec: weights::ESTIMATED_TIME_POINTS_PER_SEC, + cap_secs: weights::ESTIMATED_TIME_CAP_SECS, + } + } +} + +impl ScoreRule for EstimatedTimeRule { + fn name(&self) -> &'static str { + "EstimatedTimeRule" + } + + fn score( + &self, + job: &JobContext<'_>, + worker: &WorkerContext<'_>, + instance: &InstanceContext, + ) -> f64 { + let estimate = estimated_secs(job, worker, instance).min(self.cap_secs); + self.points_per_sec * (self.cap_secs - estimate) + } + + fn description(&self) -> &'static str { + "Ranks builds by the seconds they are expected to take on this worker: downloading the missing inputs, building, and uploading the outputs. Each second costs one point, up to a cap." + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::score::context::{ScoredJob, Windowed}; + use gradient_types::ids::ProjectId; + + const GIGABYTE: f64 = 1_000_000_000.0; + + fn transfer(worker_mbps: Option, in_flight: u32, stored_per_nar_byte: f64) -> f64 { + transfer_secs(Transfer { + nar_bytes: GIGABYTE, + worker_mbps, + storage_mbps: 1_200.0, + stored_per_nar_byte, + in_flight, + }) + } + + #[test] + fn a_transfer_runs_at_the_slower_of_the_worker_and_its_storage_share() { + assert_eq!(transfer(Some(100.0), 0, 1.0), 80.0); + assert_eq!(transfer(Some(100.0), 23, 1.0), 160.0); + assert_eq!(transfer(None, 23, 1.0), 160.0); + } + + #[test] + fn compressed_storage_moves_more_nar_bytes_per_second() { + assert_eq!(transfer(None, 23, 0.5), 80.0); + } + + #[test] + fn nothing_to_move_takes_no_time() { + let none = Transfer { + nar_bytes: 0.0, + worker_mbps: Some(1.0), + storage_mbps: 1.0, + stored_per_nar_byte: 1.0, + in_flight: 0, + }; + assert_eq!(transfer_secs(none), 0.0); + } + + fn built(ms: u64, core_score: Option) -> HistoryPrediction { + HistoryPrediction { + build_time_ms: Some(ms), + uncontended_build_time_ms: Some(ms), + build_core_score: core_score, + samples: 3, + ..Default::default() + } + } + + fn on(cpu_core_score: u32, running_builds: u32) -> WorkerMetricsView { + WorkerMetricsView { + cpu_core_score, + running_builds, + ..Default::default() + } + } + + #[test] + fn a_build_takes_longer_on_a_slower_core_and_beside_running_builds() { + let history = built(100_000, Some(3_000)); + let inst = InstanceContext::default(); + + assert_eq!(build_secs(&history, Some(&on(3_000, 0)), &inst), 100.0); + assert_eq!(build_secs(&history, Some(&on(1_500, 0)), &inst), 200.0); + assert!((build_secs(&history, Some(&on(1_500, 5)), &inst) - 240.0).abs() < 1e-9); + } + + #[test] + fn the_core_score_ratio_is_bounded() { + let history = built(100_000, Some(3_000)); + let inst = InstanceContext::default(); + + assert_eq!(build_secs(&history, Some(&on(1, 0)), &inst), 200.0); + assert_eq!(build_secs(&history, Some(&on(1_000_000, 0)), &inst), 50.0); + } + + #[test] + fn an_unknown_core_score_is_the_fleet_mean() { + let inst = InstanceContext { + cpu_core_score_mean: Some(3_000.0), + ..Default::default() + }; + + assert_eq!( + build_secs(&built(100_000, None), Some(&on(1_500, 0)), &inst), + 200.0 + ); + assert_eq!( + build_secs(&built(100_000, Some(1_500)), Some(&on(0, 0)), &inst), + 50.0 + ); + } + + #[test] + fn without_history_the_instance_build_time_is_the_estimate() { + let inst = InstanceContext { + build_time_ms: Windowed { + w24h: Some(60_000.0), + ..Default::default() + }, + ..Default::default() + }; + + assert_eq!( + build_secs(&HistoryPrediction::default(), Some(&on(2_000, 0)), &inst), + 60.0 + ); + assert_eq!( + build_secs( + &HistoryPrediction::default(), + None, + &InstanceContext::default() + ), + 0.0 + ); + } + + fn build_job(history: HistoryPrediction) -> ScoredJob<'static> { + ScoredJob::new_build( + "j", + ProjectId::now_v7(), + "x86_64-linux", + false, + false, + None, + None, + history, + ) + } + + fn ctx<'a>( + job: &'a ScoredJob<'a>, + missing_nar_size: Option, + outputs_present: bool, + ) -> JobContext<'a> { + JobContext { + job, + missing_count: Some(0), + missing_nar_size, + outputs_present, + dependency_count: 0, + queued_at: gradient_types::now(), + ready_at: gradient_types::now(), + project_work_share: None, + prioritized: false, + build_request: false, + rescore_count: 0, + now: gradient_types::now(), + } + } + + fn worker(metrics: WorkerMetricsView) -> WorkerContext<'static> { + WorkerContext { + architectures: &[], + system_features: &[], + fetch: false, + metrics: Some(metrics), + } + } + + #[test] + fn every_missing_path_costs_the_learned_overhead() { + let job = build_job(HistoryPrediction::default()); + let inst = InstanceContext { + per_path_secs: Some(0.5), + missing_paths: Windowed { + w1h: Some(8.0), + ..Default::default() + }, + ..Default::default() + }; + let scored = JobContext { + missing_count: Some(6), + ..ctx(&job, Some(0), false) + }; + let unscored = JobContext { + missing_count: None, + ..ctx(&job, None, false) + }; + + assert_eq!(path_secs(&scored, &inst), 3.0); + assert_eq!(path_secs(&unscored, &inst), 4.0); + } + + #[test] + fn the_score_is_the_capped_time_left_unspent() { + let rule = EstimatedTimeRule::default(); + let inst = InstanceContext::default(); + let short = build_job(built(100_000, None)); + let endless = build_job(built(100_000_000, None)); + let w = worker(on(1_000, 0)); + + assert_eq!( + rule.score(&ctx(&short, Some(0), false), &w, &inst), + rule.points_per_sec * (rule.cap_secs - 100.0) + ); + assert_eq!(rule.score(&ctx(&endless, Some(0), false), &w, &inst), 0.0); + assert_eq!( + rule.score(&ctx(&endless, Some(0), true), &w, &inst), + rule.points_per_sec * rule.cap_secs, + "a worker holding the outputs builds nothing" + ); + } + + #[test] + fn a_build_that_may_run_out_of_memory_expects_to_lose_its_time_again() { + let history = HistoryPrediction { + predicted_peak_ram_mb: Some(3_000), + oom_rate: 0.25, + ..built(100_000, None) + }; + let with_free = |ram_free_mb| WorkerMetricsView { + ram_free_mb: Some(ram_free_mb), + ..Default::default() + }; + + assert_eq!( + oom_retry_secs(&history, Some(&with_free(8_000)), 100.0), + 25.0 + ); + assert_eq!( + oom_retry_secs(&history, Some(&with_free(2_000)), 100.0), + 75.0 + ); + assert_eq!( + oom_retry_secs(&history, Some(&with_free(1_000)), 100.0), + 100.0 + ); + } + + #[test] + fn a_worker_holding_the_outputs_only_uploads_them() { + let job = build_job(HistoryPrediction { + output_nar_size: Some(GIGABYTE as u64), + ..built(100_000, None) + }); + let w = worker(WorkerMetricsView { + upload_speed_mbps: Some(400.0), + ..Default::default() + }); + let inst = InstanceContext { + storage_write_mbps: Some(1_000_000.0), + ..Default::default() + }; + + assert!((estimated_secs(&ctx(&job, Some(0), true), &w, &inst) - 20.0).abs() < 1e-9); + } + + #[test] + fn the_estimate_sums_download_build_and_upload() { + let history = HistoryPrediction { + output_nar_size: Some(GIGABYTE as u64), + ..built(30_000, None) + }; + let job = build_job(history); + let inst = InstanceContext { + storage_read_mbps: Some(1_000_000.0), + storage_write_mbps: Some(1_000_000.0), + ..Default::default() + }; + let w = worker(WorkerMetricsView { + download_speed_mbps: Some(800.0), + upload_speed_mbps: Some(400.0), + ..Default::default() + }); + + let estimate = estimated_secs(&ctx(&job, Some(GIGABYTE as u64), false), &w, &inst); + assert!((estimate - (10.0 + 30.0 + 20.0)).abs() < 1e-9, "{estimate}"); + } + + #[test] + fn a_large_download_goes_to_the_faster_downloading_worker() { + let rule = EstimatedTimeRule::default(); + let job = build_job(HistoryPrediction::default()); + let c = ctx(&job, Some(4 * GIGABYTE as u64), false); + let downloading = |download_speed_mbps| { + worker(WorkerMetricsView { + download_speed_mbps: Some(download_speed_mbps), + ..Default::default() + }) + }; + let inst = InstanceContext::default(); + + assert!( + rule.score(&c, &downloading(1_000.0), &inst) + > rule.score(&c, &downloading(100.0), &inst) + ); + } + + #[test] + fn an_evaluation_is_estimated_by_its_run_time_alone() { + let rule = EstimatedTimeRule::default(); + let eval = ScoredJob::new_eval("e", ProjectId::now_v7(), true, built(100_000, None)); + let unknown = + ScoredJob::new_eval("e", ProjectId::now_v7(), true, HistoryPrediction::default()); + let inst = InstanceContext { + per_path_secs: Some(1.0), + build_time_ms: Windowed { + w1h: Some(999_000.0), + ..Default::default() + }, + ..Default::default() + }; + let w = worker(on(1_000, 0)); + + assert_eq!( + rule.score(&ctx(&eval, Some(10), false), &w, &inst), + rule.points_per_sec * (rule.cap_secs - 100.0) + ); + assert_eq!( + rule.score(&ctx(&unknown, None, false), &w, &inst), + rule.points_per_sec * rule.cap_secs, + "an evaluation never takes the build time of the instance" + ); + } +} diff --git a/backend/gradient-pool/src/score/rules/mod.rs b/backend/gradient-pool/src/score/rules/mod.rs index a95e63275..90391b508 100644 --- a/backend/gradient-pool/src/score/rules/mod.rs +++ b/backend/gradient-pool/src/score/rules/mod.rs @@ -4,19 +4,16 @@ * SPDX-License-Identifier: AGPL-3.0-only */ -pub mod affinity; pub mod builtin; +pub mod estimated_time; pub mod fair_share; -pub mod prefer_local; pub mod qos; pub mod resource; +pub mod transfer_limit; -pub use affinity::{CpuAffinityRule, DiskAffinityRule, NetworkAffinityRule}; -pub use builtin::{ - BuiltinDeprioritizeRule, DependencyCountRule, MissingNarSizeRule, MissingPathsRule, - RealisedOutputsRule, RescoreWaitRule, ReserveFetchWorkersRule, WaitTimeRule, -}; +pub use builtin::{RescoreWaitRule, ReserveFetchWorkersRule, WaitTimeRule}; +pub use estimated_time::EstimatedTimeRule; pub use fair_share::FairShareRule; -pub use prefer_local::PreferLocalBuildRule; pub use qos::QosRule; -pub use resource::{ResourceFitRule, ResourceSaturationRule}; +pub use resource::ResourceSaturationRule; +pub use transfer_limit::TransferLimitRule; diff --git a/backend/gradient-pool/src/score/rules/prefer_local.rs b/backend/gradient-pool/src/score/rules/prefer_local.rs deleted file mode 100644 index f9432f43d..000000000 --- a/backend/gradient-pool/src/score/rules/prefer_local.rs +++ /dev/null @@ -1,166 +0,0 @@ -/* - * SPDX-FileCopyrightText: 2026 Wavelens GmbH - * - * SPDX-License-Identifier: AGPL-3.0-only - */ - -use crate::score::context::InstanceContext; -use crate::score::rule::{JobContext, ScoreRule, WorkerContext}; - -#[derive(Debug)] -pub struct PreferLocalBuildRule { - pub local_bonus: f64, - pub miss_penalty: f64, -} - -impl Default for PreferLocalBuildRule { - fn default() -> Self { - Self { - local_bonus: crate::score::weights::PREFER_LOCAL_BONUS, - miss_penalty: crate::score::weights::PREFER_LOCAL_MISS_PENALTY, - } - } -} - -impl ScoreRule for PreferLocalBuildRule { - fn name(&self) -> &'static str { - "PreferLocalBuildRule" - } - - fn score( - &self, - job: &JobContext<'_>, - _worker: &WorkerContext<'_>, - instance: &InstanceContext, - ) -> f64 { - let Some(b) = job.job.build() else { return 0.0 }; - if !b.prefer_local_build { - return 0.0; - } - - let knee = 2.0 * instance.missing_paths.w1h.unwrap_or(0.0); - let slope = if knee > 0.0 { - self.local_bonus / knee - } else { - self.miss_penalty - }; - match job.missing_count { - Some(0) => self.local_bonus, - Some(n) => (self.local_bonus - n as f64 * slope).max(0.0), - None => 0.0, - } - } - - fn description(&self) -> &'static str { - "Rewards keeping a `preferLocalBuild` derivation on a worker that already has its inputs, since shipping it elsewhere rarely pays off." - } -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::score::context::{HistoryPrediction, ScoredJob}; - use gradient_types::ids::ProjectId; - - fn job(prefer_local_build: bool) -> ScoredJob<'static> { - ScoredJob::new_build( - "test", - ProjectId::now_v7(), - "x86_64-linux", - prefer_local_build, - false, - None, - None, - HistoryPrediction::default(), - ) - } - - fn ctx<'a>(job: &'a ScoredJob<'a>, missing_count: Option) -> JobContext<'a> { - JobContext { - job, - missing_count, - missing_nar_size: None, - outputs_present: false, - dependency_count: 0, - queued_at: gradient_types::now(), - ready_at: gradient_types::now(), - project_work_share: None, - prioritized: false, - build_request: false, - rescore_count: 0, - now: gradient_types::now(), - } - } - - fn worker() -> WorkerContext<'static> { - WorkerContext { - architectures: &[], - system_features: &[], - fetch: false, - metrics: None, - } - } - - #[test] - fn local_worker_with_full_cache_gets_full_bonus() { - let rule = PreferLocalBuildRule::default(); - let j = job(true); - assert_eq!( - rule.score(&ctx(&j, Some(0)), &worker(), &InstanceContext::default()), - rule.local_bonus - ); - } - - #[test] - fn more_missing_paths_lowers_bonus_floored_at_zero() { - let rule = PreferLocalBuildRule::default(); - let j = job(true); - let few = rule.score(&ctx(&j, Some(2)), &worker(), &InstanceContext::default()); - let many = rule.score(&ctx(&j, Some(100)), &worker(), &InstanceContext::default()); - assert!(few < rule.local_bonus); - assert!(many < few); - assert_eq!(many, 0.0, "deeply-missing closure floors at 0"); - } - - #[test] - fn unknown_missing_count_is_zero() { - let rule = PreferLocalBuildRule::default(); - let j = job(true); - assert_eq!( - rule.score(&ctx(&j, None), &worker(), &InstanceContext::default()), - 0.0 - ); - } - - #[test] - fn not_prefer_local_is_zero_regardless_of_missing_count() { - let rule = PreferLocalBuildRule::default(); - let j = job(false); - assert_eq!( - rule.score(&ctx(&j, Some(0)), &worker(), &InstanceContext::default()), - 0.0 - ); - assert_eq!( - rule.score(&ctx(&j, Some(5)), &worker(), &InstanceContext::default()), - 0.0 - ); - assert_eq!( - rule.score(&ctx(&j, None), &worker(), &InstanceContext::default()), - 0.0 - ); - } - - #[test] - fn knee_tracks_instance_missing_paths() { - let rule = PreferLocalBuildRule::default(); - let j = job(true); - let mut inst = InstanceContext::default(); - inst.missing_paths.w1h = Some(5.0); - - assert_eq!(rule.score(&ctx(&j, Some(10)), &worker(), &inst), 0.0); - - let at_half = rule.score(&ctx(&j, Some(5)), &worker(), &inst); - assert!(at_half > 0.0); - assert!(at_half < rule.local_bonus); - } -} diff --git a/backend/gradient-pool/src/score/rules/resource.rs b/backend/gradient-pool/src/score/rules/resource.rs index 18772c4e7..4c810338f 100644 --- a/backend/gradient-pool/src/score/rules/resource.rs +++ b/backend/gradient-pool/src/score/rules/resource.rs @@ -7,59 +7,6 @@ use crate::score::context::InstanceContext; use crate::score::rule::{JobContext, ScoreRule, WorkerContext}; -#[derive(Debug)] -pub struct ResourceFitRule { - pub ram_overshoot_penalty: f64, - pub max_overshoot: f64, -} - -impl Default for ResourceFitRule { - fn default() -> Self { - Self { - ram_overshoot_penalty: crate::score::weights::RESOURCE_FIT_RAM_PENALTY, - max_overshoot: crate::score::weights::RESOURCE_FIT_MAX_OVERSHOOT, - } - } -} - -impl ScoreRule for ResourceFitRule { - fn name(&self) -> &'static str { - "ResourceFitRule" - } - - fn score( - &self, - job: &JobContext<'_>, - worker: &WorkerContext<'_>, - instance: &InstanceContext, - ) -> f64 { - let Some(m) = worker.metrics else { return 0.0 }; - let h = job.build_history(); - if h.samples == 0 { - return 0.0; - } - - let mut s = 0.0; - if let Some(free) = m.ram_free_mb - && free > 0 - && h.predicted_peak_ram_mb > free - { - let overshoot = - ((h.predicted_peak_ram_mb - free) as f64 / free as f64).min(self.max_overshoot); - s -= self.ram_overshoot_penalty - * overshoot - * (1.0 + h.oom_rate as f64) - * (1.0 + instance.oom_rate.w1h.unwrap_or(0.0)); - } - - s - } - - fn description(&self) -> &'static str { - "Uses historical peak RAM to penalize workers that would likely run out of memory." - } -} - /// Substitute-only `builtin` jobs are getting a more lenient CPU threshold. RAM saturation is still /// applying because a RAM-starved worker can fail a fetch too. #[derive(Debug)] @@ -115,10 +62,9 @@ impl ScoreRule for ResourceSaturationRule { s -= self.penalty; } - let h = job.build_history(); - if h.samples > 0 + if let Some(peak) = job.build_history().predicted_peak_ram_mb && m.ram_free_mb - .is_some_and(|f| h.predicted_peak_ram_mb as f64 * self.ram_fit_headroom > f as f64) + .is_some_and(|f| peak as f64 * self.ram_fit_headroom > f as f64) { s -= self.penalty; } @@ -134,7 +80,7 @@ impl ScoreRule for ResourceSaturationRule { #[cfg(test)] mod tests { use super::*; - use crate::score::context::{HistoryPrediction, ScoredJob, Windowed, WorkerMetricsView}; + use crate::score::context::{HistoryPrediction, ScoredJob, WorkerMetricsView}; use crate::score::weights::RESOURCE_SATURATION_PENALTY as PENALTY; use gradient_types::ids::ProjectId; @@ -181,152 +127,6 @@ mod tests { } } - #[test] - fn ram_overshoot_is_negative_and_scales_with_overshoot() { - let rule = ResourceFitRule::default(); - let m = WorkerMetricsView { - ram_free_mb: Some(1000), - ..Default::default() - }; - let w = worker_with(m); - - let small = job_with_history(HistoryPrediction { - predicted_peak_ram_mb: 1500, - samples: 5, - ..Default::default() - }); - let large = job_with_history(HistoryPrediction { - predicted_peak_ram_mb: 3000, - samples: 5, - ..Default::default() - }); - - let s_small = rule.score(&ctx(&small), &w, &InstanceContext::default()); - let s_large = rule.score(&ctx(&large), &w, &InstanceContext::default()); - assert!(s_small < 0.0); - assert!( - s_large < s_small, - "larger overshoot must be more negative: {s_large} vs {s_small}" - ); - } - - #[test] - fn ram_overshoot_penalty_is_bounded() { - let rule = ResourceFitRule::default(); - let w = worker_with(WorkerMetricsView { - ram_free_mb: Some(100), - ..Default::default() - }); - let job = job_with_history(HistoryPrediction { - predicted_peak_ram_mb: 1_000_000, - oom_rate: 1.0, - samples: 5, - ..Default::default() - }); - let inst = InstanceContext { - oom_rate: Windowed { - w1h: Some(1.0), - ..Default::default() - }, - ..Default::default() - }; - - let s = rule.score(&ctx(&job), &w, &inst); - assert!( - s >= -(rule.ram_overshoot_penalty * rule.max_overshoot * 4.0) - 0.001, - "penalty must be bounded by clamp, got {s}" - ); - assert!( - s > -4000.0, - "penalty must stay below WaitTimeRule cap so wait can overcome it, got {s}" - ); - } - - #[test] - fn higher_oom_rate_is_more_negative_for_same_overshoot() { - let rule = ResourceFitRule::default(); - let w = worker_with(WorkerMetricsView { - ram_free_mb: Some(1000), - ..Default::default() - }); - - let low = job_with_history(HistoryPrediction { - predicted_peak_ram_mb: 2000, - oom_rate: 0.0, - samples: 5, - ..Default::default() - }); - let high = job_with_history(HistoryPrediction { - predicted_peak_ram_mb: 2000, - oom_rate: 0.5, - samples: 5, - ..Default::default() - }); - - assert!( - rule.score(&ctx(&high), &w, &InstanceContext::default()) - < rule.score(&ctx(&low), &w, &InstanceContext::default()) - ); - } - - #[test] - fn eval_ram_overshoot_routes_to_big_ram_worker() { - let rule = ResourceFitRule::default(); - let job = || { - eval_job_with_history(HistoryPrediction { - predicted_peak_ram_mb: 40_000, - samples: 5, - ..Default::default() - }) - }; - let small = worker_with(WorkerMetricsView { - ram_free_mb: Some(16_000), - ..Default::default() - }); - let big = worker_with(WorkerMetricsView { - ram_free_mb: Some(64_000), - ..Default::default() - }); - assert!(rule.score(&ctx(&job()), &small, &InstanceContext::default()) < 0.0); - assert_eq!( - rule.score(&ctx(&job()), &big, &InstanceContext::default()), - 0.0 - ); - } - - #[test] - fn no_samples_is_zero() { - let rule = ResourceFitRule::default(); - let w = worker_with(WorkerMetricsView { - ram_free_mb: Some(100), - ..Default::default() - }); - let job = job_with_history(HistoryPrediction { - predicted_peak_ram_mb: 9000, - avg_cpu_time_ms: 999_999, - samples: 0, - ..Default::default() - }); - assert_eq!(rule.score(&ctx(&job), &w, &InstanceContext::default()), 0.0); - } - - #[test] - fn no_metrics_is_zero() { - let rule = ResourceFitRule::default(); - let w = WorkerContext { - architectures: &[], - system_features: &[], - fetch: false, - metrics: None, - }; - let job = job_with_history(HistoryPrediction { - predicted_peak_ram_mb: 9000, - samples: 5, - ..Default::default() - }); - assert_eq!(rule.score(&ctx(&job), &w, &InstanceContext::default()), 0.0); - } - fn builtin_job() -> ScoredJob<'static> { ScoredJob::new_build( "test", @@ -435,20 +235,15 @@ mod tests { ..Default::default() }; let job = job_with_history(HistoryPrediction { - predicted_peak_ram_mb: 64_000, + predicted_peak_ram_mb: Some(64_000), samples: 5, ..Default::default() }); let sat = ResourceSaturationRule::default(); - let fit = ResourceFitRule::default(); assert_eq!( sat.score(&ctx(&job), &worker_with(cold), &InstanceContext::default()), 0.0 ); - assert_eq!( - fit.score(&ctx(&job), &worker_with(cold), &InstanceContext::default()), - 0.0 - ); let starved = WorkerMetricsView { ram_total_mb: 16_000, @@ -469,7 +264,7 @@ mod tests { fn ram_prediction_exceeding_free_penalizes_and_stacks_with_saturation() { let rule = ResourceSaturationRule::default(); let job = job_with_history(HistoryPrediction { - predicted_peak_ram_mb: 10_000, + predicted_peak_ram_mb: Some(10_000), samples: 5, ..Default::default() }); @@ -507,25 +302,4 @@ mod tests { -2.0 * PENALTY ); } - - #[test] - fn ram_overshoot_more_negative_with_high_instance_oom() { - let rule = ResourceFitRule::default(); - let w = worker_with(WorkerMetricsView { - ram_free_mb: Some(1000), - ..Default::default() - }); - let job = job_with_history(HistoryPrediction { - predicted_peak_ram_mb: 2000, - samples: 5, - ..Default::default() - }); - - let mut low = InstanceContext::default(); - low.oom_rate.w1h = Some(0.0); - let mut high = InstanceContext::default(); - high.oom_rate.w1h = Some(1.0); - - assert!(rule.score(&ctx(&job), &w, &low) > rule.score(&ctx(&job), &w, &high)); - } } diff --git a/backend/gradient-pool/src/score/rules/transfer_limit.rs b/backend/gradient-pool/src/score/rules/transfer_limit.rs new file mode 100644 index 000000000..9c8e092ce --- /dev/null +++ b/backend/gradient-pool/src/score/rules/transfer_limit.rs @@ -0,0 +1,171 @@ +/* + * SPDX-FileCopyrightText: 2026 Wavelens GmbH + * + * SPDX-License-Identifier: AGPL-3.0-only + */ + +use gradient_wire::messages::SMALL_UPLOAD_BYTES; + +use crate::score::context::InstanceContext; +use crate::score::rule::{JobContext, ScoreRule, WorkerContext}; + +const BYTES_PER_MIB: f64 = 1_048_576.0; + +fn slots_full(in_flight: u32, slots: u32) -> bool { + slots > 0 && in_flight >= slots +} + +fn holds_download(job: &JobContext<'_>, instance: &InstanceContext) -> bool { + let mean_bytes = instance.nar_size_mb.w1h.unwrap_or(0.0) * BYTES_PER_MIB; + !job.outputs_present + && slots_full(instance.downloads_in_flight, instance.download_slots) + && job + .missing_nar_size + .is_some_and(|bytes| bytes as f64 > mean_bytes) +} + +fn holds_upload(job: &JobContext<'_>, instance: &InstanceContext) -> bool { + slots_full(instance.uploads_in_flight, instance.upload_slots) + && job + .job + .history() + .output_nar_size + .is_some_and(|bytes| bytes > SMALL_UPLOAD_BYTES) +} + +#[derive(Debug)] +pub struct TransferLimitRule { + pub hold: f64, +} + +impl Default for TransferLimitRule { + fn default() -> Self { + Self { + hold: crate::score::weights::TRANSFER_LIMIT_HOLD, + } + } +} + +impl ScoreRule for TransferLimitRule { + fn name(&self) -> &'static str { + "TransferLimitRule" + } + + fn score( + &self, + job: &JobContext<'_>, + _worker: &WorkerContext<'_>, + instance: &InstanceContext, + ) -> f64 { + if job.job.build().is_none() { + return 0.0; + } + + if holds_download(job, instance) || holds_upload(job, instance) { + -self.hold + } else { + 0.0 + } + } + + fn description(&self) -> &'static str { + "Holds a build with a larger than average download, or an output over 1 MiB, below the assignment floor while the server's download or upload slots are full. Jobs with local inputs take the slot, and the wait bonus releases a held job." + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::score::context::{HistoryPrediction, ScoredJob, Windowed}; + use gradient_types::ids::ProjectId; + + fn build(output_nar_size: Option) -> ScoredJob<'static> { + ScoredJob::new_build( + "j", + ProjectId::now_v7(), + "x86_64-linux", + false, + false, + None, + None, + HistoryPrediction { + output_nar_size, + ..Default::default() + }, + ) + } + + fn ctx<'a>(job: &'a ScoredJob<'a>, missing_nar_size: u64) -> JobContext<'a> { + JobContext { + job, + missing_count: Some(0), + missing_nar_size: Some(missing_nar_size), + outputs_present: false, + dependency_count: 0, + queued_at: gradient_types::now(), + ready_at: gradient_types::now(), + project_work_share: None, + prioritized: false, + build_request: false, + rescore_count: 0, + now: gradient_types::now(), + } + } + + fn worker() -> WorkerContext<'static> { + WorkerContext { + architectures: &[], + system_features: &[], + fetch: false, + metrics: None, + } + } + + fn downloads(in_flight: u32) -> InstanceContext { + InstanceContext { + downloads_in_flight: in_flight, + download_slots: 16, + nar_size_mb: Windowed { + w1h: Some(100.0), + ..Default::default() + }, + ..Default::default() + } + } + + #[test] + fn a_large_download_waits_while_the_download_slots_are_full() { + let rule = TransferLimitRule::default(); + let job = build(None); + let large = 500 * 1_048_576; + + assert_eq!( + rule.score(&ctx(&job, large), &worker(), &downloads(16)), + -rule.hold + ); + assert_eq!( + rule.score(&ctx(&job, large), &worker(), &downloads(15)), + 0.0 + ); + assert_eq!( + rule.score(&ctx(&job, 1_048_576), &worker(), &downloads(16)), + 0.0, + "a job with local inputs takes the slot" + ); + } + + #[test] + fn a_large_output_waits_while_the_upload_slots_are_full() { + let rule = TransferLimitRule::default(); + let full = InstanceContext { + uploads_in_flight: 8, + upload_slots: 8, + ..Default::default() + }; + let large = build(Some(SMALL_UPLOAD_BYTES + 1)); + let small = build(Some(SMALL_UPLOAD_BYTES)); + + assert_eq!(rule.score(&ctx(&large, 0), &worker(), &full), -rule.hold); + assert_eq!(rule.score(&ctx(&small, 0), &worker(), &full), 0.0); + } +} diff --git a/backend/gradient-pool/src/score/weights.rs b/backend/gradient-pool/src/score/weights.rs index 3da5d38ce..032c477c0 100644 --- a/backend/gradient-pool/src/score/weights.rs +++ b/backend/gradient-pool/src/score/weights.rs @@ -5,59 +5,38 @@ */ //! Scores are additive, largest first. QOS_PRIORITIZED is outranking any wait, and WAIT_TIME_CAP is -//! out-budgeting the rest against starvation. Resource penalties are keeping doomed placements out. -//! REALISED_OUTPUTS_BONUS, CpuAffinityRule and the cache-warmth caps are next, and the rest are -//! tie-breakers. +//! out-budgeting the estimated time against starvation. Resource penalties are keeping doomed +//! placements out, and the estimated time is ranking everything else. pub const ASSIGN_FLOOR: f64 = 0.0; -pub const MISSING_PATHS_CAP: f64 = 200.0; -pub const MISSING_PATHS_BASELINE_K: f64 = 2.0; -pub const MISSING_PATHS_FALLBACK_AVG: f64 = 20.0; - -pub const MISSING_NAR_SIZE_CAP: f64 = 500.0; -pub const MISSING_NAR_SIZE_BASELINE_K: f64 = 2.0; - -pub const REALISED_OUTPUTS_BONUS: f64 = 2500.0; - -pub const REAL_BUILD_BONUS: f64 = 50.0; -pub const ARCHLESS_BUILTIN_BONUS: f64 = 100.0; - -pub const DEPENDENCY_COUNT_CAP: f64 = 50.0; -pub const DEPENDENCY_COUNT_BASELINE_K: f64 = 2.0; -pub const DEPENDENCY_COUNT_FALLBACK_AVG: f64 = 10.0; - pub const WAIT_TIME_GAIN: f64 = 60.0; pub const WAIT_TIME_FALLBACK_AVG_SECS: f64 = 60.0; pub const WAIT_TIME_CAP: f64 = 4000.0; +pub const TRANSFER_LIMIT_HOLD: f64 = WAIT_TIME_CAP; + pub const RESERVE_FETCH_PENALTY: f64 = 300.0; pub const RESCORE_MAX_ROUNDS: u32 = 4; -pub const RESOURCE_FIT_RAM_PENALTY: f64 = 400.0; -pub const RESOURCE_FIT_MAX_OVERSHOOT: f64 = 2.0; - -pub const CPU_AFFINITY_WEIGHT: f64 = 400.0; -pub const CPU_HEAVINESS_CAP: f64 = 3.0; -pub const CPU_HEAVY_THRESHOLD_MS: u64 = 60_000; - -pub const RESOURCE_SATURATION_PENALTY: f64 = 5000.0; +pub const ESTIMATED_TIME_POINTS_PER_SEC: f64 = 1.0; +pub const ESTIMATED_TIME_CAP_SECS: f64 = 3600.0; +pub const BUILD_CONTENTION_PER_BUILD: f64 = 0.04; +pub const CORE_SCORE_RATIO_MIN: f64 = 0.5; +pub const CORE_SCORE_RATIO_MAX: f64 = 2.0; +pub const STORAGE_READ_FALLBACK_MBPS: f64 = 1200.0; +pub const STORAGE_WRITE_FALLBACK_MBPS: f64 = 1120.0; +pub const STORED_PER_NAR_BYTE_FALLBACK: f64 = 1.0; +pub const PER_PATH_FALLBACK_SECS: f64 = 0.18; + +pub const RESOURCE_SATURATION_PENALTY: f64 = + 5000.0 + ESTIMATED_TIME_POINTS_PER_SEC * ESTIMATED_TIME_CAP_SECS; pub const CPU_SATURATED_PCT: f64 = 80.0; pub const CPU_SATURATED_PCT_BUILTIN: f64 = 90.0; pub const RAM_SATURATED_FREE_FRAC: f64 = 0.10; pub const RAM_FIT_HEADROOM: f64 = 1.1; -pub const PREFER_LOCAL_BONUS: f64 = 150.0; -pub const PREFER_LOCAL_MISS_PENALTY: f64 = 20.0; - -pub const NETWORK_AFFINITY_BONUS: f64 = 80.0; -pub const NETWORK_REFERENCE_MBPS: f64 = 100.0; - -pub const DISK_AFFINITY_BONUS: f64 = 60.0; -pub const DISK_HEAVY_THRESHOLD_BYTES: u64 = 100 * 1_048_576; -pub const DISK_REFERENCE_MBPS: f64 = 500.0; - pub const FAIR_SHARE_WEIGHT: f64 = 500.0; pub const QOS_PRIORITIZED: f64 = 5000.0; diff --git a/backend/gradient-pool/src/worker_pool.rs b/backend/gradient-pool/src/worker_pool.rs index 3958b826c..e98836b0f 100644 --- a/backend/gradient-pool/src/worker_pool.rs +++ b/backend/gradient-pool/src/worker_pool.rs @@ -9,11 +9,11 @@ use std::sync::Arc; use std::sync::atomic::{AtomicI64, Ordering}; use gradient_types::ids::ProjectId; -use gradient_wire::types::{GradientCapabilities, JobKind}; +use gradient_wire::types::{BuildStage, GradientCapabilities, JobKind}; use crate::peer_auth::PeerAuth; use crate::session_port::{SessionPort, SessionSignal}; -use crate::worker_state::{Active, Draining, TypedWorker}; +use crate::worker_state::{Active, Draining, TypedWorker, WorkerShared}; pub enum WorkerSlot { Active(TypedWorker), @@ -165,14 +165,16 @@ impl WorkerPool { cpu_usage_pct: f32, ram_free_mb: u64, disk_speed_mbps: Option, - network_speed_mbps: Option, + upload_speed_mbps: Option, + download_speed_mbps: Option, ) { if let Some(slot) = self.workers.get_mut(id) { let s = slot.shared_mut(); s.cpu_usage_pct = Some(cpu_usage_pct); s.ram_free_mb = Some(ram_free_mb); s.disk_speed_mbps = disk_speed_mbps; - s.network_speed_mbps = network_speed_mbps; + s.upload_speed_mbps = upload_speed_mbps; + s.download_speed_mbps = download_speed_mbps; } } @@ -186,7 +188,9 @@ impl WorkerPool { ram_free_mb: s.ram_free_mb, cpu_usage_pct: s.cpu_usage_pct, disk_speed_mbps: s.disk_speed_mbps, - network_speed_mbps: s.network_speed_mbps, + upload_speed_mbps: s.upload_speed_mbps, + download_speed_mbps: s.download_speed_mbps, + running_builds: s.jobs_in(BuildStage::Build), } }) } @@ -201,7 +205,7 @@ impl WorkerPool { shared.session.signal(SessionSignal::Close { reason: "unregistered by the scheduler".into(), }); - shared.assigned_jobs.iter().cloned().collect() + shared.assigned_jobs.keys().cloned().collect() }) .unwrap_or_default() } @@ -265,10 +269,29 @@ impl WorkerPool { pub fn assign_job(&mut self, worker_id: &str, job_id: &str) { if let Some(slot) = self.workers.get_mut(worker_id) { - slot.shared_mut().assigned_jobs.insert(job_id.to_owned()); + slot.shared_mut() + .assigned_jobs + .insert(job_id.to_owned(), None); } } + pub fn enter_stage(&mut self, worker_id: &str, job_id: &str, stage: BuildStage) { + if let Some(current) = self + .workers + .get_mut(worker_id) + .and_then(|slot| slot.shared_mut().assigned_jobs.get_mut(job_id)) + { + *current = Some(stage); + } + } + + pub fn fleet_jobs_in(&self, stage: BuildStage) -> u32 { + self.workers + .values() + .map(|slot| slot.shared().jobs_in(stage)) + .sum() + } + pub fn release_job(&mut self, worker_id: &str, job_id: &str) -> bool { match self.workers.get_mut(worker_id) { Some(slot) => { @@ -295,14 +318,28 @@ impl WorkerPool { } pub fn mean_cpu_core_score(&self) -> Option { - let scores: Vec = self + self.mean_of(|s| { + Some(s.cpu_core_score) + .filter(|score| *score > 0) + .map(f64::from) + }) + } + + pub fn mean_upload_speed_mbps(&self) -> Option { + self.mean_of(|s| s.upload_speed_mbps.map(f64::from)) + } + + pub fn mean_download_speed_mbps(&self) -> Option { + self.mean_of(|s| s.download_speed_mbps.map(f64::from)) + } + + fn mean_of(&self, value: impl Fn(&WorkerShared) -> Option) -> Option { + let values: Vec = self .workers .values() - .map(|slot| slot.shared().cpu_core_score) - .filter(|s| *s > 0) - .map(f64::from) + .filter_map(|slot| value(slot.shared())) .collect(); - (!scores.is_empty()).then(|| scores.iter().sum::() / scores.len() as f64) + (!values.is_empty()).then(|| values.iter().sum::() / values.len() as f64) } fn info_for(&self, id: &str, slot: &WorkerSlot) -> WorkerInfo { @@ -320,7 +357,8 @@ impl WorkerPool { ram_free_mb: s.ram_free_mb, ram_total_mb: s.ram_total_mb, disk_speed_mbps: s.disk_speed_mbps, - network_speed_mbps: s.network_speed_mbps, + upload_speed_mbps: s.upload_speed_mbps, + download_speed_mbps: s.download_speed_mbps, } } @@ -361,7 +399,9 @@ pub struct WorkerInfo { #[serde(skip)] pub disk_speed_mbps: Option, #[serde(skip)] - pub network_speed_mbps: Option, + pub upload_speed_mbps: Option, + #[serde(skip)] + pub download_speed_mbps: Option, } #[cfg(test)] @@ -445,6 +485,18 @@ mod tests { assert_eq!(pool.worker_count(), 0); } + #[test] + fn fleet_speed_means_skip_workers_that_measured_nothing() { + let mut pool = WorkerPool::new(); + for (id, upload) in [("w1", Some(100.0)), ("w2", Some(300.0)), ("w3", None)] { + pool.register(id.into(), caps(), HashSet::new(), port().0); + pool.update_metrics(id, 0.0, 0, None, upload, None); + } + + assert_eq!(pool.mean_upload_speed_mbps(), Some(200.0)); + assert_eq!(pool.mean_download_speed_mbps(), None); + } + #[test] fn mean_cpu_core_score_skips_workers_without_one() { let mut pool = WorkerPool::new(); @@ -552,14 +604,16 @@ mod tests { assert_eq!(view.cpu_usage_pct, None); assert_eq!(view.ram_free_mb, None); assert_eq!(view.disk_speed_mbps, None); - assert_eq!(view.network_speed_mbps, None); + assert_eq!(view.upload_speed_mbps, None); + assert_eq!(view.download_speed_mbps, None); - pool.update_metrics("w1", 42.5, 3000, Some(550.0), Some(120.0)); + pool.update_metrics("w1", 42.5, 3000, Some(550.0), Some(120.0), Some(900.0)); let view = pool.metrics_for("w1").unwrap(); assert_eq!(view.cpu_usage_pct, Some(42.5)); assert_eq!(view.ram_free_mb, Some(3000)); assert_eq!(view.disk_speed_mbps, Some(550.0)); - assert_eq!(view.network_speed_mbps, Some(120.0)); + assert_eq!(view.upload_speed_mbps, Some(120.0)); + assert_eq!(view.download_speed_mbps, Some(900.0)); assert_eq!(view.cpu_count, 4); assert_eq!(view.ram_total_mb, 8192); @@ -591,7 +645,7 @@ mod tests { ..Default::default() }, ); - pool.update_metrics("w1", 12.5, 9000, None, None); + pool.update_metrics("w1", 12.5, 9000, None, None, None); let caps = pool.worker_caps("w1").unwrap(); assert!(caps.fetch); @@ -726,6 +780,30 @@ mod tests { assert_eq!(pool.all_workers()[0].assigned_job_count, 0); } + #[test] + fn a_jobs_stage_counts_for_its_worker_and_the_fleet_until_it_is_released() { + let mut pool = WorkerPool::new(); + for id in ["w1", "w2"] { + pool.register(id.into(), caps(), HashSet::new(), port().0); + } + pool.assign_job("w1", "a"); + pool.assign_job("w1", "b"); + pool.assign_job("w2", "c"); + pool.enter_stage("w1", "a", BuildStage::Build); + pool.enter_stage("w1", "b", BuildStage::Prefetch); + pool.enter_stage("w2", "c", BuildStage::Prefetch); + pool.enter_stage("w2", "never-assigned", BuildStage::Build); + + assert_eq!(pool.metrics_for("w1").unwrap().running_builds, 1); + assert_eq!(pool.metrics_for("w2").unwrap().running_builds, 0); + assert_eq!(pool.fleet_jobs_in(BuildStage::Prefetch), 2); + + pool.enter_stage("w1", "b", BuildStage::Build); + pool.release_job("w1", "a"); + assert_eq!(pool.metrics_for("w1").unwrap().running_builds, 1); + assert_eq!(pool.fleet_jobs_in(BuildStage::Prefetch), 1); + } + #[test] fn test_all_workers_info() { let mut pool = WorkerPool::new(); diff --git a/backend/gradient-pool/src/worker_state.rs b/backend/gradient-pool/src/worker_state.rs index d6b7b77ac..0d1891277 100644 --- a/backend/gradient-pool/src/worker_state.rs +++ b/backend/gradient-pool/src/worker_state.rs @@ -4,14 +4,14 @@ * SPDX-License-Identifier: AGPL-3.0-only */ -use std::collections::HashSet; +use std::collections::{HashMap, HashSet}; use std::marker::PhantomData; use std::sync::Arc; use std::sync::atomic::AtomicI64; use gradient_types::ids::ProjectId; -use gradient_wire::types::GradientCapabilities; +use gradient_wire::types::{BuildStage, GradientCapabilities}; use crate::peer_auth::PeerAuth; use crate::session_port::SessionPort; @@ -58,8 +58,9 @@ pub struct WorkerShared { pub cpu_usage_pct: Option, pub ram_free_mb: Option, pub disk_speed_mbps: Option, - pub network_speed_mbps: Option, - pub assigned_jobs: HashSet, + pub upload_speed_mbps: Option, + pub download_speed_mbps: Option, + pub assigned_jobs: HashMap>, pub peer_auth: PeerAuth, pub sent_candidates: HashSet, pub session: Arc, @@ -69,6 +70,13 @@ pub struct WorkerShared { } impl WorkerShared { + pub fn jobs_in(&self, stage: BuildStage) -> u32 { + self.assigned_jobs + .values() + .filter(|s| **s == Some(stage)) + .count() as u32 + } + pub fn profile(&self) -> WorkerProfile { WorkerProfile { architectures: self.architectures.clone(), @@ -144,8 +152,9 @@ impl TypedWorker { cpu_usage_pct: None, ram_free_mb: None, disk_speed_mbps: None, - network_speed_mbps: None, - assigned_jobs: HashSet::new(), + upload_speed_mbps: None, + download_speed_mbps: None, + assigned_jobs: HashMap::new(), peer_auth: PeerAuth::from_peers(authorized_peers), sent_candidates: HashSet::new(), session, diff --git a/backend/gradient-proto/src/handler/cache_session.rs b/backend/gradient-proto/src/handler/cache_session.rs index e32a6b819..c8e9948ac 100644 --- a/backend/gradient-proto/src/handler/cache_session.rs +++ b/backend/gradient-proto/src/handler/cache_session.rs @@ -146,9 +146,11 @@ pub async fn handle_cache_socket( debug!(%cache_id, %store_path, "skipping path not in this cache"); continue; } - let Some(slot) = - super::nar_serve::ServeSlot::acquire(Arc::clone(&nar_serve_semaphore)) - .await + let Some(slot) = super::nar_serve::ServeSlot::acquire( + Arc::clone(&nar_serve_semaphore), + Arc::clone(&state.nar_downloads), + ) + .await else { warn!("nar serve semaphore closed"); return; @@ -232,6 +234,7 @@ mod tests { missing_paths: vec![], spans: vec![], elapsed_ms: 0, + metrics: None, }) .is_some() ); diff --git a/backend/gradient-proto/src/handler/inbound.rs b/backend/gradient-proto/src/handler/inbound.rs index 2a5752072..adb22955a 100644 --- a/backend/gradient-proto/src/handler/inbound.rs +++ b/backend/gradient-proto/src/handler/inbound.rs @@ -190,14 +190,16 @@ impl<'a> InboundContext<'a> { cpu_usage_pct, ram_free_mb, disk_speed_mbps, - network_speed_mbps, + upload_speed_mbps, + download_speed_mbps, } => { - self.spawn_worker_metrics( + self.spawn_worker_metrics(WorkerMetrics { cpu_usage_pct, ram_free_mb, disk_speed_mbps, - network_speed_mbps, - ); + upload_speed_mbps, + download_speed_mbps, + }); true } ClientMessage::RequestJobList => self.on_request_job_list().await, @@ -298,6 +300,7 @@ impl<'a> InboundContext<'a> { missing_paths, spans, elapsed_ms, + metrics, } => { let report = ReportedTimeline::received(spans, elapsed_ms); self.forget_uploads(&job_id, uploads).await; @@ -321,6 +324,7 @@ impl<'a> InboundContext<'a> { error, kind, missing_paths, + metrics, }) .await; } @@ -330,6 +334,10 @@ impl<'a> InboundContext<'a> { self.on_draining().await; true } + ClientMessage::HandoverDone | ClientMessage::PathsAdded { .. } => { + warn!(peer_id = %self.peer_id, "worker sent a shared-worker message the server does not handle"); + true + } ClientMessage::NarRequest { job_id, paths } => { self.on_nar_request(job_id, paths).await; true @@ -431,22 +439,10 @@ impl<'a> InboundContext<'a> { } } - fn spawn_worker_metrics( - &self, - cpu_usage_pct: f32, - ram_free_mb: u64, - disk_speed_mbps: Option, - network_speed_mbps: Option, - ) { + fn spawn_worker_metrics(&self, metrics: WorkerMetrics) { let rpc = self.rpc(); self.state.shutdown.spawn(async move { - rpc.on_worker_metrics( - cpu_usage_pct, - ram_free_mb, - disk_speed_mbps, - network_speed_mbps, - ) - .await; + rpc.on_worker_metrics(metrics).await; }); } @@ -721,7 +717,8 @@ impl<'a> InboundContext<'a> { let peer_id = self.peer_id.to_owned(); let job_id = job_id.clone(); shutdown.spawn(async move { - let Some(_slot) = ServeSlot::acquire(permit).await else { + let server = Arc::clone(&state.nar_downloads); + let Some(_slot) = ServeSlot::acquire(permit, server).await else { return; }; @@ -748,7 +745,8 @@ impl<'a> InboundContext<'a> { let peer_id = self.peer_id.to_owned(); let shutdown = self.state.shutdown.clone(); shutdown.spawn(async move { - let Some(_slot) = ServeSlot::acquire(permit).await else { + let server = Arc::clone(&state.nar_downloads); + let Some(_slot) = ServeSlot::acquire(permit, server).await else { return; }; @@ -922,24 +920,10 @@ impl RpcContext { } } - async fn on_worker_metrics( - &self, - cpu_usage_pct: f32, - ram_free_mb: u64, - disk_speed_mbps: Option, - network_speed_mbps: Option, - ) { - debug!(peer_id = %self.peer_id, cpu_usage_pct, ram_free_mb, ?disk_speed_mbps, ?network_speed_mbps, "WorkerMetrics"); + async fn on_worker_metrics(&self, metrics: WorkerMetrics) { + debug!(peer_id = %self.peer_id, ?metrics, "WorkerMetrics"); self.scheduler - .update_worker_metrics( - &self.peer_id, - WorkerMetrics { - cpu_usage_pct, - ram_free_mb, - disk_speed_mbps, - network_speed_mbps, - }, - ) + .update_worker_metrics(&self.peer_id, metrics) .await; } } diff --git a/backend/gradient-proto/src/handler/job_events.rs b/backend/gradient-proto/src/handler/job_events.rs index 4d521917f..8beb5c059 100644 --- a/backend/gradient-proto/src/handler/job_events.rs +++ b/backend/gradient-proto/src/handler/job_events.rs @@ -14,7 +14,7 @@ use gradient_entity::dispatched_job::DispatchedJobOutcome; use gradient_scheduler::{ReportedTimeline, Scheduler}; use gradient_types::ids::DispatchedJobId; use gradient_util::shutdown::Shutdown; -use gradient_wire::types::BuildFailureKind; +use gradient_wire::types::{BuildFailureKind, BuildMetrics}; use tokio::sync::mpsc; use tokio::task::JoinHandle; use tracing::{Instrument as _, debug, debug_span, error, info, warn}; @@ -41,6 +41,7 @@ pub(super) enum JobEvent { error: String, kind: BuildFailureKind, missing_paths: Vec, + metrics: Option, }, } @@ -151,7 +152,11 @@ impl ApplyJobEvent for SchedulerJobEvents { error, kind, missing_paths, - } => self.failed(job_id, error, kind, missing_paths).await, + metrics, + } => { + self.failed(job_id, error, kind, missing_paths, metrics) + .await + } } } } @@ -213,6 +218,9 @@ impl SchedulerJobEvents { } } JobUpdateKind::Compressing => {} + JobUpdateKind::Stage(stage) => { + scheduler.enter_build_stage(peer_id, &job_id, stage).await; + } JobUpdateKind::EvalStats(report) => { if let Err(e) = scheduler.record_eval_metrics(&job_id, report).await { error!(%peer_id, %job_id, error = %e, "record_eval_metrics failed"); @@ -257,11 +265,12 @@ impl SchedulerJobEvents { error: String, kind: BuildFailureKind, missing_paths: Vec, + metrics: Option, ) { let peer_id = self.peer_id.as_str(); if let Err(e) = self .scheduler - .handle_job_failed(peer_id, &job_id, &error, kind, &missing_paths) + .handle_job_failed(peer_id, &job_id, &error, kind, &missing_paths, metrics) .await { error!(%peer_id, %job_id, error = %e, "handle_job_failed failed"); diff --git a/backend/gradient-proto/src/handler/nar_serve.rs b/backend/gradient-proto/src/handler/nar_serve.rs index f6a7d5b96..649b00289 100644 --- a/backend/gradient-proto/src/handler/nar_serve.rs +++ b/backend/gradient-proto/src/handler/nar_serve.rs @@ -19,7 +19,8 @@ use tracing::warn; use super::socket::ProtoWriter; pub(super) struct ServeSlot { - _permit: OwnedSemaphorePermit, + _connection: OwnedSemaphorePermit, + _server: OwnedSemaphorePermit, gauges: &'static Gauges, } @@ -32,22 +33,30 @@ impl Drop for Waiting { } impl ServeSlot { - pub(super) async fn acquire(semaphore: Arc) -> Option { - Self::acquire_with(semaphore, &GAUGES).await + pub(super) async fn acquire( + connection: Arc, + server: Arc, + ) -> Option { + Self::acquire_with(connection, server, &GAUGES).await } + /// The connection's own permit comes first. A connection queueing many paths must not hold + /// server-wide permits while it waits for its own. pub(super) async fn acquire_with( - semaphore: Arc, + connection: Arc, + server: Arc, gauges: &'static Gauges, ) -> Option { gauges.serves_waiting.inc(); let waiting = Waiting(gauges); - let permit = semaphore.acquire_owned().await.ok()?; + let connection = connection.acquire_owned().await.ok()?; + let server = server.acquire_owned().await.ok()?; drop(waiting); gauges.serves_active.inc(); Some(Self { - _permit: permit, + _connection: connection, + _server: server, gauges, }) } @@ -324,12 +333,22 @@ mod serve_slot_tests { Box::leak(Box::new(Gauges::new())) } + fn permits(n: usize) -> Arc { + Arc::new(Semaphore::new(n)) + } + + async fn take( + connection: &Arc, + server: &Arc, + g: &'static Gauges, + ) -> Option { + ServeSlot::acquire_with(Arc::clone(connection), Arc::clone(server), g).await + } + #[tokio::test] async fn a_slot_is_active_until_dropped() { let g = gauges(); - let slot = ServeSlot::acquire_with(Arc::new(Semaphore::new(1)), g) - .await - .expect("slot"); + let slot = take(&permits(1), &permits(1), g).await.expect("slot"); assert_eq!(g.serves_active.get(), 1); assert_eq!(g.serves_waiting.get(), 0); @@ -340,11 +359,9 @@ mod serve_slot_tests { #[tokio::test] async fn a_queued_serve_counts_as_waiting() { let g = gauges(); - let sem = Arc::new(Semaphore::new(1)); - let held = ServeSlot::acquire_with(Arc::clone(&sem), g) - .await - .expect("slot"); - let mut queued = Box::pin(ServeSlot::acquire_with(Arc::clone(&sem), g)); + let (sem, server) = (permits(1), permits(8)); + let held = take(&sem, &server, g).await.expect("slot"); + let mut queued = Box::pin(take(&sem, &server, g)); let still_queued = tokio::time::timeout(Duration::from_millis(20), &mut queued).await; assert!(still_queued.is_err()); @@ -356,11 +373,28 @@ mod serve_slot_tests { drop(slot); } + #[tokio::test] + async fn the_server_wide_limit_holds_back_another_connection() { + let g = gauges(); + let server = permits(1); + let held = take(&permits(8), &server, g).await.expect("slot"); + let other_connection = permits(8); + let mut other = Box::pin(take(&other_connection, &server, g)); + + assert!( + tokio::time::timeout(Duration::from_millis(20), &mut other) + .await + .is_err() + ); + drop(held); + assert!(other.await.is_some()); + } + #[tokio::test] async fn cancelled_wait_releases_waiting() { let g = gauges(); - let sem = Arc::new(Semaphore::new(0)); - let waiting = ServeSlot::acquire_with(Arc::clone(&sem), g); + let (sem, server) = (permits(0), permits(1)); + let waiting = take(&sem, &server, g); let cancelled = tokio::time::timeout(Duration::from_millis(20), waiting).await; assert!(cancelled.is_err()); @@ -371,10 +405,10 @@ mod serve_slot_tests { #[tokio::test] async fn a_closed_semaphore_yields_no_slot() { let g = gauges(); - let sem = Arc::new(Semaphore::new(0)); + let sem = permits(0); sem.close(); - assert!(ServeSlot::acquire_with(sem, g).await.is_none()); + assert!(take(&sem, &permits(1), g).await.is_none()); assert_eq!(g.serves_waiting.get(), 0); } } diff --git a/backend/gradient-proto/src/handler/upload/mod.rs b/backend/gradient-proto/src/handler/upload/mod.rs index 35b7d759e..0d8ab11b9 100644 --- a/backend/gradient-proto/src/handler/upload/mod.rs +++ b/backend/gradient-proto/src/handler/upload/mod.rs @@ -552,6 +552,7 @@ mod tests { missing_paths: Vec::new(), spans: Vec::new(), elapsed_ms: 0, + metrics: None, }; ctx.handle(failed, uploads).await; diff --git a/backend/gradient-report/src/tables.rs b/backend/gradient-report/src/tables.rs index 2fce7327a..c1233ed36 100644 --- a/backend/gradient-report/src/tables.rs +++ b/backend/gradient-report/src/tables.rs @@ -511,8 +511,8 @@ pub fn instance_tables() -> &'static [TableSpec] { ), spec!( "worker_sample", - "CREATE TABLE worker_sample (id TEXT, worker_id TEXT, at TEXT, cpu_usage_pct REAL, ram_free_mb INTEGER, ram_total_mb INTEGER, disk_speed_mbps REAL, network_speed_mbps REAL, assigned_jobs INTEGER, max_concurrent_builds INTEGER, state INTEGER, capabilities TEXT)", - "WITH w AS (SELECT created_at AS started, COALESCE(finished_at, (now() AT TIME ZONE \'UTC\')) AS ended FROM evaluation WHERE id = $1) SELECT s.id::text, s.worker_id::text, s.at::text, s.cpu_usage_pct::text, s.ram_free_mb::text, s.ram_total_mb::text, s.disk_speed_mbps::text, s.network_speed_mbps::text, s.assigned_jobs::text, s.max_concurrent_builds::text, s.state::text, s.capabilities::text FROM worker_sample s, w WHERE s.worker_id IN (SELECT worker_id FROM dispatched_job WHERE evaluation_id = $1) AND s.at BETWEEN w.started AND w.ended", + "CREATE TABLE worker_sample (id TEXT, worker_id TEXT, at TEXT, cpu_usage_pct REAL, ram_free_mb INTEGER, ram_total_mb INTEGER, disk_speed_mbps REAL, upload_speed_mbps REAL, download_speed_mbps REAL, assigned_jobs INTEGER, max_concurrent_builds INTEGER, state INTEGER, capabilities TEXT)", + "WITH w AS (SELECT created_at AS started, COALESCE(finished_at, (now() AT TIME ZONE \'UTC\')) AS ended FROM evaluation WHERE id = $1) SELECT s.id::text, s.worker_id::text, s.at::text, s.cpu_usage_pct::text, s.ram_free_mb::text, s.ram_total_mb::text, s.disk_speed_mbps::text, s.upload_speed_mbps::text, s.download_speed_mbps::text, s.assigned_jobs::text, s.max_concurrent_builds::text, s.state::text, s.capabilities::text FROM worker_sample s, w WHERE s.worker_id IN (SELECT worker_id FROM dispatched_job WHERE evaluation_id = $1) AND s.at BETWEEN w.started AND w.ended", "the workers that ran this evaluation, while it ran", [ "id", @@ -522,7 +522,8 @@ pub fn instance_tables() -> &'static [TableSpec] { "ram_free_mb", "ram_total_mb", "disk_speed_mbps", - "network_speed_mbps", + "upload_speed_mbps", + "download_speed_mbps", "assigned_jobs", "max_concurrent_builds", "state", diff --git a/backend/gradient-scheduler/src/actor.rs b/backend/gradient-scheduler/src/actor.rs index a1e1a45f6..150664bb8 100644 --- a/backend/gradient-scheduler/src/actor.rs +++ b/backend/gradient-scheduler/src/actor.rs @@ -13,7 +13,9 @@ use gradient_pool::score::{InstanceContext, ScoringPolicy}; use gradient_types::ids::{ ClusterAttemptId, DerivationBuildId, DispatchedJobId, EvaluationId, ProjectId, }; -use gradient_wire::types::{CandidateScore, GradientCapabilities, JobCandidate, JobKind}; +use gradient_wire::types::{ + BuildStage, CandidateScore, GradientCapabilities, JobCandidate, JobKind, +}; use ractor::{Actor, ActorProcessingErr, ActorRef, RpcReplyPort}; use tracing::{debug, info}; @@ -45,7 +47,8 @@ pub struct WorkerMetrics { pub cpu_usage_pct: f32, pub ram_free_mb: u64, pub disk_speed_mbps: Option, - pub network_speed_mbps: Option, + pub upload_speed_mbps: Option, + pub download_speed_mbps: Option, } #[derive(Debug, Clone, Default)] @@ -78,6 +81,8 @@ pub struct Counts { pub pending_builds: u32, pub active_builds: u32, pub cpu_core_score_mean: Option, + pub upload_speed_mean_mbps: Option, + pub download_speed_mean_mbps: Option, } #[allow( @@ -126,6 +131,11 @@ pub enum SchedulerMsg { worker: String, reply: RpcReplyPort<()>, }, + EnterStage { + worker: String, + job_id: String, + stage: BuildStage, + }, Enqueue { job_id: String, job: PendingJob, @@ -315,6 +325,14 @@ impl SchedulerCore { self.pool.signal_active(SessionSignal::Offers(self.offers)); } + fn with_live_transfers(&self, instance: &InstanceContext) -> InstanceContext { + InstanceContext { + downloads_in_flight: self.pool.fleet_jobs_in(BuildStage::Prefetch), + uploads_in_flight: self.pool.fleet_jobs_in(BuildStage::Upload), + ..*instance + } + } + fn auth_and_caps(&self, worker: &str) -> (Option>, Option) { let authorized = self .pool @@ -361,13 +379,14 @@ impl SchedulerCore { return self.idle(worker, slot, caps.as_ref()); } let policy = Arc::clone(&self.policy); + let instance = self.with_live_transfers(instance); match self.tracker.take_best_of_kind( worker, authorized.as_ref(), caps.as_ref(), kind, &*policy, - instance, + &instance, ) { Some(assignment) => { self.idle.clear(worker, slot); @@ -427,6 +446,7 @@ impl SchedulerCore { .collect(); let policy = Arc::clone(&self.policy); + let instance = self.with_live_transfers(instance); let seats: Vec = placement .seats .iter() @@ -444,7 +464,7 @@ impl SchedulerCore { &m.key, &job, &*policy, - instance, + &instance, ), job, role: m.role.clone(), @@ -664,7 +684,8 @@ impl Actor for CoreActor { metrics.cpu_usage_pct, metrics.ram_free_mb, metrics.disk_speed_mbps, - metrics.network_speed_mbps, + metrics.upload_speed_mbps, + metrics.download_speed_mbps, ); let _ = reply.send(()); } @@ -673,6 +694,11 @@ impl Actor for CoreActor { core.idle.forget_worker(&worker); let _ = reply.send(()); } + SchedulerMsg::EnterStage { + worker, + job_id, + stage, + } => core.pool.enter_stage(&worker, &job_id, stage), SchedulerMsg::Enqueue { job_id, job, reply } => { core.tracker.add_pending(job_id.clone(), job); core.pool.remove_sent_candidate(&job_id); @@ -922,6 +948,8 @@ impl Actor for CoreActor { pending_builds, active_builds, cpu_core_score_mean: core.pool.mean_cpu_core_score(), + upload_speed_mean_mbps: core.pool.mean_upload_speed_mbps(), + download_speed_mean_mbps: core.pool.mean_download_speed_mbps(), }); } SchedulerMsg::PendingSnapshot { reply } => { diff --git a/backend/gradient-scheduler/src/cluster/recovery.rs b/backend/gradient-scheduler/src/cluster/recovery.rs index 9a560ea29..80a852b40 100644 --- a/backend/gradient-scheduler/src/cluster/recovery.rs +++ b/backend/gradient-scheduler/src/cluster/recovery.rs @@ -143,7 +143,7 @@ impl Scheduler { for disposition in dispositions { let settled = match disposition { Disposition::Complete(job) => self.settle_completed(job, false).await, - Disposition::Fail(job, failure) => self.settle_failed(job, &failure).await, + Disposition::Fail(job, failure) => self.settle_failed(job, &failure, None).await, Disposition::Requeue(job) => { requeue.push(job); Ok(()) @@ -160,7 +160,7 @@ impl Scheduler { async fn settle_single(&self, report: MemberReport) -> Result<()> { match report { MemberReport::Completed { job } => self.settle_completed(job, false).await, - MemberReport::Failed { job, failure } => self.settle_failed(job, &failure).await, + MemberReport::Failed { job, failure } => self.settle_failed(job, &failure, None).await, MemberReport::Lost { job } => { crate::build::requeue_orphaned_jobs(&self.state, &[job]).await; Ok(()) diff --git a/backend/gradient-scheduler/src/history.rs b/backend/gradient-scheduler/src/history.rs index 7fd15bfcd..694ca7f74 100644 --- a/backend/gradient-scheduler/src/history.rs +++ b/backend/gradient-scheduler/src/history.rs @@ -4,7 +4,12 @@ * SPDX-License-Identifier: AGPL-3.0-only */ -use gradient_types::{CDerivationMetric, EDerivationMetric, MDerivationMetric}; +use std::collections::{HashMap, HashSet}; + +use gradient_types::ids::DerivationId; +use gradient_types::{ + CDerivationMetric, CDerivationOutput, EDerivationMetric, EDerivationOutput, MDerivationMetric, +}; use sea_orm::{ColumnTrait, ConnectionTrait, EntityTrait, QueryFilter, QueryOrder, QuerySelect}; /// Older builds are drifting with toolchain and host changes. @@ -27,7 +32,40 @@ pub async fn predict( Err(_) => return gradient_pool::score::HistoryPrediction::default(), }; - summarize(&rows) + gradient_pool::score::HistoryPrediction { + output_nar_size: mean_output_nar_size(&output_nar_sizes(db, &rows).await), + ..summarize(&rows) + } +} + +async fn output_nar_sizes( + db: &impl ConnectionTrait, + rows: &[MDerivationMetric], +) -> Vec<(DerivationId, Option)> { + let derivations: HashSet = rows.iter().map(|r| r.derivation).collect(); + if derivations.is_empty() { + return Vec::new(); + } + + EDerivationOutput::find() + .select_only() + .columns([CDerivationOutput::Derivation, CDerivationOutput::NarSize]) + .filter(CDerivationOutput::Derivation.is_in(derivations)) + .into_tuple() + .all(db) + .await + .unwrap_or_default() +} + +fn mean_output_nar_size(outputs: &[(DerivationId, Option)]) -> Option { + let mut per_derivation: HashMap = HashMap::new(); + for (derivation, size) in outputs { + if let Some(size) = size { + *per_derivation.entry(*derivation).or_default() += size; + } + } + + mean(&per_derivation.into_values().collect::>()) } fn summarize(rows: &[MDerivationMetric]) -> gradient_pool::score::HistoryPrediction { @@ -38,24 +76,26 @@ fn summarize(rows: &[MDerivationMetric]) -> gradient_pool::score::HistoryPredict let samples = rows.len() as u32; let mut peaks: Vec = rows.iter().filter_map(|r| r.peak_ram_mb).collect(); - let predicted_peak_ram_mb = percentile_or_max(&mut peaks, 0.95).max(0) as u64; + let predicted_peak_ram_mb = percentile_or_max(&mut peaks, 0.95); let cpu: Vec = rows.iter().filter_map(|r| r.cpu_time_ms).collect(); - let avg_cpu_time_ms = mean_nonnull(&cpu); + let avg_cpu_time_ms = mean(&cpu); let durations: Vec = rows.iter().filter_map(|r| r.build_time_ms).collect(); - let build_time_ms = mean_nonnull(&durations); + let build_time_ms = mean(&durations); + let uncontended: Vec = rows.iter().filter_map(uncontended_build_time_ms).collect(); + let built_on: Vec = rows + .iter() + .filter(|r| r.build_time_ms.is_some()) + .filter_map(|r| r.cpu_core_score.map(i64::from)) + .collect(); let disk: Vec = rows .iter() .map(|r| r.disk_read_bytes.unwrap_or(0) + r.disk_write_bytes.unwrap_or(0)) .filter(|&b| b > 0) .collect(); - let avg_disk_bytes = if disk.is_empty() { - 0 - } else { - (disk.iter().sum::() / disk.len() as i64).max(0) as u64 - }; + let avg_disk_bytes = mean(&disk); let oom = rows.iter().filter(|r| r.oom_killed).count(); let oom_rate = oom as f32 / samples as f32; @@ -64,30 +104,37 @@ fn summarize(rows: &[MDerivationMetric]) -> gradient_pool::score::HistoryPredict predicted_peak_ram_mb, avg_cpu_time_ms, build_time_ms, + uncontended_build_time_ms: mean(&uncontended), + build_core_score: mean(&built_on).map(|score| score as u32), avg_disk_bytes, + output_nar_size: None, oom_rate, samples, } } -fn mean_nonnull(vals: &[i64]) -> u64 { +fn uncontended_build_time_ms(row: &MDerivationMetric) -> Option { + let others = row.concurrent_builds.unwrap_or(0).max(0) as u32; + row.build_time_ms + .map(|ms| (ms as f64 / gradient_pool::score::contention_factor(others)).round() as i64) +} + +fn mean(vals: &[i64]) -> Option { if vals.is_empty() { - return 0; + return None; } - (vals.iter().sum::() / vals.len() as i64).max(0) as u64 + Some((vals.iter().sum::() / vals.len() as i64).max(0) as u64) } -fn percentile_or_max(values: &mut [i64], p: f64) -> i64 { - if values.is_empty() { - return 0; - } +fn percentile_or_max(values: &mut [i64], p: f64) -> Option { values.sort_unstable(); - if values.len() < 20 { - return values[values.len() - 1]; - } - let idx = ((values.len() as f64 - 1.0) * p).round() as usize; - values[idx] + let idx = if values.len() < 20 { + values.len().checked_sub(1)? + } else { + ((values.len() as f64 - 1.0) * p).round() as usize + }; + Some(values[idx].max(0) as u64) } #[cfg(test)] @@ -109,7 +156,21 @@ mod tests { fn empty_rows_yield_default() { let p = summarize(&[]); assert_eq!(p.samples, 0); - assert_eq!(p.predicted_peak_ram_mb, 0); + assert_eq!(p.predicted_peak_ram_mb, None); + } + + #[test] + fn a_value_no_build_measured_stays_unknown_instead_of_zero() { + let unmeasured = MDerivationMetric { + build_time_ms: Some(300), + ..Default::default() + }; + let p = summarize(&[unmeasured.clone(), unmeasured]); + assert_eq!(p.samples, 2); + assert_eq!(p.build_time_ms, Some(300)); + assert_eq!(p.predicted_peak_ram_mb, None); + assert_eq!(p.avg_cpu_time_ms, None); + assert_eq!(p.avg_disk_bytes, None); } #[test] @@ -121,11 +182,30 @@ mod tests { ]; let p = summarize(&rows); assert_eq!(p.samples, 3); - assert_eq!(p.predicted_peak_ram_mb, 300); - assert_eq!(p.avg_cpu_time_ms, 2000); + assert_eq!(p.predicted_peak_ram_mb, Some(300)); + assert_eq!(p.avg_cpu_time_ms, Some(2000)); assert!((p.oom_rate - (1.0 / 3.0)).abs() < 1e-6); } + #[test] + fn the_build_time_drops_the_slowdown_of_the_builds_beside_it() { + let row = |build_time_ms, concurrent_builds, cpu_core_score| MDerivationMetric { + build_time_ms: Some(build_time_ms), + concurrent_builds, + cpu_core_score, + ..Default::default() + }; + let p = summarize(&[ + row(120_000, Some(5), Some(2_000)), + row(100_000, None, Some(4_000)), + row(80_000, Some(0), None), + ]); + + assert_eq!(p.build_time_ms, Some(100_000)); + assert_eq!(p.uncontended_build_time_ms, Some(93_333)); + assert_eq!(p.build_core_score, Some(3_000)); + } + #[test] fn summarize_aggregates_disk_bytes() { let rows = vec![ @@ -133,17 +213,42 @@ mod tests { metric(Some(200), Some(2000), false), ]; let p = summarize(&rows); - assert_eq!(p.avg_disk_bytes, 50_000_000); + assert_eq!(p.avg_disk_bytes, Some(50_000_000)); + } + + #[test] + fn the_output_size_is_the_mean_over_builds_of_their_summed_outputs() { + let (one, two) = (DerivationId::now_v7(), DerivationId::now_v7()); + let outputs = [ + (one, Some(100)), + (one, Some(300)), + (two, Some(200)), + (two, None), + ]; + assert_eq!(mean_output_nar_size(&outputs), Some(300)); + assert_eq!(mean_output_nar_size(&[(one, None)]), None); } #[tokio::test] async fn predict_reads_the_latest_rows_of_the_same_pname_and_architecture() { + let derivation = DerivationId::now_v7(); + let output: std::collections::BTreeMap = [ + ("derivation".to_owned(), derivation.into_inner().into()), + ("nar_size".to_owned(), 4_096_i64.into()), + ] + .into_iter() + .collect(); let db = sea_orm::MockDatabase::new(sea_orm::DatabaseBackend::Postgres) - .append_query_results([vec![metric(Some(100), Some(1000), false)]]) + .append_query_results([vec![MDerivationMetric { + derivation, + ..metric(Some(100), Some(1000), false) + }]]) + .append_query_results([vec![output]]) .into_connection(); let p = predict(&db, "hello", "x86_64-linux").await; assert_eq!(p.samples, 1); + assert_eq!(p.output_nar_size, Some(4_096)); let sql: Vec = db .into_transaction_log() @@ -151,9 +256,13 @@ mod tests { .flat_map(|t| t.statements()) .map(|stmt| stmt.sql.clone()) .collect(); - let [sql] = sql.as_slice() else { - panic!("one statement expected: {sql:?}") + let [sql, outputs_sql] = sql.as_slice() else { + panic!("two statements expected: {sql:?}") }; + assert!( + outputs_sql.contains(r#""derivation_output"."derivation" IN ($1)"#), + "{outputs_sql}" + ); assert!(sql.contains(r#""pname" = $1"#), "{sql}"); assert!(sql.contains(r#""architecture" = $2"#), "{sql}"); assert!(!sql.contains(r#""closure_size" >="#), "{sql}"); diff --git a/backend/gradient-scheduler/src/instance.rs b/backend/gradient-scheduler/src/instance.rs index b3f8e15d8..6e0a6a875 100644 --- a/backend/gradient-scheduler/src/instance.rs +++ b/backend/gradient-scheduler/src/instance.rs @@ -16,6 +16,10 @@ pub struct InstanceCounts { pub total_workers: u32, pub idle_workers: u32, pub cpu_core_score_mean: Option, + pub upload_speed_mean_mbps: Option, + pub download_speed_mean_mbps: Option, + pub download_slots: u32, + pub upload_slots: u32, } /// A window with no data must stay distinguishable from a measured zero. @@ -41,9 +45,6 @@ struct MetricRow { disk_5m: Option, disk_1h: Option, disk_24h: Option, - network_5m: Option, - network_1h: Option, - network_24h: Option, build_time_5m: Option, build_time_1h: Option, build_time_24h: Option, @@ -89,9 +90,6 @@ gradient_db::sql! { (AVG(disk_read_bytes + disk_write_bytes) FILTER (WHERE created_at >= $1))::float8 AS disk_5m, (AVG(disk_read_bytes + disk_write_bytes) FILTER (WHERE created_at >= $2))::float8 AS disk_1h, (AVG(disk_read_bytes + disk_write_bytes) FILTER (WHERE created_at >= $3))::float8 AS disk_24h, - (AVG(peak_network_mbps) FILTER (WHERE created_at >= $1))::float8 AS network_5m, - (AVG(peak_network_mbps) FILTER (WHERE created_at >= $2))::float8 AS network_1h, - (AVG(peak_network_mbps) FILTER (WHERE created_at >= $3))::float8 AS network_24h, (AVG(build_time_ms) FILTER (WHERE created_at >= $1))::float8 AS build_time_5m, (AVG(build_time_ms) FILTER (WHERE created_at >= $2))::float8 AS build_time_1h, (AVG(build_time_ms) FILTER (WHERE created_at >= $3))::float8 AS build_time_24h, @@ -139,6 +137,180 @@ fn assignment_windows_sql() -> String { ) } +#[derive(Debug, Default, FromQueryResult)] +struct StorageRow { + read_mbps: Option, + write_mbps: Option, +} + +gradient_db::sql! { + STORAGE_PEAK_THROUGHPUT = r#" + WITH span AS ( + SELECT p.phase, + d.finished_at - make_interval(secs => (d.worker_elapsed_ms - p.start_ms) / 1000.0) AS started, + d.finished_at - make_interval(secs => (d.worker_elapsed_ms - p.end_ms) / 1000.0) AS ended, + p.bytes * 8.0 / (1000.0 * (p.end_ms - p.start_ms)) AS mbps + FROM dispatched_job_phase p + JOIN dispatched_job d ON d.id = p.dispatched_job + WHERE p.phase IN ($2, $3) AND p.created_at >= $1 + AND p.bytes >= 1048576 AND p.end_ms - p.start_ms >= 1000 + AND d.finished_at IS NOT NULL AND d.worker_elapsed_ms IS NOT NULL + ), + change AS ( + SELECT phase, started AS at, mbps AS delta FROM span + UNION ALL + SELECT phase, ended, -mbps FROM span + ), + running AS ( + SELECT phase, SUM(delta) OVER (PARTITION BY phase ORDER BY at, delta ROWS UNBOUNDED PRECEDING) AS mbps + FROM change + ) + SELECT (MAX(mbps) FILTER (WHERE phase = $2))::float8 AS read_mbps, + (MAX(mbps) FILTER (WHERE phase = $3))::float8 AS write_mbps + FROM running + "#, + params = [Now, Int(16), Int(12)]; +} + +#[derive(Debug, Default, FromQueryResult)] +struct CompressionRow { + ratio: Option, +} + +gradient_db::sql! { + STORED_TO_NAR_RATIO = r#" + SELECT (SUM(push.bytes)::float8 / NULLIF(SUM(c.bytes), 0))::float8 AS ratio + FROM ( + SELECT dispatched_job, parent_seq, SUM(bytes) AS bytes + FROM dispatched_job_phase + WHERE phase = $2 AND created_at >= $1 AND parent_seq IS NOT NULL + GROUP BY dispatched_job, parent_seq + ) push + JOIN dispatched_job_phase c + ON c.dispatched_job = push.dispatched_job AND c.seq = push.parent_seq + WHERE c.phase = $3 AND c.bytes > 0 + "#, + params = [Now, Int(12), Int(11)]; +} + +const STORAGE_WINDOW_HOURS: i64 = 1; + +#[derive(Debug, Default, FromQueryResult)] +struct PathFitRow { + n: f64, + sb: Option, + sp: Option, + sy: Option, + sbb: Option, + spp: Option, + sbp: Option, + sby: Option, + spy: Option, +} + +gradient_db::sql! { + PREFETCH_TIME_FIT = r#" + SELECT COUNT(*)::float8 AS n, + SUM(b)::float8 AS sb, SUM(p)::float8 AS sp, SUM(y)::float8 AS sy, + SUM(b * b)::float8 AS sbb, SUM(p * p)::float8 AS spp, SUM(b * p)::float8 AS sbp, + SUM(b * y)::float8 AS sby, SUM(p * y)::float8 AS spy + FROM ( + SELECT (end_ms - start_ms) / 1000.0 AS y, bytes / 1000000.0 AS b, paths::numeric AS p + FROM dispatched_job_phase + WHERE phase = $2 AND created_at >= $1 + ) span + "#, + params = [Now, Int(8)]; +} + +const PER_PATH_WINDOW_HOURS: i64 = 24; +const PER_PATH_MIN_SPANS: f64 = 100.0; + +fn det3(m: [[f64; 3]; 3]) -> f64 { + m[0][0] * (m[1][1] * m[2][2] - m[1][2] * m[2][1]) + - m[0][1] * (m[1][0] * m[2][2] - m[1][2] * m[2][0]) + + m[0][2] * (m[1][0] * m[2][1] - m[1][1] * m[2][0]) +} + +fn per_path_secs(row: &PathFitRow) -> Option { + if row.n < PER_PATH_MIN_SPANS { + return None; + } + + let normal = [ + [row.n, row.sb?, row.sp?], + [row.sb?, row.sbb?, row.sbp?], + [row.sp?, row.sbp?, row.spp?], + ]; + let det = det3(normal); + if det.abs() < f64::EPSILON { + return None; + } + + let mut paths_column = normal; + for (row_index, value) in [row.sy?, row.sby?, row.spy?].into_iter().enumerate() { + paths_column[row_index][2] = value; + } + + Some(det3(paths_column) / det).filter(|secs| secs.is_finite() && *secs > 0.0) +} + +async fn learned_per_path_secs( + db: &impl ConnectionTrait, + now: chrono::NaiveDateTime, +) -> Option { + use gradient_wire::types::JobPhase; + + let since = now - chrono::Duration::hours(PER_PATH_WINDOW_HOURS); + let row = PathFitRow::find_by_statement( + PREFETCH_TIME_FIT.bind([since.into(), JobPhase::Prefetch.as_i16().into()]), + ) + .one(db) + .await + .unwrap_or_else(|e| { + error!(error = %e, "instance metrics: per-path fit query failed"); + None + })?; + + per_path_secs(&row) +} + +async fn storage_throughput( + db: &impl ConnectionTrait, + now: chrono::NaiveDateTime, +) -> (StorageRow, Option) { + use gradient_wire::types::JobPhase; + + let since = now - chrono::Duration::hours(STORAGE_WINDOW_HOURS); + let peak = StorageRow::find_by_statement(STORAGE_PEAK_THROUGHPUT.bind([ + since.into(), + JobPhase::NarFetch.as_i16().into(), + JobPhase::NarPush.as_i16().into(), + ])) + .one(db) + .await + .unwrap_or_else(|e| { + error!(error = %e, "instance metrics: storage throughput query failed"); + None + }) + .unwrap_or_default(); + + let ratio = CompressionRow::find_by_statement(STORED_TO_NAR_RATIO.bind([ + since.into(), + JobPhase::NarPush.as_i16().into(), + JobPhase::Compress.as_i16().into(), + ])) + .one(db) + .await + .unwrap_or_else(|e| { + error!(error = %e, "instance metrics: compression ratio query failed"); + None + }) + .and_then(|row| row.ratio); + + (peak, ratio) +} + pub async fn compute_instance_context( db: &impl ConnectionTrait, counts: InstanceCounts, @@ -179,6 +351,9 @@ pub async fn compute_instance_context( } }; + let (storage, compression_ratio) = storage_throughput(db, now).await; + let per_path_secs = learned_per_path_secs(db, now).await; + gradient_pool::score::InstanceContext { wait_secs: windowed( assignment_id.wait_5m, @@ -194,7 +369,6 @@ pub async fn compute_instance_context( cpu_time_ms: windowed(metric.cpu_time_5m, metric.cpu_time_1h, metric.cpu_time_24h), avg_cpu_pct: windowed(metric.cpu_pct_5m, metric.cpu_pct_1h, metric.cpu_pct_24h), disk_bytes: windowed(metric.disk_5m, metric.disk_1h, metric.disk_24h), - network_mbps: windowed(metric.network_5m, metric.network_1h, metric.network_24h), oom_rate: windowed(metric.oom_5m, metric.oom_1h, metric.oom_24h), closure_size: windowed(metric.closure_5m, metric.closure_1h, metric.closure_24h), nar_size_mb: windowed( @@ -222,6 +396,15 @@ pub async fn compute_instance_context( total_workers: counts.total_workers, idle_workers: counts.idle_workers, cpu_core_score_mean: counts.cpu_core_score_mean, + upload_speed_mean_mbps: counts.upload_speed_mean_mbps, + download_speed_mean_mbps: counts.download_speed_mean_mbps, + storage_read_mbps: storage.read_mbps, + storage_write_mbps: storage.write_mbps, + compression_ratio, + per_path_secs, + download_slots: counts.download_slots, + upload_slots: counts.upload_slots, + ..Default::default() } } @@ -245,35 +428,104 @@ gradient_db::sql! { params = [Now]; } +#[derive(Debug, Default, FromQueryResult)] +struct EvalDurationRow { + task: TaskId, + elapsed_ms: f64, + samples: i64, +} + +gradient_db::sql! { + EVAL_HISTORY_DURATION = r#" + SELECT task, + AVG(worker_elapsed_ms)::float8 AS elapsed_ms, + COUNT(*)::bigint AS samples + FROM dispatched_job + WHERE kind = $2 AND outcome = $3 AND dispatched_at >= $1 + AND task IS NOT NULL AND worker_elapsed_ms IS NOT NULL + GROUP BY task + "#, + params = [Now, Int(0), Int(0)]; +} + +const EVAL_DURATION_WINDOW_DAYS: i64 = 7; + +#[derive(Debug, Default)] +pub struct EvalHistory { + tasks: HashMap, + fleet_elapsed_ms: Option, +} + +impl EvalHistory { + pub fn for_task(&self, task: TaskId) -> gradient_pool::score::HistoryPrediction { + let mut history = self.tasks.get(&task).copied().unwrap_or_default(); + if history.uncontended_build_time_ms.is_none() { + history.uncontended_build_time_ms = self.fleet_elapsed_ms; + } + + history + } + + fn from_rows(ram: Vec, durations: Vec) -> Self { + let mut tasks: HashMap = ram + .into_iter() + .map(|r| { + ( + r.task, + gradient_pool::score::HistoryPrediction { + predicted_peak_ram_mb: Some(r.p95_ram.max(0.0) as u64), + samples: r.samples.max(0) as u32, + ..Default::default() + }, + ) + }) + .collect(); + + let (mut weighted_ms, mut runs) = (0.0, 0.0); + for row in durations { + let elapsed_ms = row.elapsed_ms.max(0.0) as u64; + let history = tasks.entry(row.task).or_default(); + history.build_time_ms = Some(elapsed_ms); + history.uncontended_build_time_ms = Some(elapsed_ms); + weighted_ms += row.elapsed_ms.max(0.0) * row.samples as f64; + runs += row.samples as f64; + } + + Self { + tasks, + fleet_elapsed_ms: (runs > 0.0).then(|| (weighted_ms / runs) as u64), + } + } +} + pub async fn compute_eval_history( db: &impl ConnectionTrait, now: chrono::NaiveDateTime, -) -> HashMap { - let since = now - chrono::Duration::hours(24); +) -> EvalHistory { + use gradient_entity::dispatched_job::{DispatchedJobKind, DispatchedJobOutcome}; - let rows = match EvalHistoryRow::find_by_statement(EVAL_HISTORY_P95_RAM.bind([since.into()])) + let since = now - chrono::Duration::hours(24); + let ram = EvalHistoryRow::find_by_statement(EVAL_HISTORY_P95_RAM.bind([since.into()])) .all(db) .await - { - Ok(r) => r, - Err(e) => { + .unwrap_or_else(|e| { error!(error = %e, "eval history query failed"); - return HashMap::new(); - } - }; + Vec::new() + }); - rows.into_iter() - .map(|r| { - ( - r.task, - gradient_pool::score::HistoryPrediction { - predicted_peak_ram_mb: r.p95_ram.max(0.0) as u64, - samples: r.samples.max(0) as u32, - ..Default::default() - }, - ) - }) - .collect() + let durations = EvalDurationRow::find_by_statement(EVAL_HISTORY_DURATION.bind([ + (now - chrono::Duration::days(EVAL_DURATION_WINDOW_DAYS)).into(), + i16::from(DispatchedJobKind::Eval).into(), + i16::from(DispatchedJobOutcome::Completed).into(), + ])) + .all(db) + .await + .unwrap_or_else(|e| { + error!(error = %e, "eval duration query failed"); + Vec::new() + }); + + EvalHistory::from_rows(ram, durations) } #[cfg(test)] @@ -298,9 +550,6 @@ mod tests { f("disk_5m", 4.0), f("disk_1h", 5.0), f("disk_24h", 6.0), - f("network_5m", 7.0), - f("network_1h", 8.0), - f("network_24h", 9.0), f("build_time_5m", 11.0), f("build_time_1h", 12.0), f("build_time_24h", 13.0), @@ -333,9 +582,17 @@ mod tests { .into_iter() .collect(); + let storage: BTreeMap = [f("read_mbps", 1_200.0), f("write_mbps", 900.0)] + .into_iter() + .collect(); + let compression: BTreeMap = [f("ratio", 0.4)].into_iter().collect(); + let db = MockDatabase::new(DatabaseBackend::Postgres) .append_query_results([vec![metric]]) .append_query_results([vec![assignment_id]]) + .append_query_results([vec![storage]]) + .append_query_results([vec![compression]]) + .append_query_results([vec![BTreeMap::from([f("n", 0.0)])]]) .into_connection(); let counts = InstanceCounts { @@ -344,6 +601,10 @@ mod tests { total_workers: 5, idle_workers: 1, cpu_core_score_mean: None, + upload_speed_mean_mbps: Some(400.0), + download_speed_mean_mbps: None, + download_slots: 16, + upload_slots: 8, }; let ic = compute_instance_context(&db, counts, gradient_types::now()).await; @@ -360,6 +621,40 @@ mod tests { assert_eq!(ic.pending_builds, 3); assert_eq!(ic.total_workers, 5); assert_eq!(ic.idle_workers, 1); + assert_eq!(ic.upload_speed_mean_mbps, Some(400.0)); + assert_eq!(ic.storage_read_mbps, Some(1_200.0)); + assert_eq!(ic.storage_write_mbps, Some(900.0)); + assert_eq!(ic.compression_ratio, Some(0.4)); + } + + #[test] + fn the_path_coefficient_is_the_time_beyond_the_bytes() { + let spans: Vec<(f64, f64)> = (0..200) + .map(|i| (f64::from(i % 7) * 30.0, f64::from(i % 11))) + .collect(); + let mut row = PathFitRow { + n: spans.len() as f64, + ..Default::default() + }; + let add = |sum: &mut Option, value: f64| *sum = Some(sum.unwrap_or(0.0) + value); + for (b, p) in spans { + let y = 2.0 + 0.04 * b + 0.25 * p; + add(&mut row.sb, b); + add(&mut row.sp, p); + add(&mut row.sy, y); + add(&mut row.sbb, b * b); + add(&mut row.spp, p * p); + add(&mut row.sbp, b * p); + add(&mut row.sby, b * y); + add(&mut row.spy, p * y); + } + + assert!((per_path_secs(&row).unwrap() - 0.25).abs() < 1e-9); + assert_eq!( + per_path_secs(&PathFitRow { n: 10.0, ..row }), + None, + "too few spans" + ); } #[tokio::test] @@ -375,11 +670,42 @@ mod tests { let db = MockDatabase::new(DatabaseBackend::Postgres) .append_query_results([vec![row]]) + .append_query_results([Vec::>::new()]) .into_connection(); - let history = compute_eval_history(&db, gradient_types::now()).await; - let h = history.get(&pid).expect("task present"); - assert_eq!(h.predicted_peak_ram_mb, 42_000); + let h = compute_eval_history(&db, gradient_types::now()) + .await + .for_task(pid); + assert_eq!(h.predicted_peak_ram_mb, Some(42_000)); assert_eq!(h.samples, 7); } + + #[test] + fn a_task_without_eval_runs_takes_the_mean_of_every_run() { + let (seen, unseen) = (TaskId::now_v7(), TaskId::now_v7()); + let history = EvalHistory::from_rows( + Vec::new(), + vec![ + EvalDurationRow { + task: seen, + elapsed_ms: 10_000.0, + samples: 3, + }, + EvalDurationRow { + task: TaskId::now_v7(), + elapsed_ms: 50_000.0, + samples: 1, + }, + ], + ); + + assert_eq!( + history.for_task(seen).uncontended_build_time_ms, + Some(10_000) + ); + assert_eq!( + history.for_task(unseen).uncontended_build_time_ms, + Some(20_000) + ); + } } diff --git a/backend/gradient-scheduler/src/job_handlers/build_status.rs b/backend/gradient-scheduler/src/job_handlers/build_status.rs index 9b045a0db..b744d5e60 100644 --- a/backend/gradient-scheduler/src/job_handlers/build_status.rs +++ b/backend/gradient-scheduler/src/job_handlers/build_status.rs @@ -201,6 +201,7 @@ impl Scheduler { error: &str, kind: BuildFailureKind, missing_paths: &[String], + metrics: Option, ) -> Result<()> { let worker = worker_id.to_owned(); let released = self @@ -225,10 +226,15 @@ impl Scheduler { .await; } - self.settle_failed(job, &failure).await + self.settle_failed(job, &failure, metrics).await } - pub(crate) async fn settle_failed(&self, job: PendingJob, failure: &Failure) -> Result<()> { + pub(crate) async fn settle_failed( + &self, + job: PendingJob, + failure: &Failure, + metrics: Option, + ) -> Result<()> { match job { PendingJob::Eval(j) => { let r = self @@ -254,6 +260,7 @@ impl Scheduler { log_banner: gradient_sources::strip_nix_log_tail(&failure.error), kind: failure.kind, missing_paths: failure.missing_paths.clone(), + metrics, }) .await .map(|_| ()), diff --git a/backend/gradient-scheduler/src/job_handlers/cluster.rs b/backend/gradient-scheduler/src/job_handlers/cluster.rs index a723cc8a8..b240b4761 100644 --- a/backend/gradient-scheduler/src/job_handlers/cluster.rs +++ b/backend/gradient-scheduler/src/job_handlers/cluster.rs @@ -491,7 +491,7 @@ impl Scheduler { missing_paths: Vec::new(), }; for job in waiting.members.into_iter().filter_map(|m| m.job) { - if let Err(e) = self.settle_failed(job, &failure).await { + if let Err(e) = self.settle_failed(job, &failure, None).await { warn!(error = %e, %cluster, "settling a member of an aborted cluster failed"); } } diff --git a/backend/gradient-scheduler/src/job_handlers/queue.rs b/backend/gradient-scheduler/src/job_handlers/queue.rs index 818d909d9..6c0591f9f 100644 --- a/backend/gradient-scheduler/src/job_handlers/queue.rs +++ b/backend/gradient-scheduler/src/job_handlers/queue.rs @@ -62,7 +62,7 @@ impl Scheduler { kind: gradient_wire::types::BuildFailureKind::Aborted, missing_paths: Vec::new(), }; - self.settle_failed(job, &failure).await + self.settle_failed(job, &failure, None).await } } } diff --git a/backend/gradient-scheduler/src/jobs.rs b/backend/gradient-scheduler/src/jobs.rs index 33333d1be..28a5489e7 100644 --- a/backend/gradient-scheduler/src/jobs.rs +++ b/backend/gradient-scheduler/src/jobs.rs @@ -809,11 +809,12 @@ impl JobTracker { if policy.uses_project_work_share() { for ActiveJob { job, .. } in self.active.values() { if let PendingJob::Build(b) = job { - let w = if b.history.build_time_ms > 0 { - b.history.build_time_ms as f64 - } else { - (if b.prefer_local_build { 0.5 } else { 1.0 }) - * instance.build_time_ms.w1h.unwrap_or(0.0) + let w = match b.history.build_time_ms.filter(|ms| *ms > 0) { + Some(ms) => ms as f64, + None => { + (if b.prefer_local_build { 0.5 } else { 1.0 }) + * instance.build_time_ms.w1h.unwrap_or(0.0) + } }; *by_project.entry(b.project_id).or_default() += w; total += w; diff --git a/backend/gradient-scheduler/src/lib.rs b/backend/gradient-scheduler/src/lib.rs index 56a63e327..546866c3b 100644 --- a/backend/gradient-scheduler/src/lib.rs +++ b/backend/gradient-scheduler/src/lib.rs @@ -57,14 +57,7 @@ pub struct Scheduler { pub(crate) kick_gen: Arc, pub(crate) policy: Arc, pub(crate) instance: Arc>, - pub(crate) eval_history: Arc< - arc_swap::ArcSwap< - std::collections::HashMap< - gradient_types::ids::TaskId, - gradient_pool::score::HistoryPrediction, - >, - >, - >, + pub(crate) eval_history: Arc>, pub draining: Arc, pub(crate) assessments: Arc>, pub(crate) cluster_wake: Arc, @@ -93,7 +86,7 @@ impl Scheduler { gradient_pool::score::InstanceContext::default(), )), eval_history: Arc::new(arc_swap::ArcSwap::from_pointee( - std::collections::HashMap::new(), + crate::instance::EvalHistory::default(), )), draining: Arc::new(AtomicBool::new(false)), assessments: Arc::default(), diff --git a/backend/gradient-scheduler/src/loops/background.rs b/backend/gradient-scheduler/src/loops/background.rs index 4b6844b9d..2abfb2030 100644 --- a/backend/gradient-scheduler/src/loops/background.rs +++ b/backend/gradient-scheduler/src/loops/background.rs @@ -90,6 +90,10 @@ pub(super) async fn instance_metrics_pass(scheduler: Arc) -> anyhow:: total_workers: c.workers as u32, idle_workers: c.idle_workers as u32, cpu_core_score_mean: c.cpu_core_score_mean, + upload_speed_mean_mbps: c.upload_speed_mean_mbps, + download_speed_mean_mbps: c.download_speed_mean_mbps, + download_slots: scheduler.state.config.nar.max_concurrent_downloads as u32, + upload_slots: scheduler.state.config.upload.concurrency as u32, }; let ctx = crate::instance::compute_instance_context( &scheduler.state.worker_db, diff --git a/backend/gradient-scheduler/src/loops/eval.rs b/backend/gradient-scheduler/src/loops/eval.rs index cea4fda05..4bb2c6eff 100644 --- a/backend/gradient-scheduler/src/loops/eval.rs +++ b/backend/gradient-scheduler/src/loops/eval.rs @@ -117,7 +117,7 @@ pub(crate) async fn assign_queued_evals(scheduler: &Scheduler) -> anyhow::Result continue; }; - let history = eval_history.get(&task_id).copied().unwrap_or_default(); + let history = eval_history.for_task(task_id); let pending = PendingEvalJob { evaluation_id: eval.id, diff --git a/backend/gradient-scheduler/src/views.rs b/backend/gradient-scheduler/src/views.rs index b1e636305..db4b146fe 100644 --- a/backend/gradient-scheduler/src/views.rs +++ b/backend/gradient-scheduler/src/views.rs @@ -21,7 +21,9 @@ pub struct WorkerContextView { pub ram_free_mb: Option, pub cpu_usage_pct: Option, pub disk_speed_mbps: Option, - pub network_speed_mbps: Option, + pub upload_speed_mbps: Option, + pub download_speed_mbps: Option, + pub running_builds: u32, } impl WorkerContextView { @@ -37,17 +39,22 @@ impl WorkerContextView { ram_free_mb: m.ram_free_mb, cpu_usage_pct: m.cpu_usage_pct, disk_speed_mbps: m.disk_speed_mbps, - network_speed_mbps: m.network_speed_mbps, + upload_speed_mbps: m.upload_speed_mbps, + download_speed_mbps: m.download_speed_mbps, + running_builds: m.running_builds, } } } #[derive(Serialize)] pub struct HistoryView { - pub peak_ram_mb: u64, - pub avg_cpu_time_ms: u64, - pub build_time_ms: u64, - pub avg_disk_bytes: u64, + pub peak_ram_mb: Option, + pub avg_cpu_time_ms: Option, + pub build_time_ms: Option, + pub uncontended_build_time_ms: Option, + pub build_core_score: Option, + pub avg_disk_bytes: Option, + pub output_nar_size: Option, pub oom_rate: f32, pub samples: u32, } @@ -58,7 +65,10 @@ impl From<&HistoryPrediction> for HistoryView { peak_ram_mb: h.predicted_peak_ram_mb, avg_cpu_time_ms: h.avg_cpu_time_ms, build_time_ms: h.build_time_ms, + uncontended_build_time_ms: h.uncontended_build_time_ms, + build_core_score: h.build_core_score, avg_disk_bytes: h.avg_disk_bytes, + output_nar_size: h.output_nar_size, oom_rate: h.oom_rate, samples: h.samples, } diff --git a/backend/gradient-scheduler/src/worker_lifecycle.rs b/backend/gradient-scheduler/src/worker_lifecycle.rs index ed9ea5a3e..9f5ea000d 100644 --- a/backend/gradient-scheduler/src/worker_lifecycle.rs +++ b/backend/gradient-scheduler/src/worker_lifecycle.rs @@ -16,7 +16,7 @@ use sea_orm::{ use tracing::{debug, info, warn}; use gradient_types::ids::ProjectId; -use gradient_wire::types::GradientCapabilities; +use gradient_wire::types::{BuildStage, GradientCapabilities}; use crate::Scheduler; use crate::actor::{Registered, Registration, SchedulerMsg, WorkerCapabilities, WorkerMetrics}; @@ -35,7 +35,8 @@ pub(crate) async fn record_worker_sample( ram_free_mb: info.ram_free_mb.map(|v| v as i64), ram_total_mb: Some(info.ram_total_mb as i64), disk_speed_mbps: info.disk_speed_mbps, - network_speed_mbps: info.network_speed_mbps, + upload_speed_mbps: info.upload_speed_mbps, + download_speed_mbps: info.download_speed_mbps, assigned_jobs: info.assigned_job_count as i32, max_concurrent_builds: info.max_concurrent_builds as i32, state: info.draining.into(), @@ -252,6 +253,16 @@ impl Scheduler { .await; } + pub async fn enter_build_stage(&self, worker_id: &str, job_id: &str, stage: BuildStage) { + let _ = self + .cast(SchedulerMsg::EnterStage { + worker: worker_id.to_owned(), + job_id: job_id.to_owned(), + stage, + }) + .await; + } + pub async fn unregister_worker(&self, worker_id: &str) { self.close_worker_connection(worker_id).await; let worker = worker_id.to_owned(); diff --git a/backend/gradient-ssh/src/daemon_build.rs b/backend/gradient-ssh/src/daemon_build.rs index f8c0eda9c..22a596131 100644 --- a/backend/gradient-ssh/src/daemon_build.rs +++ b/backend/gradient-ssh/src/daemon_build.rs @@ -164,6 +164,10 @@ fn build_result(inner: BuildResultInner) -> BuildResult { stop_time: 0, cpu_user: None, cpu_system: None, + memory_peak: None, + io_read_bytes: None, + io_write_bytes: None, + oom_kills: None, } } diff --git a/backend/gradient-storage/src/nar.rs b/backend/gradient-storage/src/nar.rs index 0eca83e9c..9fc6366f7 100644 --- a/backend/gradient-storage/src/nar.rs +++ b/backend/gradient-storage/src/nar.rs @@ -11,7 +11,7 @@ use anyhow::{Context, Result}; use bytes::{Bytes, BytesMut}; use futures::StreamExt as _; use futures::stream::BoxStream; -use gradient_util::telemetry::STATS; +use gradient_util::telemetry::{GAUGES, STATS}; use gradient_wire::constants::{BULK_CHUNK_SIZE, MULTIPART_NAR_BYTES, PRESIGN_TTL}; use object_store::{ClientOptions, ObjectStore, ObjectStoreExt as _, PutPayload, path::Path}; pub use object_store::{MultipartUpload, WriteMultipart}; @@ -123,7 +123,7 @@ impl NarStore { let store = object_store::local::LocalFileSystem::new_with_prefix(base_path) .context("Failed to create local NAR storage")?; Ok(Self { - inner: Arc::new(TimedStore::new(Arc::new(store), &STATS)), + inner: Arc::new(TimedStore::new(Arc::new(store), &STATS, &GAUGES)), prefix: String::new(), local_base: Some(base_path.to_string()), s3_signer: None, @@ -183,6 +183,7 @@ impl NarStore { inner: Arc::new(TimedStore::new( Arc::clone(&store) as Arc, &STATS, + &GAUGES, )), prefix: crate::layout::normalize_prefix(prefix), local_base: None, @@ -256,7 +257,8 @@ impl NarStore { .await; guard.finish_with(opened.is_ok()); - return opened.map(|found| found.map(|(size, s)| (size, watch_stream(s, &STATS)))); + return opened + .map(|found| found.map(|(size, s)| (size, watch_stream(s, &STATS, &GAUGES)))); } let Some(offset) = offset else { diff --git a/backend/gradient-storage/src/timed.rs b/backend/gradient-storage/src/timed.rs index af5cf5588..495448751 100644 --- a/backend/gradient-storage/src/timed.rs +++ b/backend/gradient-storage/src/timed.rs @@ -16,7 +16,7 @@ use async_trait::async_trait; use bytes::Bytes; use futures::StreamExt; use futures::stream::BoxStream; -use gradient_util::telemetry::{MinuteStats, metric}; +use gradient_util::telemetry::{Gauges, MinuteStats, metric}; use object_store::path::Path; use object_store::{ CopyOptions, GetOptions, GetResult, GetResultPayload, ListResult, MultipartUpload, ObjectMeta, @@ -27,11 +27,20 @@ use object_store::{ pub(crate) struct TimedStore { inner: Arc, stats: &'static MinuteStats, + gauges: &'static Gauges, } impl TimedStore { - pub(crate) fn new(inner: Arc, stats: &'static MinuteStats) -> Self { - Self { inner, stats } + pub(crate) fn new( + inner: Arc, + stats: &'static MinuteStats, + gauges: &'static Gauges, + ) -> Self { + Self { + inner, + stats, + gauges, + } } } @@ -99,11 +108,15 @@ impl Drop for OpGuard { pub(crate) fn watch_stream( stream: BoxStream<'static, Result>, stats: &'static MinuteStats, + gauges: &'static Gauges, ) -> BoxStream<'static, Result> { + let reading = gauges.storage_reads.enter(); stream .inspect(move |item| { - if item.is_err() { - stats.record(metric::STORAGE_OP_ERRORS, "read/error", 1.0); + let _reading = &reading; + match item { + Ok(chunk) => stats.record(metric::STORAGE_READ_BYTES, "", chunk.len() as f64), + Err(_) => stats.record(metric::STORAGE_OP_ERRORS, "read/error", 1.0), } }) .boxed() @@ -112,6 +125,7 @@ pub(crate) fn watch_stream( struct TimedUpload { inner: Box, stats: &'static MinuteStats, + gauges: &'static Gauges, } impl fmt::Debug for TimedUpload { @@ -124,8 +138,17 @@ impl fmt::Debug for TimedUpload { impl MultipartUpload for TimedUpload { fn put_part(&mut self, data: PutPayload) -> UploadPart { let guard = OpGuard::start(self.stats, "put_part"); + self.stats.record( + metric::STORAGE_WRITE_BYTES, + "", + data.content_length() as f64, + ); + let writing = self.gauges.storage_writes.enter(); let part = self.inner.put_part(data); - Box::pin(async move { guard.finish(part.await) }) + Box::pin(async move { + let _writing = writing; + guard.finish(part.await) + }) } async fn complete(&mut self) -> Result { @@ -147,6 +170,12 @@ impl ObjectStore for TimedStore { opts: PutOptions, ) -> Result { let guard = OpGuard::start(self.stats, "put"); + self.stats.record( + metric::STORAGE_WRITE_BYTES, + "", + payload.content_length() as f64, + ); + let _writing = self.gauges.storage_writes.enter(); guard.finish(self.inner.put_opts(location, payload, opts).await) } @@ -160,6 +189,7 @@ impl ObjectStore for TimedStore { Ok(Box::new(TimedUpload { inner, stats: self.stats, + gauges: self.gauges, })) } @@ -169,7 +199,8 @@ impl ObjectStore for TimedStore { let mut result = guard.finish(self.inner.get_opts(location, options).await)?; if let GetResultPayload::Stream(stream) = result.payload { - result.payload = GetResultPayload::Stream(watch_stream(stream, self.stats)); + result.payload = + GetResultPayload::Stream(watch_stream(stream, self.stats, self.gauges)); } Ok(result) @@ -233,7 +264,20 @@ mod tests { fn timed() -> (TimedStore, &'static MinuteStats) { let stats: &'static MinuteStats = Box::leak(Box::default()); - (TimedStore::new(Arc::new(InMemory::new()), stats), stats) + let gauges: &'static Gauges = Box::leak(Box::default()); + ( + TimedStore::new(Arc::new(InMemory::new()), stats, gauges), + stats, + ) + } + + fn sum(stats: &MinuteStats, metric: &str) -> f64 { + stats + .snapshot() + .iter() + .filter(|(k, _)| k.metric == metric) + .map(|(_, a)| a.sum) + .sum() } fn count(stats: &MinuteStats, metric: &str, label: &str) -> i64 { @@ -261,6 +305,27 @@ mod tests { assert_eq!(count(stats, metric::STORAGE_OP_MS, "get"), 1); } + #[tokio::test] + async fn written_and_streamed_bytes_are_counted() { + let (store, stats) = timed(); + let path = Path::from("a"); + store + .put(&path, PutPayload::from_static(b"nar")) + .await + .expect("put"); + let read: Vec<_> = store + .get(&path) + .await + .expect("get") + .into_stream() + .collect() + .await; + + assert_eq!(read.len(), 1); + assert_eq!(sum(stats, metric::STORAGE_WRITE_BYTES), 3.0); + assert_eq!(sum(stats, metric::STORAGE_READ_BYTES), 3.0); + } + #[tokio::test] async fn a_failed_call_counts_as_error() { let (store, stats) = timed(); @@ -305,7 +370,7 @@ mod tests { })]) .boxed(); - let mut stream = watch_stream(failing, stats); + let mut stream = watch_stream(failing, stats, Box::leak(Box::default())); assert!(stream.next().await.expect("item").is_err()); assert_eq!(count(stats, metric::STORAGE_OP_ERRORS, "read/error"), 1); } diff --git a/backend/gradient-test-support/src/cache_fixture.rs b/backend/gradient-test-support/src/cache_fixture.rs index 67565911e..eaa77d401 100644 --- a/backend/gradient-test-support/src/cache_fixture.rs +++ b/backend/gradient-test-support/src/cache_fixture.rs @@ -155,6 +155,7 @@ pub async fn public_cache_with_narinfo() -> Arc { startable_set: Default::default(), graph: gradient_core::Graph::stub(), upstream_query: std::sync::Arc::new(tokio::sync::Semaphore::new(32)), + nar_downloads: std::sync::Arc::new(tokio::sync::Semaphore::new(16)), upload_admission: gradient_storage::admission::UploadAdmission::new( gradient_storage::admission::Limits { concurrency: 16, @@ -218,6 +219,7 @@ pub async fn public_cache_state() -> Arc { startable_set: Default::default(), graph: gradient_core::Graph::stub(), upstream_query: std::sync::Arc::new(tokio::sync::Semaphore::new(32)), + nar_downloads: std::sync::Arc::new(tokio::sync::Semaphore::new(16)), upload_admission: gradient_storage::admission::UploadAdmission::new( gradient_storage::admission::Limits { concurrency: 16, @@ -291,6 +293,7 @@ async fn public_cache_storing_nar(served: bool) -> Arc { startable_set: Default::default(), graph: gradient_core::Graph::stub(), upstream_query: std::sync::Arc::new(tokio::sync::Semaphore::new(32)), + nar_downloads: std::sync::Arc::new(tokio::sync::Semaphore::new(16)), upload_admission: gradient_storage::admission::UploadAdmission::new( gradient_storage::admission::Limits { concurrency: 16, @@ -413,6 +416,7 @@ fn make_state( startable_set: Default::default(), graph: gradient_core::Graph::stub(), upstream_query: std::sync::Arc::new(tokio::sync::Semaphore::new(32)), + nar_downloads: std::sync::Arc::new(tokio::sync::Semaphore::new(16)), upload_admission: gradient_storage::admission::UploadAdmission::new( gradient_storage::admission::Limits { concurrency: 16, @@ -501,6 +505,7 @@ pub async fn private_cache_state() -> Arc { startable_set: Default::default(), graph: gradient_core::Graph::stub(), upstream_query: std::sync::Arc::new(tokio::sync::Semaphore::new(32)), + nar_downloads: std::sync::Arc::new(tokio::sync::Semaphore::new(16)), upload_admission: gradient_storage::admission::UploadAdmission::new( gradient_storage::admission::Limits { concurrency: 16, @@ -578,6 +583,7 @@ pub async fn private_cache_with_nar() -> Arc { startable_set: Default::default(), graph: gradient_core::Graph::stub(), upstream_query: std::sync::Arc::new(tokio::sync::Semaphore::new(32)), + nar_downloads: std::sync::Arc::new(tokio::sync::Semaphore::new(16)), upload_admission: gradient_storage::admission::UploadAdmission::new( gradient_storage::admission::Limits { concurrency: 16, diff --git a/backend/gradient-test-support/src/fakes/job_reporter.rs b/backend/gradient-test-support/src/fakes/job_reporter.rs index 04b714454..2c3ee4d57 100644 --- a/backend/gradient-test-support/src/fakes/job_reporter.rs +++ b/backend/gradient-test-support/src/fakes/job_reporter.rs @@ -39,7 +39,6 @@ pub enum ReportedEvent { metrics: Option, substituted: bool, }, - Compressing, LogChunk { task_index: u32, data: Vec, @@ -263,11 +262,6 @@ impl JobReporter for RecordingJobReporter { Ok(()) } - async fn report_compressing(&mut self) -> Result<()> { - self.record(ReportedEvent::Compressing); - Ok(()) - } - async fn send_log_chunk(&mut self, task_index: u32, data: Vec) -> Result<()> { self.record(ReportedEvent::LogChunk { task_index, data }); Ok(()) diff --git a/backend/gradient-test-support/src/state.rs b/backend/gradient-test-support/src/state.rs index 72ff1602b..7b89fded9 100644 --- a/backend/gradient-test-support/src/state.rs +++ b/backend/gradient-test-support/src/state.rs @@ -125,6 +125,7 @@ fn assemble( startable_set: Default::default(), graph: gradient_core::Graph::stub(), upstream_query: std::sync::Arc::new(tokio::sync::Semaphore::new(32)), + nar_downloads: std::sync::Arc::new(tokio::sync::Semaphore::new(16)), upload_admission: gradient_storage::admission::UploadAdmission::new( gradient_storage::admission::Limits { concurrency: 16, diff --git a/backend/gradient-test-support/src/web.rs b/backend/gradient-test-support/src/web.rs index 22e476ebc..a276b2ee4 100644 --- a/backend/gradient-test-support/src/web.rs +++ b/backend/gradient-test-support/src/web.rs @@ -138,6 +138,7 @@ fn server_with_pools( startable_set: Default::default(), graph: gradient_core::Graph::stub(), upstream_query: std::sync::Arc::new(tokio::sync::Semaphore::new(32)), + nar_downloads: std::sync::Arc::new(tokio::sync::Semaphore::new(16)), upload_admission: gradient_storage::admission::UploadAdmission::new( gradient_storage::admission::Limits { concurrency: 16, diff --git a/backend/gradient-types/src/cli/nar.rs b/backend/gradient-types/src/cli/nar.rs index fdd5a9157..500bd4698 100644 --- a/backend/gradient-types/src/cli/nar.rs +++ b/backend/gradient-types/src/cli/nar.rs @@ -74,6 +74,16 @@ pub struct NarArgs { )] pub max_concurrent_serves: usize, + /// NAR downloads from storage that may run at once across all connections, keeping the storage + /// near its best total throughput. + #[arg( + long = "nar-max-concurrent-downloads", + env = "GRADIENT_NAR_MAX_CONCURRENT_DOWNLOADS", + value_parser = greater_than_zero::, + default_value_t = 16 + )] + pub max_concurrent_downloads: usize, + /// Seconds since the last write of an unfinished upload staged under ``, after which /// the next deep GC is removing the upload. `0` is keeping every unfinished upload. #[arg( @@ -94,6 +104,7 @@ impl Default for NarArgs { storage_open_timeout_secs: 60, send_chunk_timeout_secs: 30, max_concurrent_serves: 8, + max_concurrent_downloads: 16, partial_ttl_secs: 86400, } } diff --git a/backend/gradient-types/src/cli/upload.rs b/backend/gradient-types/src/cli/upload.rs index 0a6e7e38b..43651eff8 100644 --- a/backend/gradient-types/src/cli/upload.rs +++ b/backend/gradient-types/src/cli/upload.rs @@ -14,7 +14,7 @@ pub struct UploadArgs { #[arg( long = "upload-concurrency", env = "GRADIENT_UPLOAD_CONCURRENCY", - default_value_t = 16 + default_value_t = 8 )] pub concurrency: usize, @@ -46,7 +46,7 @@ pub struct UploadArgs { impl Default for UploadArgs { fn default() -> Self { Self { - concurrency: 16, + concurrency: 8, bytes_budget: 8 * 1024 * 1024 * 1024, lease_idle_secs: 300, rest_wait_secs: 30, diff --git a/backend/gradient-util/src/telemetry.rs b/backend/gradient-util/src/telemetry.rs index de7d3d9ea..8dae4a305 100644 --- a/backend/gradient-util/src/telemetry.rs +++ b/backend/gradient-util/src/telemetry.rs @@ -10,7 +10,7 @@ use std::collections::HashMap; use std::sync::LazyLock; use std::sync::atomic::{AtomicI64, AtomicU32, Ordering}; -use std::time::{SystemTime, UNIX_EPOCH}; +use std::time::{Instant, SystemTime, UNIX_EPOCH}; use crate::sync::Mutex; @@ -20,6 +20,9 @@ pub mod metric { pub const PROTO_BULK_LANE_FILL: &str = "proto.bulk_lane_fill"; pub const PROTO_CONTROL_LANE_FILL: &str = "proto.control_lane_fill"; pub const PROTO_SEND_STALLS: &str = "proto.send_stalls"; + pub const STORAGE_READ_BYTES: &str = "storage.read_bytes"; + pub const STORAGE_WRITE_BYTES: &str = "storage.write_bytes"; + pub const STORAGE_BUSY_MS: &str = "storage.busy_ms"; pub const NAR_SERVES_WAITING: &str = "nar.serves_waiting"; pub const NAR_SERVES_ACTIVE: &str = "nar.serves_active"; pub const NAR_SERVE_FAILURES: &str = "nar.serve_failures"; @@ -134,6 +137,16 @@ impl MinuteStats { gauges.serves_active.get() as f64, minute, ); + + for (label, clock) in [ + ("read", &gauges.storage_reads), + ("write", &gauges.storage_writes), + ] { + let busy_ms = clock.take_ms(); + if busy_ms > 0.0 { + self.record_at(metric::STORAGE_BUSY_MS, label, busy_ms, minute); + } + } } } @@ -191,11 +204,72 @@ impl Default for Level { } } +struct Busy { + active: u32, + since: Option, + ms: f64, +} + +pub struct BusyClock(Mutex); + +impl BusyClock { + pub const fn new() -> Self { + Self(Mutex::new(Busy { + active: 0, + since: None, + ms: 0.0, + })) + } + + pub fn enter(&self) -> BusyGuard<'_> { + let mut busy = self.0.lock(); + if busy.active == 0 { + busy.since = Some(Instant::now()); + } + + busy.active += 1; + BusyGuard(self) + } + + pub fn take_ms(&self) -> f64 { + let mut busy = self.0.lock(); + if busy.active > 0 + && let Some(since) = busy.since.replace(Instant::now()) + { + busy.ms += since.elapsed().as_secs_f64() * 1000.0; + } + + std::mem::take(&mut busy.ms) + } +} + +impl Default for BusyClock { + fn default() -> Self { + Self::new() + } +} + +pub struct BusyGuard<'a>(&'a BusyClock); + +impl Drop for BusyGuard<'_> { + fn drop(&mut self) { + let mut busy = self.0.0.lock(); + busy.active -= 1; + if busy.active == 0 + && let Some(since) = busy.since.take() + { + busy.ms += since.elapsed().as_secs_f64() * 1000.0; + } + } +} + pub struct Gauges { pub bulk_lane_peak: Peak, pub control_lane_peak: Peak, pub serves_waiting: Level, pub serves_active: Level, + pub storage_reads: BusyClock, + pub storage_writes: BusyClock, } impl Gauges { @@ -205,6 +279,8 @@ impl Gauges { control_lane_peak: Peak::new(), serves_waiting: Level::new(), serves_active: Level::new(), + storage_reads: BusyClock::new(), + storage_writes: BusyClock::new(), } } } @@ -327,4 +403,22 @@ mod tests { assert_eq!(gauges.bulk_lane_peak.get(), 0); assert_eq!(gauges.serves_waiting.get(), 2, "levels are not reset"); } + + #[test] + fn overlapping_operations_count_their_busy_time_once() { + let pause = || std::thread::sleep(std::time::Duration::from_millis(100)); + let clock = BusyClock::new(); + let first = clock.enter(); + pause(); + let second = clock.enter(); + pause(); + drop(first); + pause(); + drop(second); + pause(); + + let busy_ms = clock.take_ms(); + assert!((300.0..400.0).contains(&busy_ms), "{busy_ms}"); + assert_eq!(clock.take_ms(), 0.0, "an idle clock adds nothing"); + } } diff --git a/backend/gradient-web/src/access.rs b/backend/gradient-web/src/access.rs index 2def6ebd1..fd65ffd02 100644 --- a/backend/gradient-web/src/access.rs +++ b/backend/gradient-web/src/access.rs @@ -763,6 +763,7 @@ mod tests { git_host: gradient_git_host::GitHostRegistry::with_builtin(), github_app_install_url: Default::default(), upstream_query: std::sync::Arc::new(tokio::sync::Semaphore::new(32)), + nar_downloads: std::sync::Arc::new(tokio::sync::Semaphore::new(16)), upload_admission: gradient_storage::admission::UploadAdmission::new( gradient_storage::admission::Limits { concurrency: 16, diff --git a/backend/gradient-web/src/endpoints/board.rs b/backend/gradient-web/src/endpoints/board.rs index 15daf9e35..bb93a9862 100644 --- a/backend/gradient-web/src/endpoints/board.rs +++ b/backend/gradient-web/src/endpoints/board.rs @@ -1199,7 +1199,6 @@ fn resource_metric_expr(metric: &str) -> Option<(&'static str, &'static str)> { "(coalesce(dm.disk_read_bytes,0) + coalesce(dm.disk_write_bytes,0))::double precision", "bytes", ), - "network" => ("dm.peak_network_mbps", "Mbps"), _ => return None, }) } @@ -1546,7 +1545,8 @@ mod tests { ram_free_mb: None, ram_total_mb: 0, disk_speed_mbps: None, - network_speed_mbps: None, + upload_speed_mbps: None, + download_speed_mbps: None, } } @@ -1818,7 +1818,7 @@ mod tests { #[test] fn expensive_resources_skip_zero_values_for_every_metric() { - for metric in ["ram", "cpu", "disk", "network"] { + for metric in ["ram", "cpu", "disk"] { let (value_expr, _) = resource_metric_expr(metric).unwrap(); let sql = expensive_by_resource_sql(value_expr, 30, None); assert!(sql.contains(&format!("{value_expr} > 0")), "sql = {sql}"); diff --git a/backend/gradient-web/src/endpoints/board_metrics.rs b/backend/gradient-web/src/endpoints/board_metrics.rs index fc594c388..07a98bcb2 100644 --- a/backend/gradient-web/src/endpoints/board_metrics.rs +++ b/backend/gradient-web/src/endpoints/board_metrics.rs @@ -293,7 +293,8 @@ pub async fn get_board_upstream_caches( #[derive(Serialize)] pub struct WorkerNet { pub worker_id: Option, - pub network_speed_mbps: Option, + pub upload_speed_mbps: Option, + pub download_speed_mbps: Option, pub disk_speed_mbps: Option, } @@ -318,7 +319,7 @@ fn workers_serving(project_list: &str) -> String { fn board_network_sql(project_filter: Option<&str>) -> String { let mut sql = String::from( - "SELECT DISTINCT ON (worker_id) worker_id, network_speed_mbps, disk_speed_mbps \ + "SELECT DISTINCT ON (worker_id) worker_id, upload_speed_mbps, download_speed_mbps, disk_speed_mbps \ FROM worker_sample \ WHERE at >= (now() AT TIME ZONE 'UTC') - interval '1 hour'", ); @@ -365,7 +366,8 @@ pub async fn get_board_network( .into_iter() .map(|r| WorkerNet { worker_id: r.try_get("", "worker_id").ok(), - network_speed_mbps: r.try_get("", "network_speed_mbps").ok().flatten(), + upload_speed_mbps: r.try_get("", "upload_speed_mbps").ok().flatten(), + download_speed_mbps: r.try_get("", "download_speed_mbps").ok().flatten(), disk_speed_mbps: r.try_get("", "disk_speed_mbps").ok().flatten(), }) .collect(); diff --git a/backend/gradient-web/src/endpoints/builds/query.rs b/backend/gradient-web/src/endpoints/builds/query.rs index 4d3ae9c46..aedc8f685 100644 --- a/backend/gradient-web/src/endpoints/builds/query.rs +++ b/backend/gradient-web/src/endpoints/builds/query.rs @@ -95,6 +95,13 @@ pub async fn get_build( None => None, }; + let prioritized = gradient_db::scheduling::priority::shared_builds_with_qos( + &state.web_db, + &[shared_build.id], + ) + .await? + .contains(&shared_build.id); + let build_with_outputs = BuildWithOutputs { id: build_job.id, evaluation: build_job.evaluation, @@ -104,7 +111,7 @@ pub async fn get_build( worker, dispatched_job: attempt.map(|a| a.dispatched_job), output: outputs, - prioritized: shared_build.prioritized, + prioritized, created_at: build_job.created_at, updated_at: shared_build.updated_at, progress: running_progress( diff --git a/backend/gradient-web/src/endpoints/evals/query.rs b/backend/gradient-web/src/endpoints/evals/query.rs index 5c2f2082c..f1e1ea1b6 100644 --- a/backend/gradient-web/src/endpoints/evals/query.rs +++ b/backend/gradient-web/src/endpoints/evals/query.rs @@ -130,6 +130,10 @@ pub async fn get_evaluation( None }; + let prioritized = + gradient_db::scheduling::priority::evaluations_with_qos(&state.web_db, &[evaluation.id]) + .await? + .contains(&evaluation.id); let progress = live_progress( &state.eval_progress, evaluation.id, @@ -157,7 +161,7 @@ pub async fn get_evaluation( warning_count, error, entry_points, - prioritized: evaluation.prioritized, + prioritized, trigger, triggered_by, waiting_reason, @@ -321,6 +325,11 @@ pub async fn get_evaluation_builds( &page_shared_build_ids, ) .await?; + let with_qos = gradient_db::scheduling::priority::shared_builds_with_qos( + &state.web_db, + &page_shared_build_ids, + ) + .await?; let mut page = Vec::with_capacity(page_slice.len()); for (_, layer, _, j, status) in &page_slice { @@ -343,7 +352,7 @@ pub async fn get_evaluation_builds( build_started_at: attempt.and_then(|a| a.build_started_at), dispatched_job: attempt.map(|a| a.dispatched_job), depth: *layer, - prioritized: shared_build.prioritized, + prioritized: with_qos.contains(&j.derivation_build), }); } diff --git a/backend/gradient-web/src/endpoints/projects/workers.rs b/backend/gradient-web/src/endpoints/projects/workers.rs index 205a98a03..6d74b7d86 100644 --- a/backend/gradient-web/src/endpoints/projects/workers.rs +++ b/backend/gradient-web/src/endpoints/projects/workers.rs @@ -349,7 +349,8 @@ pub struct WorkerSamplePoint { pub ram_free_mb: Option, pub ram_total_mb: Option, pub disk_speed_mbps: Option, - pub network_speed_mbps: Option, + pub upload_speed_mbps: Option, + pub download_speed_mbps: Option, pub assigned_jobs: i32, pub max_concurrent_builds: i32, pub state: i16, @@ -418,7 +419,8 @@ pub async fn get_project_worker_metrics( ram_free_mb: s.ram_free_mb, ram_total_mb: s.ram_total_mb, disk_speed_mbps: s.disk_speed_mbps, - network_speed_mbps: s.network_speed_mbps, + upload_speed_mbps: s.upload_speed_mbps, + download_speed_mbps: s.download_speed_mbps, assigned_jobs: s.assigned_jobs, max_concurrent_builds: s.max_concurrent_builds, state: i16::from(s.state), @@ -652,7 +654,8 @@ mod tests { ram_free_mb: None, ram_total_mb: 0, disk_speed_mbps: None, - network_speed_mbps: None, + upload_speed_mbps: None, + download_speed_mbps: None, } } diff --git a/backend/gradient-web/src/endpoints/tasks/evaluations.rs b/backend/gradient-web/src/endpoints/tasks/evaluations.rs index 891da649f..30c508f8c 100644 --- a/backend/gradient-web/src/endpoints/tasks/evaluations.rs +++ b/backend/gradient-web/src/endpoints/tasks/evaluations.rs @@ -111,6 +111,7 @@ pub(super) async fn evaluations_to_summaries( let message_counts = gradient_db::task_board::evaluation_message_counts(db, &eval_ids).await?; let eval_jobs = gradient_db::scheduling::assignment_record::latest_eval_jobs(db, &eval_ids).await?; + let with_qos = gradient_db::scheduling::priority::evaluations_with_qos(db, &eval_ids).await?; let mut out = Vec::with_capacity(evaluations.len()); for evaluation in evaluations { @@ -171,7 +172,7 @@ pub(super) async fn evaluations_to_summaries( errors, warnings, dispatched_job: eval_jobs.get(&evaluation.id).copied(), - prioritized: evaluation.prioritized, + prioritized: with_qos.contains(&evaluation.id), created_at: evaluation.created_at, started_at: evaluation.fetch_started_at, finished_at: evaluation.finished_at, @@ -622,6 +623,7 @@ pub async fn get_task_entry_points( struct EntryPointRelatedData { shared_builds: HashMap, + with_qos: HashSet, build_jobs: HashMap, derivations: HashMap, has_products: HashMap, @@ -736,9 +738,13 @@ impl EntryPointRelatedData { m }; + let shared_build_ids: Vec = + shared_builds.values().map(|a| a.id).collect(); + let with_qos = + gradient_db::scheduling::priority::shared_builds_with_qos(db, &shared_build_ids) + .await?; + let attempts: HashMap = { - let shared_build_ids: Vec = - shared_builds.values().map(|a| a.id).collect(); let mut by_shared_build = gradient_db::scheduling::build_attempt::latest_attempts(db, &shared_build_ids) .await?; @@ -770,6 +776,7 @@ impl EntryPointRelatedData { Ok(Self { shared_builds, + with_qos, build_jobs, derivations, has_products, @@ -816,7 +823,7 @@ impl EntryPointRelatedData { prioritized: self .shared_builds .get(&ep.derivation) - .is_some_and(|a| a.prioritized), + .is_some_and(|a| self.with_qos.contains(&a.id)), created_at: ep.created_at, }); } diff --git a/backend/gradient-web/tests/actions.rs b/backend/gradient-web/tests/actions.rs index 870ec502d..6edc40607 100644 --- a/backend/gradient-web/tests/actions.rs +++ b/backend/gradient-web/tests/actions.rs @@ -170,6 +170,7 @@ fn server_with_email( git_host: gradient_git_host::GitHostRegistry::with_builtin(), github_app_install_url: Default::default(), upstream_query: std::sync::Arc::new(tokio::sync::Semaphore::new(32)), + nar_downloads: std::sync::Arc::new(tokio::sync::Semaphore::new(16)), upload_admission: gradient_storage::admission::UploadAdmission::new( gradient_storage::admission::Limits { concurrency: 16, diff --git a/backend/gradient-web/tests/auth_hardening.rs b/backend/gradient-web/tests/auth_hardening.rs index 73222b2d0..ed1fdcc05 100644 --- a/backend/gradient-web/tests/auth_hardening.rs +++ b/backend/gradient-web/tests/auth_hardening.rs @@ -97,6 +97,7 @@ fn server_with(web_db_setup: impl FnOnce(MockDatabase) -> MockDatabase) -> TestS git_host: gradient_git_host::GitHostRegistry::with_builtin(), github_app_install_url: Default::default(), upstream_query: std::sync::Arc::new(tokio::sync::Semaphore::new(32)), + nar_downloads: std::sync::Arc::new(tokio::sync::Semaphore::new(16)), upload_admission: gradient_storage::admission::UploadAdmission::new( gradient_storage::admission::Limits { concurrency: 16, diff --git a/backend/gradient-web/tests/auth_middleware.rs b/backend/gradient-web/tests/auth_middleware.rs index f0b2467b5..6dfcaea6d 100644 --- a/backend/gradient-web/tests/auth_middleware.rs +++ b/backend/gradient-web/tests/auth_middleware.rs @@ -55,6 +55,7 @@ fn server() -> TestServer { git_host: gradient_git_host::GitHostRegistry::with_builtin(), github_app_install_url: Default::default(), upstream_query: std::sync::Arc::new(tokio::sync::Semaphore::new(32)), + nar_downloads: std::sync::Arc::new(tokio::sync::Semaphore::new(16)), upload_admission: gradient_storage::admission::UploadAdmission::new( gradient_storage::admission::Limits { concurrency: 16, diff --git a/backend/gradient-web/tests/body_size_limit.rs b/backend/gradient-web/tests/body_size_limit.rs index 4c20035ef..8e4ec72a1 100644 --- a/backend/gradient-web/tests/body_size_limit.rs +++ b/backend/gradient-web/tests/body_size_limit.rs @@ -51,6 +51,7 @@ fn make_state_with_limits(max_request_size: usize) -> Arc { git_host: gradient_git_host::GitHostRegistry::with_builtin(), github_app_install_url: Default::default(), upstream_query: std::sync::Arc::new(tokio::sync::Semaphore::new(32)), + nar_downloads: std::sync::Arc::new(tokio::sync::Semaphore::new(16)), upload_admission: gradient_storage::admission::UploadAdmission::new( gradient_storage::admission::Limits { concurrency: 16, diff --git a/backend/gradient-web/tests/cache_local_priority.rs b/backend/gradient-web/tests/cache_local_priority.rs index 0cd9d87de..ca9d09606 100644 --- a/backend/gradient-web/tests/cache_local_priority.rs +++ b/backend/gradient-web/tests/cache_local_priority.rs @@ -89,6 +89,7 @@ fn build_server(cache: gradient_entity::cache::Model, peer: &str) -> TestServer git_host: gradient_git_host::GitHostRegistry::with_builtin(), github_app_install_url: Default::default(), upstream_query: std::sync::Arc::new(tokio::sync::Semaphore::new(32)), + nar_downloads: std::sync::Arc::new(tokio::sync::Semaphore::new(16)), upload_admission: gradient_storage::admission::UploadAdmission::new( gradient_storage::admission::Limits { concurrency: 16, diff --git a/backend/gradient-web/tests/cli_device_authorization.rs b/backend/gradient-web/tests/cli_device_authorization.rs index 0e4e37f94..9a1682831 100644 --- a/backend/gradient-web/tests/cli_device_authorization.rs +++ b/backend/gradient-web/tests/cli_device_authorization.rs @@ -119,6 +119,7 @@ fn server_with(web_db_setup: impl FnOnce(MockDatabase) -> MockDatabase) -> TestS git_host: gradient_git_host::GitHostRegistry::with_builtin(), github_app_install_url: Default::default(), upstream_query: std::sync::Arc::new(tokio::sync::Semaphore::new(32)), + nar_downloads: std::sync::Arc::new(tokio::sync::Semaphore::new(16)), upload_admission: gradient_storage::admission::UploadAdmission::new( gradient_storage::admission::Limits { concurrency: 16, diff --git a/backend/gradient-web/tests/commits_authorization.rs b/backend/gradient-web/tests/commits_authorization.rs index d48816a95..3fbd2cdc9 100644 --- a/backend/gradient-web/tests/commits_authorization.rs +++ b/backend/gradient-web/tests/commits_authorization.rs @@ -155,6 +155,7 @@ fn make_server(db: sea_orm::DatabaseConnection) -> TestServer { git_host: gradient_git_host::GitHostRegistry::with_builtin(), github_app_install_url: Default::default(), upstream_query: std::sync::Arc::new(tokio::sync::Semaphore::new(32)), + nar_downloads: std::sync::Arc::new(tokio::sync::Semaphore::new(16)), upload_admission: gradient_storage::admission::UploadAdmission::new( gradient_storage::admission::Limits { concurrency: 16, diff --git a/backend/gradient-web/tests/evaluations_search.rs b/backend/gradient-web/tests/evaluations_search.rs index 84ff5c3ad..bf1d49d23 100644 --- a/backend/gradient-web/tests/evaluations_search.rs +++ b/backend/gradient-web/tests/evaluations_search.rs @@ -86,6 +86,7 @@ fn with_summary_rollups(db: MockDatabase) -> MockDatabase { db.append_query_results([Vec::::new()]) .append_query_results([Vec::::new()]) .append_query_results([Vec::::new()]) + .append_query_results([Vec::::new()]) } fn base_db(session_id: SessionId) -> MockDatabase { diff --git a/backend/gradient-web/tests/git_host_hooks.rs b/backend/gradient-web/tests/git_host_hooks.rs index 9df59da5a..7ec2385f9 100644 --- a/backend/gradient-web/tests/git_host_hooks.rs +++ b/backend/gradient-web/tests/git_host_hooks.rs @@ -90,6 +90,7 @@ fn make_state( git_host: gradient_git_host::GitHostRegistry::with_builtin(), github_app_install_url: Default::default(), upstream_query: std::sync::Arc::new(tokio::sync::Semaphore::new(32)), + nar_downloads: std::sync::Arc::new(tokio::sync::Semaphore::new(16)), upload_admission: gradient_storage::admission::UploadAdmission::new( gradient_storage::admission::Limits { concurrency: 16, diff --git a/backend/gradient-web/tests/metrics.rs b/backend/gradient-web/tests/metrics.rs index 8e3f8cefa..eff84be3f 100644 --- a/backend/gradient-web/tests/metrics.rs +++ b/backend/gradient-web/tests/metrics.rs @@ -63,6 +63,7 @@ fn state_with_metrics(enabled: bool, db: DatabaseConnection) -> Arc git_host: gradient_git_host::GitHostRegistry::with_builtin(), github_app_install_url: Default::default(), upstream_query: std::sync::Arc::new(tokio::sync::Semaphore::new(32)), + nar_downloads: std::sync::Arc::new(tokio::sync::Semaphore::new(16)), upload_admission: gradient_storage::admission::UploadAdmission::new( gradient_storage::admission::Limits { concurrency: 16, diff --git a/backend/gradient-web/tests/nar_serve.rs b/backend/gradient-web/tests/nar_serve.rs index 7b4d6b962..b58d624af 100644 --- a/backend/gradient-web/tests/nar_serve.rs +++ b/backend/gradient-web/tests/nar_serve.rs @@ -113,6 +113,7 @@ fn state( git_host: gradient_git_host::GitHostRegistry::with_builtin(), github_app_install_url: Default::default(), upstream_query: Arc::new(tokio::sync::Semaphore::new(32)), + nar_downloads: std::sync::Arc::new(tokio::sync::Semaphore::new(16)), upload_admission: gradient_storage::admission::UploadAdmission::new( gradient_storage::admission::Limits { concurrency: 16, diff --git a/backend/gradient-web/tests/narinfo.rs b/backend/gradient-web/tests/narinfo.rs index 9dbaa8060..319267339 100644 --- a/backend/gradient-web/tests/narinfo.rs +++ b/backend/gradient-web/tests/narinfo.rs @@ -140,6 +140,7 @@ async fn narinfo_served_from_db_without_daemon_probe() { git_host: gradient_git_host::GitHostRegistry::with_builtin(), github_app_install_url: Default::default(), upstream_query: std::sync::Arc::new(tokio::sync::Semaphore::new(32)), + nar_downloads: std::sync::Arc::new(tokio::sync::Semaphore::new(16)), upload_admission: gradient_storage::admission::UploadAdmission::new( gradient_storage::admission::Limits { concurrency: 16, @@ -285,6 +286,7 @@ async fn narinfo_returns_404_when_signature_null() { git_host: gradient_git_host::GitHostRegistry::with_builtin(), github_app_install_url: Default::default(), upstream_query: std::sync::Arc::new(tokio::sync::Semaphore::new(32)), + nar_downloads: std::sync::Arc::new(tokio::sync::Semaphore::new(16)), upload_admission: gradient_storage::admission::UploadAdmission::new( gradient_storage::admission::Limits { concurrency: 16, diff --git a/backend/gradient-web/tests/oidc_errors.rs b/backend/gradient-web/tests/oidc_errors.rs index ed58dc273..9c6bfd4bc 100644 --- a/backend/gradient-web/tests/oidc_errors.rs +++ b/backend/gradient-web/tests/oidc_errors.rs @@ -69,6 +69,7 @@ fn server_with_broken_oidc() -> TestServer { git_host: gradient_git_host::GitHostRegistry::with_builtin(), github_app_install_url: Default::default(), upstream_query: std::sync::Arc::new(tokio::sync::Semaphore::new(32)), + nar_downloads: std::sync::Arc::new(tokio::sync::Semaphore::new(16)), upload_admission: gradient_storage::admission::UploadAdmission::new( gradient_storage::admission::Limits { concurrency: 16, diff --git a/backend/gradient-web/tests/oidc_pkce.rs b/backend/gradient-web/tests/oidc_pkce.rs index 560ef0c80..fccab42ba 100644 --- a/backend/gradient-web/tests/oidc_pkce.rs +++ b/backend/gradient-web/tests/oidc_pkce.rs @@ -122,6 +122,7 @@ async fn authorize_redirect_carries_pkce_and_cookie_holds_verifier() { git_host: gradient_git_host::GitHostRegistry::with_builtin(), github_app_install_url: Default::default(), upstream_query: std::sync::Arc::new(tokio::sync::Semaphore::new(32)), + nar_downloads: std::sync::Arc::new(tokio::sync::Semaphore::new(16)), upload_admission: gradient_storage::admission::UploadAdmission::new( gradient_storage::admission::Limits { concurrency: 16, diff --git a/backend/gradient-web/tests/old_direct_build_gone.rs b/backend/gradient-web/tests/old_direct_build_gone.rs index f1a4eb19a..f33553ae5 100644 --- a/backend/gradient-web/tests/old_direct_build_gone.rs +++ b/backend/gradient-web/tests/old_direct_build_gone.rs @@ -46,6 +46,7 @@ fn make_state() -> Arc { git_host: gradient_git_host::GitHostRegistry::with_builtin(), github_app_install_url: Default::default(), upstream_query: std::sync::Arc::new(tokio::sync::Semaphore::new(32)), + nar_downloads: std::sync::Arc::new(tokio::sync::Semaphore::new(16)), upload_admission: gradient_storage::admission::UploadAdmission::new( gradient_storage::admission::Limits { concurrency: 16, diff --git a/backend/gradient-web/tests/rate_limit.rs b/backend/gradient-web/tests/rate_limit.rs index d6317c469..048cbdfba 100644 --- a/backend/gradient-web/tests/rate_limit.rs +++ b/backend/gradient-web/tests/rate_limit.rs @@ -46,6 +46,7 @@ fn make_state() -> Arc { git_host: gradient_git_host::GitHostRegistry::with_builtin(), github_app_install_url: Default::default(), upstream_query: std::sync::Arc::new(tokio::sync::Semaphore::new(32)), + nar_downloads: std::sync::Arc::new(tokio::sync::Semaphore::new(16)), upload_admission: gradient_storage::admission::UploadAdmission::new( gradient_storage::admission::Limits { concurrency: 16, diff --git a/backend/gradient-web/tests/request_id.rs b/backend/gradient-web/tests/request_id.rs index 70ce915b5..e574a1596 100644 --- a/backend/gradient-web/tests/request_id.rs +++ b/backend/gradient-web/tests/request_id.rs @@ -47,6 +47,7 @@ fn make_state() -> Arc { git_host: gradient_git_host::GitHostRegistry::with_builtin(), github_app_install_url: Default::default(), upstream_query: std::sync::Arc::new(tokio::sync::Semaphore::new(32)), + nar_downloads: std::sync::Arc::new(tokio::sync::Semaphore::new(16)), upload_admission: gradient_storage::admission::UploadAdmission::new( gradient_storage::admission::Limits { concurrency: 16, diff --git a/backend/gradient-web/tests/scim.rs b/backend/gradient-web/tests/scim.rs index 38b7d426f..5e8cff0ac 100644 --- a/backend/gradient-web/tests/scim.rs +++ b/backend/gradient-web/tests/scim.rs @@ -76,6 +76,7 @@ fn build_server(db: DatabaseConnection, hard_delete: bool) -> TestServer { git_host: gradient_git_host::GitHostRegistry::with_builtin(), github_app_install_url: Default::default(), upstream_query: std::sync::Arc::new(tokio::sync::Semaphore::new(32)), + nar_downloads: std::sync::Arc::new(tokio::sync::Semaphore::new(16)), upload_admission: gradient_storage::admission::UploadAdmission::new( gradient_storage::admission::Limits { concurrency: 16, diff --git a/backend/gradient-web/tests/tasks_sign_cache.rs b/backend/gradient-web/tests/tasks_sign_cache.rs index 068778409..0cc40a08b 100644 --- a/backend/gradient-web/tests/tasks_sign_cache.rs +++ b/backend/gradient-web/tests/tasks_sign_cache.rs @@ -97,6 +97,7 @@ fn make_server(db: sea_orm::DatabaseConnection) -> TestServer { git_host: gradient_git_host::GitHostRegistry::with_builtin(), github_app_install_url: Default::default(), upstream_query: std::sync::Arc::new(tokio::sync::Semaphore::new(32)), + nar_downloads: std::sync::Arc::new(tokio::sync::Semaphore::new(16)), upload_admission: gradient_storage::admission::UploadAdmission::new( gradient_storage::admission::Limits { concurrency: 16, diff --git a/backend/gradient-wire-derive/src/lib.rs b/backend/gradient-wire-derive/src/lib.rs index 7bb9405be..2f9a44d47 100644 --- a/backend/gradient-wire-derive/src/lib.rs +++ b/backend/gradient-wire-derive/src/lib.rs @@ -17,11 +17,17 @@ pub fn derive_proto(input: proc_macro::TokenStream) -> proc_macro::TokenStream { .into() } +struct Removed { + until: u16, + ty: Type, +} + #[derive(Default)] struct Versions { since: u16, default: bool, oldest: Option, + removed: Vec, } fn versions(attrs: &[Attribute]) -> syn::Result { @@ -51,10 +57,18 @@ fn parse_versions(input: ParseStream, out: &mut Versions) -> syn::Result<()> { input.parse::()?; out.oldest = Some(input.parse::()?.base10_parse()?); } + "removed" => { + let content; + syn::parenthesized!(content in input); + let until = content.parse::()?.base10_parse()?; + content.parse::()?; + let ty = content.parse()?; + out.removed.push(Removed { until, ty }); + } _ => { return Err(syn::Error::new( word.span(), - "expected a version, `default` or `oldest = N`", + "expected a version, `default`, `oldest = N` or `removed(N, Type)`", )); } } @@ -86,6 +100,13 @@ fn fields(fields: &Fields) -> syn::Result> { return Err(syn::Error::new_spanned(f, "`oldest` belongs on the type")); } + if !v.removed.is_empty() { + return Err(syn::Error::new_spanned( + f, + "`removed` belongs on the struct or variant that lost the field", + )); + } + Ok(Field { member: f.ident.clone(), binding: format_ident!("__f{i}"), @@ -140,6 +161,22 @@ fn field_versions(fields: &[Field], present_since: u16) -> (Vec, Ve .unzip() } +fn encode_removed(removed: &[Removed]) -> TokenStream { + let codec = codec(); + let steps = removed.iter().map(|Removed { until, ty }| { + quote!(if version < #until { #codec::Proto::encode(&<#ty as ::core::default::Default>::default(), version, out)?; }) + }); + quote!(#(#steps)*) +} + +fn decode_removed(removed: &[Removed]) -> TokenStream { + let codec = codec(); + let steps = removed.iter().map(|Removed { until, ty }| { + quote!(if version < #until { <#ty as #codec::Proto>::decode(input, version)?; }) + }); + quote!(#(#steps)*) +} + fn encode_fields(fields: &[Field], present_since: u16) -> TokenStream { let codec = codec(); let steps = fields.iter().map(|f| { @@ -169,26 +206,30 @@ fn decode_fields(fields: &[Field], present_since: u16) -> TokenStream { quote!(#(#steps)*) } -fn describe_fields(fields: &[Field], present_since: u16) -> TokenStream { - if fields.is_empty() { +fn describe_fields(fields: &[Field], removed: &[Removed], present_since: u16) -> TokenStream { + if fields.is_empty() && removed.is_empty() { return quote!(out.push_str("{}");); } let codec = codec(); - let steps = fields.iter().map(|f| { - let ty = &f.ty; - when_present( - f.since.max(present_since), - quote! { - #codec::push_separator(out, &mut first); - <#ty as #codec::Proto>::describe(version, out); - }, - ) + let describe = |ty: &Type| { + quote! { + #codec::push_separator(out, &mut first); + <#ty as #codec::Proto>::describe(version, out); + } + }; + let steps = fields + .iter() + .map(|f| when_present(f.since.max(present_since), describe(&f.ty))); + let removed_steps = removed.iter().map(|Removed { until, ty }| { + let step = describe(ty); + quote!(if version < #until { #step }) }); quote!({ out.push('{'); let mut first = true; #(#steps)* + #(#removed_steps)* out.push('}'); }) } @@ -201,19 +242,22 @@ struct Body { describe: TokenStream, } -fn struct_body(shape: &Fields) -> syn::Result { +fn struct_body(shape: &Fields, removed: &[Removed]) -> syn::Result { let fs = fields(shape)?; - let (oldest, newest) = field_versions(&fs, 0); + let (oldest, mut newest) = field_versions(&fs, 0); + newest.extend(removed.iter().map(|r| r.until).map(|until| quote!(#until))); let pat = pattern(quote!(Self), &fs, shape); let encode = encode_fields(&fs, 0); + let encode_removed = encode_removed(removed); let decode = decode_fields(&fs, 0); + let decode_removed = decode_removed(removed); let bind = (!fs.is_empty()).then(|| quote!(let #pat = self;)); Ok(Body { oldest, newest, - encode: quote! { #bind #encode Ok(()) }, - decode: quote! { #decode Ok(#pat) }, - describe: describe_fields(&fs, 0), + encode: quote! { #bind #encode #encode_removed Ok(()) }, + decode: quote! { #decode #decode_removed Ok(#pat) }, + describe: describe_fields(&fs, removed, 0), }) } @@ -235,7 +279,7 @@ fn enum_body(name: &Ident, data: &syn::DataEnum) -> syn::Result { if v.default || v.oldest.is_some() { return Err(syn::Error::new_spanned( variant, - "a variant takes only its version", + "a variant takes only its version and `removed(N, Type)`", )); } @@ -257,10 +301,16 @@ fn enum_body(name: &Ident, data: &syn::DataEnum) -> syn::Result { body.oldest.extend(oldest); body.newest.extend(newest); body.newest.push(quote!(#since)); + body.newest.extend( + v.removed + .iter() + .map(|r| r.until) + .map(|until| quote!(#until)), + ); let pat = pattern(quote!(Self::#ident), &fs, &variant.fields); - let encode = encode_fields(&fs, since); - let decode = decode_fields(&fs, since); - let describe = describe_fields(&fs, since); + let (encode, encode_removed) = (encode_fields(&fs, since), encode_removed(&v.removed)); + let (decode, decode_removed) = (decode_fields(&fs, since), decode_removed(&v.removed)); + let describe = describe_fields(&fs, &v.removed, since); let refuse_older_peer = (since > 0).then(|| { quote! { if version < #since { @@ -270,10 +320,10 @@ fn enum_body(name: &Ident, data: &syn::DataEnum) -> syn::Result { }); let guard = (since > 0).then(|| quote!(if version >= #since)); encode_arms.push(quote! { - #pat => { #refuse_older_peer #codec::put_varint(out, #tag); #encode } + #pat => { #refuse_older_peer #codec::put_varint(out, #tag); #encode #encode_removed } }); decode_arms.push(quote! { - #tag #guard => { #decode Ok(#pat) } + #tag #guard => { #decode #decode_removed Ok(#pat) } }); describe_arms.push(when_present( since, @@ -321,10 +371,11 @@ fn param(body: &TokenStream, name: &str) -> Ident { fn expand(input: &DeriveInput) -> syn::Result { let name = &input.ident; let top = versions(&input.attrs)?; - if top.since != 0 || top.default { + let removed_on_enum = matches!(input.data, Data::Enum(_)) && !top.removed.is_empty(); + if top.since != 0 || top.default || removed_on_enum { return Err(syn::Error::new_spanned( name, - "only `oldest = N` belongs on the type", + "only `oldest = N` belongs on the type, plus `removed(N, Type)` on a struct", )); } @@ -337,7 +388,7 @@ fn expand(input: &DeriveInput) -> syn::Result { decode, describe, } = match &input.data { - Data::Struct(data) => struct_body(&data.fields)?, + Data::Struct(data) => struct_body(&data.fields, &top.removed)?, Data::Enum(data) => enum_body(name, data)?, Data::Union(_) => { return Err(syn::Error::new_spanned(name, "unions are not supported")); diff --git a/backend/gradient-wire/schema/v28.txt b/backend/gradient-wire/schema/v28.txt new file mode 100644 index 000000000..6487bca26 --- /dev/null +++ b/backend/gradient-wire/schema/v28.txt @@ -0,0 +1,56 @@ +client +0:{{bool,bool,bool,bool,bool,bool},str} +1:{seq<(str,str)>} +2:{} +3:{u16,str} +4:{seq,seq,u32,u32,u64,u32,opt,opt} +5:{f32,u64,opt,opt,opt} +6:{} +7:{seq<{str,u32,u64,bool}>,bool} +8:{str,bool,opt} +9:{str,str,[0:{},1:{opt},2:{},3:{},4:{seq<{str,str,seq<{str,str}>,seq,seq,str,seq,opt,opt,bool,bool,bool,opt}>,seq,seq},5:{str},6:{str,seq<{str,str,str,opt,opt,seq<{str,str,str,str,opt}>}>,opt<{opt,opt,opt,opt,opt,bool,opt,opt,opt,opt}>,bool},7:{},8:{{u64,u64,u64,u64,u64,u64,u64,u64,str,seq<{str,u64,u64,u64,u64}>,seq<{str,opt,str,str,bool,opt}>}},9:{str,seq<{str,opt,str}>},10:{seq},11:{[0:{},1:{},2:{}]}]} +10:{str,str,seq<{[0:{},1:{},2:{},3:{},4:{},5:{},6:{},7:{},8:{},9:{},10:{},11:{},12:{},13:{},14:{},15:{},16:{},17:{}],u64,u64,opt,u32,u64}>,u64} +11:{str,str,str,[0:{},1:{},2:{},3:{},4:{},5:{},6:{}],seq,seq<{[0:{},1:{},2:{},3:{},4:{},5:{},6:{},7:{},8:{},9:{},10:{},11:{},12:{},13:{},14:{},15:{},16:{},17:{}],u64,u64,opt,u32,u64}>,u64,opt<{opt,opt,opt,opt,opt,bool,opt,opt,opt,opt}>} +12:{} +13:{str,str,str,[0:{},1:{},2:{}],u64,opt,u32,opt} +14:{str,str,[0:{seq<{str,[0:{},1:{},2:{},3:{}],u64,u64}>},1:{u64}]} +15:{str,u32,bytes} +16:{str,seq} +17:{str,str,u64,str} +18:{str,str} +19:{[0:{},1:{}]} +20:{str,opt<{str,u32}>,bytes} +21:{str,str,seq,[0:{},1:{},2:{}],seq>,bool} +22:{str,[0:{},1:{},2:{}],str,str} +23:{str,str,seq} +24:{str,u64,[0:{str},1:{str}],u64} +25:{u64,bytes,u64,bool} +26:{u64,[0:{{str,u64,u64,str,seq,opt,opt,opt<{str,seq}>}},1:{u64}]} +27:{u64} +server +0:{seq} +1:{{bool,bool,bool,bool,bool,bool},seq,seq<{str,str}>} +2:{seq,seq<{str,str}>} +3:{u16,str} +4:{u16,str} +5:{} +6:{seq<{str,seq<{str,opt<{u64,u64}>}>,seq,seq,opt<{str,seq}>}>,bool} +7:{seq<{str,seq<{str,opt<{u64,u64}>}>,seq,seq,opt<{str,seq}>}>} +8:{str,str,[0:{{seq<[0:{},1:{},2:{}]>,[0:{str,str},1:{str}],seq,opt,seq<{str,opt}>,opt<{str,seq,bool}>}},1:{{seq<{str,str,[0:{},1:{},2:{}],bool,seq<{str,str}>,opt,opt}>,{str,seq}}}],opt<{str,str,u32,u32}>} +9:{str,str} +10:{str,seq<{str,u32,str,opt,opt}>} +11:{str,{str,u32},bytes} +12:{str,str} +13:{[0:{}],bytes} +14:{str,str,bytes,u64,bool} +15:{str,str,str} +16:{str,str,str} +17:{str,str,u64,str} +18:{str,[0:{},1:{str},2:{u64,str}]} +19:{str,bytes,u64,bool} +20:{str,seq<{str,bool,opt,opt,opt,opt,opt,opt>,opt>,opt,opt}>} +21:{str,seq} +22:{str,str} +23:{u64,[0:{},1:{u64},2:{str},3:{{str,u64,seq}}]} +24:{u64,[0:{},1:{str},2:{str}]} +25:{str,seq<(str,str)>} diff --git a/backend/gradient-wire/schema/v29.txt b/backend/gradient-wire/schema/v29.txt new file mode 100644 index 000000000..ce74e8fc1 --- /dev/null +++ b/backend/gradient-wire/schema/v29.txt @@ -0,0 +1,59 @@ +client +0:{{bool,bool,bool,bool,bool,bool},str} +1:{seq<(str,str)>} +2:{} +3:{u16,str} +4:{seq,seq,u32,u32,u64,u32,opt,opt} +5:{f32,u64,opt,opt,opt} +6:{} +7:{seq<{str,u32,u64,bool}>,bool} +8:{str,bool,opt} +9:{str,str,[0:{},1:{opt},2:{},3:{},4:{seq<{str,str,seq<{str,str}>,seq,seq,str,seq,opt,opt,bool,bool,bool,opt}>,seq,seq},5:{str},6:{str,seq<{str,str,str,opt,opt,seq<{str,str,str,str,opt}>}>,opt<{opt,opt,opt,opt,opt,bool,opt,opt,opt,opt}>,bool},7:{},8:{{u64,u64,u64,u64,u64,u64,u64,u64,str,seq<{str,u64,u64,u64,u64}>,seq<{str,opt,str,str,bool,opt}>}},9:{str,seq<{str,opt,str}>},10:{seq},11:{[0:{},1:{},2:{}]}]} +10:{str,str,seq<{[0:{},1:{},2:{},3:{},4:{},5:{},6:{},7:{},8:{},9:{},10:{},11:{},12:{},13:{},14:{},15:{},16:{},17:{}],u64,u64,opt,u32,u64}>,u64} +11:{str,str,str,[0:{},1:{},2:{},3:{},4:{},5:{},6:{}],seq,seq<{[0:{},1:{},2:{},3:{},4:{},5:{},6:{},7:{},8:{},9:{},10:{},11:{},12:{},13:{},14:{},15:{},16:{},17:{}],u64,u64,opt,u32,u64}>,u64,opt<{opt,opt,opt,opt,opt,bool,opt,opt,opt,opt}>} +12:{} +13:{str,str,str,[0:{},1:{},2:{}],u64,opt,u32,opt} +14:{str,str,[0:{seq<{str,[0:{},1:{},2:{},3:{}],u64,u64}>},1:{u64}]} +15:{str,u32,bytes} +16:{str,seq} +17:{str,str,u64,str} +18:{str,str} +19:{[0:{},1:{}]} +20:{str,opt<{str,u32}>,bytes} +21:{str,str,seq,[0:{},1:{},2:{}],seq>,bool} +22:{str,[0:{},1:{},2:{}],str,str} +23:{str,str,seq} +24:{str,u64,[0:{str},1:{str}],u64} +25:{u64,bytes,u64,bool} +26:{u64,[0:{{str,u64,u64,str,seq,opt,opt,opt<{str,seq}>}},1:{u64}]} +27:{u64} +28:{} +29:{seq} +server +0:{seq} +1:{{bool,bool,bool,bool,bool,bool},seq,seq<{str,str}>} +2:{seq,seq<{str,str}>} +3:{u16,str} +4:{u16,str} +5:{} +6:{seq<{str,seq<{str,opt<{u64,u64}>}>,seq,seq,opt<{str,seq}>}>,bool} +7:{seq<{str,seq<{str,opt<{u64,u64}>}>,seq,seq,opt<{str,seq}>}>} +8:{str,str,[0:{{seq<[0:{},1:{},2:{}]>,[0:{str,str},1:{str}],seq,opt,seq<{str,opt}>,opt<{str,seq,bool}>}},1:{{seq<{str,str,[0:{},1:{},2:{}],bool,seq<{str,str}>,opt,opt}>,{str,seq}}}],opt<{str,str,u32,u32}>} +9:{str,str} +10:{str,seq<{str,u32,str,opt,opt}>} +11:{str,{str,u32},bytes} +12:{str,str} +13:{[0:{}],bytes} +14:{str,str,bytes,u64,bool} +15:{str,str,str} +16:{str,str,str} +17:{str,str,u64,str} +18:{str,[0:{},1:{str},2:{u64,str}]} +19:{str,bytes,u64,bool} +20:{str,seq<{str,bool,opt,opt,opt,opt,opt,opt>,opt>,opt,opt}>} +21:{str,seq} +22:{str,str} +23:{u64,[0:{},1:{u64},2:{str},3:{{str,u64,seq}}]} +24:{u64,[0:{},1:{str},2:{str}]} +25:{str,seq<(str,str)>} +26:{u32,seq,bool} diff --git a/backend/gradient-wire/src/messages/client.rs b/backend/gradient-wire/src/messages/client.rs index b09dd3205..29bbc3d7d 100644 --- a/backend/gradient-wire/src/messages/client.rs +++ b/backend/gradient-wire/src/messages/client.rs @@ -8,9 +8,9 @@ use bytes::Bytes; use crate::codec::Proto; use crate::types::{ - BuildFailureKind, BuildProgressPhase, CandidateScore, ClusterAddress, EvalMessageLevel, - EvalProgress, GradientCapabilities, JobKind, JobPhaseSpan, JobUpdateKind, QueryMode, - UploadMetadata, UploadObject, + BuildFailureKind, BuildMetrics, BuildProgressPhase, CandidateScore, ClusterAddress, + EvalMessageLevel, EvalProgress, GradientCapabilities, JobKind, JobPhaseSpan, JobUpdateKind, + QueryMode, UploadMetadata, UploadObject, }; #[derive(Proto, Debug, Clone, PartialEq)] @@ -43,11 +43,15 @@ pub enum ClientMessage { endpoint: Option, }, + #[proto(removed(28, Option))] WorkerMetrics { cpu_usage_pct: f32, ram_free_mb: u64, disk_speed_mbps: Option, - network_speed_mbps: Option, + #[proto(28, default)] + upload_speed_mbps: Option, + #[proto(28, default)] + download_speed_mbps: Option, }, RequestJobList, @@ -84,6 +88,8 @@ pub enum ClientMessage { missing_paths: Vec, spans: Vec, elapsed_ms: u64, + #[proto(28, default)] + metrics: Option, }, Draining, @@ -184,6 +190,12 @@ pub enum ClientMessage { UploadCancel { request_id: u64, }, + #[proto(29)] + HandoverDone, + #[proto(29)] + PathsAdded { + paths: Vec, + }, } impl ClientMessage { @@ -237,6 +249,8 @@ impl ClientMessage { ClientMessage::UploadChunk { .. } => "UploadChunk", ClientMessage::UploadFinished { .. } => "UploadFinished", ClientMessage::UploadCancel { .. } => "UploadCancel", + ClientMessage::HandoverDone => "HandoverDone", + ClientMessage::PathsAdded { .. } => "PathsAdded", } } diff --git a/backend/gradient-wire/src/messages/mod.rs b/backend/gradient-wire/src/messages/mod.rs index e8d93afe3..ee64daa00 100644 --- a/backend/gradient-wire/src/messages/mod.rs +++ b/backend/gradient-wire/src/messages/mod.rs @@ -9,7 +9,7 @@ pub mod server; pub use crate::types::{ BuildFailureKind, BuildJob, BuildMetrics, BuildOutput, BuildProduct, BuildProgressPhase, - BuildRequirement, BuildSpec, BuildSpecKind, BumpedInputWire, CacheInfo, CachedPath, + BuildRequirement, BuildSpec, BuildSpecKind, BuildStage, BumpedInputWire, CacheInfo, CachedPath, CandidateScore, ClusterAddress, ClusterMembership, ClusterPeer, CredentialKind, DerivationOutput, DiscoveredDerivation, EvalAttrCost, EvalCachePullOutcome, EvalMessageLevel, EvalProgress, EvalStatsReport, FlakeInputOverride, FlakeJob, FlakeOutputNode, FlakeSource, diff --git a/backend/gradient-wire/src/messages/server.rs b/backend/gradient-wire/src/messages/server.rs index 1a12bd5d2..f6b3e8a70 100644 --- a/backend/gradient-wire/src/messages/server.rs +++ b/backend/gradient-wire/src/messages/server.rs @@ -160,6 +160,12 @@ pub enum ServerMessage { worker_id: String, tokens: Vec<(String, String)>, }, + #[proto(29)] + Handover { + index: u32, + paths: Vec, + is_final: bool, + }, } impl ServerMessage { @@ -205,6 +211,7 @@ impl ServerMessage { ServerMessage::UploadGrant { .. } => "UploadGrant", ServerMessage::UploadCommitted { .. } => "UploadCommitted", ServerMessage::Authenticate { .. } => "Authenticate", + ServerMessage::Handover { .. } => "Handover", } } diff --git a/backend/gradient-wire/src/traits.rs b/backend/gradient-wire/src/traits.rs index e817e20e4..83b3df84a 100644 --- a/backend/gradient-wire/src/traits.rs +++ b/backend/gradient-wire/src/traits.rs @@ -70,7 +70,6 @@ pub trait JobReporter: Send + Sync { metrics: Option, substituted: bool, ) -> Result<()>; - async fn report_compressing(&mut self) -> Result<()>; async fn send_log_chunk(&mut self, task_index: u32, data: Vec) -> Result<()>; async fn send_eval_message( &mut self, diff --git a/backend/gradient-wire/src/types.rs b/backend/gradient-wire/src/types.rs index 4686994e3..532fcfc05 100644 --- a/backend/gradient-wire/src/types.rs +++ b/backend/gradient-wire/src/types.rs @@ -146,6 +146,17 @@ pub enum JobUpdateKind { InputUpdateExpansion { matched: Vec, }, + #[proto(28)] + Stage(BuildStage), +} + +pub const PROTO_BUILD_STAGES: u16 = 28; + +#[derive(Proto, Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub enum BuildStage { + Prefetch, + Build, + Upload, } #[derive(Proto, Debug, Clone, PartialEq)] @@ -307,6 +318,7 @@ pub struct BuildOutput { } #[derive(Proto, Debug, Clone, PartialEq, Default)] +#[proto(removed(28, Option))] pub struct BuildMetrics { pub peak_ram_mb: Option, pub cpu_time_ms: Option, @@ -315,7 +327,12 @@ pub struct BuildMetrics { pub disk_write_bytes: Option, pub oom_killed: bool, pub build_time_ms: Option, - pub peak_network_mbps: Option, + #[proto(28, default)] + pub concurrent_builds: Option, + #[proto(28, default)] + pub build_cores: Option, + #[proto(28, default)] + pub cpu_core_score: Option, } #[derive(Proto, Debug, Clone, PartialEq, Default)] @@ -401,10 +418,12 @@ pub enum JobPhase { CacheQueryWait, NarFetch, NarImport, + #[proto(28)] + UploadWait, } impl JobPhase { - pub const ALL: [Self; 17] = [ + pub const ALL: [Self; 18] = [ Self::Fetch, Self::PushInputs, Self::EvalFlake, @@ -422,6 +441,7 @@ impl JobPhase { Self::CacheQueryWait, Self::NarFetch, Self::NarImport, + Self::UploadWait, ]; pub const fn as_str(self) -> &'static str { @@ -443,6 +463,7 @@ impl JobPhase { Self::CacheQueryWait => "cache_query_wait", Self::NarFetch => "nar_fetch", Self::NarImport => "nar_import", + Self::UploadWait => "upload_wait", } } @@ -468,6 +489,7 @@ impl JobPhase { Self::Download => 15, Self::NarFetch => 16, Self::NarImport => 17, + Self::UploadWait => 18, } } @@ -490,10 +512,18 @@ impl JobPhase { 15 => Self::Download, 16 => Self::NarFetch, 17 => Self::NarImport, + 18 => Self::UploadWait, _ => return None, }) } + pub const fn known_to(self, version: u16) -> Self { + match self { + Self::UploadWait if version < PROTO_BUILD_STAGES => Self::NarPush, + phase => phase, + } + } + pub fn name_of(v: i16) -> std::borrow::Cow<'static, str> { match (Self::from_i16(v), v) { (Some(phase), _) => phase.as_str().into(), diff --git a/backend/gradient-wire/tests/derive.rs b/backend/gradient-wire/tests/derive.rs index a5edd10d3..beeb6f2ee 100644 --- a/backend/gradient-wire/tests/derive.rs +++ b/backend/gradient-wire/tests/derive.rs @@ -52,10 +52,97 @@ enum Grown { }, } +#[derive(Proto, Debug, Clone, PartialEq, Default)] +#[proto(removed(28, Option))] +struct Trimmed { + id: String, + #[proto(28, default)] + upload: Option, +} + +#[derive(Proto, Debug, Clone, PartialEq, Default)] +struct BeforeTrim { + id: String, + peak: Option, +} + +#[derive(Proto, Debug, Clone, PartialEq)] +#[proto(oldest = 27)] +enum Heartbeat { + #[proto(removed(28, Option))] + Load { + cpu: u32, + #[proto(28, default)] + upload: Option, + }, +} + +#[derive(Proto, Debug, Clone, PartialEq)] +#[proto(oldest = 27)] +enum HeartbeatBeforeTrim { + Load { cpu: u32, peak: Option }, +} + +#[derive(Proto, Debug, Clone, PartialEq)] +#[proto(oldest = 27)] +enum HeartbeatAfterTrim { + Load { cpu: u32, upload: Option }, +} + fn at(value: &T, version: u16) -> Bytes { to_bytes(value, version).expect("encodes") } +fn shape(version: u16) -> String { + let mut out = String::new(); + T::describe(version, &mut out); + out +} + +#[test] +fn a_removed_field_keeps_its_place_for_older_peers_only() { + let trimmed = Trimmed { + id: "j".into(), + upload: Some(8.0), + }; + let unset = BeforeTrim { + id: "j".into(), + peak: None, + }; + assert_eq!(at(&trimmed, 27), at(&unset, 27)); + assert_eq!(shape::(27), shape::(27)); + + let sent_by_old_peer = BeforeTrim { + peak: Some(3.0), + ..unset + }; + assert_eq!( + from_bytes::(at(&sent_by_old_peer, 27), 27), + Ok(Trimmed { + id: "j".into(), + upload: None, + }) + ); + assert_eq!(from_bytes::(at(&trimmed, 28), 28), Ok(trimmed)); + assert_eq!((Trimmed::OLDEST, Trimmed::NEWEST), (0, 28)); +} + +#[test] +fn a_variant_drops_a_removed_field_from_the_version_that_removed_it() { + assert_eq!(shape::(27), shape::(27)); + assert_eq!(shape::(28), shape::(28)); + + let load = Heartbeat::Load { + cpu: 4, + upload: Some(8.0), + }; + assert_eq!( + at(&load, 27), + at(&HeartbeatBeforeTrim::Load { cpu: 4, peak: None }, 27) + ); + assert_eq!(from_bytes::(at(&load, 28), 28), Ok(load)); +} + #[test] fn a_field_newer_than_the_peer_is_left_out_and_read_as_its_default() { let offer = Offer { diff --git a/backend/gradient-wire/tests/job_phase_names.rs b/backend/gradient-wire/tests/job_phase_names.rs index cb0c5c15c..fc5e89079 100644 --- a/backend/gradient-wire/tests/job_phase_names.rs +++ b/backend/gradient-wire/tests/job_phase_names.rs @@ -27,3 +27,16 @@ fn every_phase_round_trips_through_its_own_code() { codes.dedup(); assert_eq!(codes.len(), JobPhase::ALL.len()); } + +#[test] +fn every_phase_encodes_for_every_supported_peer_once_mapped_to_its_version() { + for version in gradient_wire::PROTO_VERSIONS { + for phase in JobPhase::ALL { + let known = phase.known_to(version); + assert!( + gradient_wire::codec::to_bytes(&known, version).is_ok(), + "{phase:?} as {known:?} at protocol {version}" + ); + } + } +} diff --git a/backend/gradient-wire/tests/wire_bytes.rs b/backend/gradient-wire/tests/wire_bytes.rs index 6666882ed..641c95bfc 100644 --- a/backend/gradient-wire/tests/wire_bytes.rs +++ b/backend/gradient-wire/tests/wire_bytes.rs @@ -41,7 +41,8 @@ fn worker_metrics_keep_their_baseline_bytes() { cpu_usage_pct: 1.5, ram_free_mb: 128, disk_speed_mbps: Some(-2.0), - network_speed_mbps: None, + upload_speed_mbps: None, + download_speed_mbps: None, }, "050000c03f800101000000c000", ); diff --git a/backend/gradient-worker-client/src/connection/mod.rs b/backend/gradient-worker-client/src/connection/mod.rs index dd76e49cf..c68d685be 100644 --- a/backend/gradient-worker-client/src/connection/mod.rs +++ b/backend/gradient-worker-client/src/connection/mod.rs @@ -91,6 +91,10 @@ pub struct ProtoWriter { } impl ProtoWriter { + pub fn version(&self) -> u16 { + self.inner.version() + } + pub async fn send(&self, msg: ClientMessage) -> Result<()> { self.inner .send_msg(&msg) diff --git a/backend/gradient-worker-client/src/nar.rs b/backend/gradient-worker-client/src/nar.rs index fefa6e04a..17388e692 100644 --- a/backend/gradient-worker-client/src/nar.rs +++ b/backend/gradient-worker-client/src/nar.rs @@ -120,12 +120,18 @@ impl std::ops::AddAssign for UploadedNar { } } +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum UploadEvent { + Granted, + Read(u64), +} + pub async fn upload_nar( uploads: &UploadClient, job_id: &str, store_path: &str, source: NarSource<'_>, - nar_read: &mut (dyn FnMut(u64) + Send), + events: &mut (dyn FnMut(UploadEvent) + Send), ) -> Result { let store_path = nix_store_path(store_path); let object = UploadObject::Nar { @@ -141,13 +147,14 @@ pub async fn upload_nar( let threads = compression_threads(Some(nar_size)); let mut upload = uploads.start(job_id, object, nar_size).await?; while let Some((request_id, target)) = upload.next_grant().await? { + events(UploadEvent::Granted); let sent = send_path( uploads.writer(), request_id, &store_path, threads, target, - nar_read, + &mut |n| events(UploadEvent::Read(n)), ) .await; if let Some(uploaded) = settle(&mut upload, sent, &path_meta).await? { @@ -173,9 +180,10 @@ pub async fn upload_nar( }; let mut upload = uploads.start(job_id, object, nar_size).await?; while let Some((request_id, target)) = upload.next_grant().await? { + events(UploadEvent::Granted); let sent = send_raw(uploads.writer(), request_id, &nar, target).await; if sent.is_ok() { - nar_read(nar_size); + events(UploadEvent::Read(nar_size)); } if let Some(uploaded) = settle(&mut upload, sent, &path_meta).await? { return Ok(uploaded); @@ -212,7 +220,6 @@ async fn send_path( match target { GrantTarget::Passthrough { resume_offset } => { debug!(store_path, resume_offset, "passthrough NAR upload"); - let started = std::time::Instant::now(); let mut passthrough = PassthroughStream::new(request_id, writer, resume_offset); let meta = pack_path_in_parts( store_path, @@ -222,8 +229,7 @@ async fn send_path( nar_read, ) .await?; - let sent = passthrough.finish().await?; - crate::throughput::NETWORK.observe_transfer(sent, started.elapsed()); + passthrough.finish().await?; Ok((meta, None)) } GrantTarget::Put { url } => { @@ -337,7 +343,7 @@ impl<'a> PassthroughStream<'a> { } } - async fn finish(self) -> Result { + async fn finish(self) -> Result<()> { self.writer .send(ClientMessage::UploadChunk { request_id: self.request_id, @@ -346,7 +352,7 @@ impl<'a> PassthroughStream<'a> { is_final: true, }) .await?; - Ok(self.produced.saturating_sub(self.resume_from)) + Ok(()) } } @@ -617,20 +623,28 @@ mod tests { } #[tokio::test] - async fn a_path_upload_reports_the_nar_bytes_it_read() { + async fn a_path_upload_reports_its_grant_and_then_the_nar_bytes_it_read() { let dir = make_temp_store_path(); let path = dir.to_str().unwrap().to_owned(); - let mut read = Vec::new(); + let mut events = Vec::new(); let served = served( GrantTarget::Passthrough { resume_offset: 0 }, async |uploads| { let source = NarSource::Path { meta: None }; - upload_nar(uploads, "job-read", &path, source, &mut |n| read.push(n)).await + upload_nar(uploads, "job-read", &path, source, &mut |e| events.push(e)).await }, ) .await; + assert_eq!(events.first(), Some(&UploadEvent::Granted)); + let read: Vec = events[1..] + .iter() + .map(|e| match e { + UploadEvent::Read(n) => *n, + UploadEvent::Granted => panic!("a second grant: {events:?}"), + }) + .collect(); assert_eq!(read.last().copied(), Some(served.nar().nar_size)); assert!(read.is_sorted()); let _ = std::fs::remove_dir_all(&dir); diff --git a/backend/gradient-worker-client/src/nar_recv.rs b/backend/gradient-worker-client/src/nar_recv.rs index 8a785e34e..707b92ba7 100644 --- a/backend/gradient-worker-client/src/nar_recv.rs +++ b/backend/gradient-worker-client/src/nar_recv.rs @@ -8,7 +8,6 @@ use gradient_util::sync::Mutex; use std::collections::HashMap; use std::path::PathBuf; use std::sync::{Arc, Weak}; -use std::time::Instant; use anyhow::Result; use bytes::Bytes; @@ -263,7 +262,6 @@ async fn stage_pull( mut rx: mpsc::Receiver, ) { let key = &spec.key; - let mut started: Option = None; while let Some(NarChunk { data, @@ -285,7 +283,6 @@ async fn stage_pull( } if !data.is_empty() { - started.get_or_insert_with(Instant::now); if let Err(e) = sink.append(offset, &data).await { stager .abandon(&spec, format!("partial append failed: {e}")) @@ -302,9 +299,6 @@ async fn stage_pull( } let staged = sink.len(); - if let Some(start) = started { - crate::throughput::NETWORK.observe_transfer(staged, start.elapsed()); - } if let Some(total) = spec.expected && staged != total diff --git a/backend/gradient-worker-client/src/object_put.rs b/backend/gradient-worker-client/src/object_put.rs index ab6957d05..39b215e11 100644 --- a/backend/gradient-worker-client/src/object_put.rs +++ b/backend/gradient-worker-client/src/object_put.rs @@ -84,12 +84,10 @@ async fn try_put( body: Bytes, content_type: Option<&str>, ) -> std::result::Result, PutError> { - let size = body.len() as u64; let mut request = crate::http::client().put(url).body(body); if let Some(content_type) = content_type { request = request.header(CONTENT_TYPE, content_type); } - let started = std::time::Instant::now(); let resp = request.send().await.map_err(|e| PutError::Retryable { error: anyhow::Error::new(e).context("object PUT failed to send"), retry_after: None, @@ -97,7 +95,6 @@ async fn try_put( let status = resp.status(); if status.is_success() { - crate::throughput::NETWORK.observe_transfer(size, started.elapsed()); return Ok(resp .headers() .get(ETAG) @@ -168,21 +165,6 @@ mod tests { assert_eq!(requests(&server).await, 3); } - #[tokio::test] - async fn a_landed_put_feeds_the_network_throughput() { - let server = MockServer::start().await; - Mock::given(method("PUT")) - .respond_with(ResponseTemplate::new(200)) - .mount(&server) - .await; - - put_with(&IMMEDIATE, &server.uri(), Bytes::from_static(b"nar"), None) - .await - .unwrap(); - - assert!(crate::throughput::NETWORK.current().is_some()); - } - #[tokio::test] async fn a_put_that_stays_throttled_fails_after_its_attempts() { let server = MockServer::start().await; diff --git a/backend/gradient-worker-client/src/testing.rs b/backend/gradient-worker-client/src/testing.rs index ce257254c..f5214a7c9 100644 --- a/backend/gradient-worker-client/src/testing.rs +++ b/backend/gradient-worker-client/src/testing.rs @@ -227,6 +227,7 @@ impl ProtoPeer { missing_paths: vec![], spans: vec![], elapsed_ms: 0, + metrics: None, }) .await } diff --git a/backend/gradient-worker-client/src/throughput.rs b/backend/gradient-worker-client/src/throughput.rs index 1104b73b9..a67712040 100644 --- a/backend/gradient-worker-client/src/throughput.rs +++ b/backend/gradient-worker-client/src/throughput.rs @@ -8,7 +8,11 @@ use std::sync::atomic::{AtomicU64, Ordering}; const ALPHA: f64 = 0.3; -pub static NETWORK: ThroughputEwma = ThroughputEwma::new(); +/// Below this size the connection setup and round trips are dominating the elapsed time. +const MIN_TRANSFER_BYTES: u64 = 1024 * 1024; + +pub static UPLOAD: ThroughputEwma = ThroughputEwma::new(); +pub static DOWNLOAD: ThroughputEwma = ThroughputEwma::new(); pub static DISK: ThroughputEwma = ThroughputEwma::new(); /// The `0` bit pattern is marking "no sample yet". @@ -46,6 +50,10 @@ impl ThroughputEwma { } pub fn observe_transfer(&self, bytes: u64, elapsed: std::time::Duration) { + if bytes < MIN_TRANSFER_BYTES { + return; + } + self.observe(bytes as f64 * 8.0 / elapsed.as_secs_f64().max(1e-6) / 1_000_000.0); } @@ -93,8 +101,15 @@ mod tests { #[test] fn a_transfer_is_observed_in_megabits_per_second() { let e = ThroughputEwma::new(); - e.observe_transfer(1_000_000, std::time::Duration::from_secs(1)); - assert_eq!(e.current(), Some(8.0)); + e.observe_transfer(4_000_000, std::time::Duration::from_secs(2)); + assert_eq!(e.current(), Some(16.0)); + } + + #[test] + fn a_transfer_under_a_mebibyte_is_ignored() { + let e = ThroughputEwma::new(); + e.observe_transfer(MIN_TRANSFER_BYTES - 1, std::time::Duration::from_millis(1)); + assert_eq!(e.current(), None); } #[test] diff --git a/backend/gradient-worker/src/config.rs b/backend/gradient-worker/src/config.rs index 03853cbd2..93cdb64b1 100644 --- a/backend/gradient-worker/src/config.rs +++ b/backend/gradient-worker/src/config.rs @@ -353,23 +353,6 @@ pub struct BuildArgs { /// The default is all available cores. #[arg(long = "build-max-cores", env = "GRADIENT_WORKER_BUILD_MAX_CORES")] pub max_cores: Option, - - /// Capture per-build resource metrics (peak RAM, CPU time, disk I/O) from the build's cgroup. - /// The daemon must enable Nix's experimental `use-cgroups` feature. - #[arg( - long = "build-metrics", - env = "GRADIENT_WORKER_BUILD_METRICS", - default_value = "false" - )] - pub metrics: bool, - - /// The nix daemon's cgroup holding each build's `nix-build@-` cgroup. - #[arg( - long = "build-cgroup-root", - env = "GRADIENT_WORKER_BUILD_CGROUP_ROOT", - default_value = "/sys/fs/cgroup/system.slice/nix-daemon.service" - )] - pub cgroup_root: String, } impl Default for BuildArgs { @@ -377,8 +360,6 @@ impl Default for BuildArgs { Self { max_concurrent: 1, max_cores: None, - metrics: false, - cgroup_root: "/sys/fs/cgroup/system.slice/nix-daemon.service".to_owned(), } } } @@ -550,6 +531,12 @@ impl WorkerConfig { self.build.max_cores.unwrap_or(0) } + pub fn cpu_core_score(&self) -> u32 { + self.system + .cpu_core_score + .unwrap_or_else(crate::metrics::cpu_core_score) + } + pub fn capabilities(&self) -> GradientCapabilities { GradientCapabilities { core: false, diff --git a/backend/gradient-worker/src/executor/build.rs b/backend/gradient-worker/src/executor/build.rs index ca947a3a8..c3e3b3ebc 100644 --- a/backend/gradient-worker/src/executor/build.rs +++ b/backend/gradient-worker/src/executor/build.rs @@ -25,9 +25,7 @@ use tracing::{debug, info, warn}; use crate::nix::store::LocalNixStore; use crate::proto::job::JobUpdater; -use super::build_metrics::{ - CgroupSampler, NetworkPeakSampler, assemble_build_metrics, daemon_cpu_usec, -}; +use super::build_metrics::{BuildHost, RUNNING_BUILDS, ResourceUsage, build_metrics}; use super::derivation::get_basic_derivation; pub use super::failure::BuildError; use super::failure::classify_build_error; @@ -66,7 +64,7 @@ impl ParsedDerivation { reason = "arg-heavy; refactor tracked in #503" )] pub(super) async fn realize( - self, + &self, store: &LocalNixStore, task_index: u32, updater: &mut JobUpdater, @@ -74,9 +72,9 @@ impl ParsedDerivation { max_silent_secs: Option, abort: &mut watch::Receiver, log_limits: crate::executor::log_limit::LogRateLimits, - log_fetch_from_store: bool, build_cores: u32, - ) -> Result<(Vec, bool, Option), BuildError> { + mode: BuildMode, + ) -> Result { let mut guard = store.acquire().await.map_err(BuildError::transient)?; debug!( @@ -121,7 +119,7 @@ impl ParsedDerivation { let abort_ref = &mut *abort; let drained = guard .execute(|client| async move { - let logs = client.build_derivation(harmonia_path, basic_drv, BuildMode::Normal); + let logs = client.build_derivation(harmonia_path, basic_drv, mode); let mut logs = pin!(logs); match drain_build_logs_with_timeout( logs.as_mut(), @@ -150,21 +148,28 @@ impl ParsedDerivation { )) })?; - let result = match drained { - Drained::Completed(r) => r, + match drained { + Drained::Completed(result) => Ok(result), Drained::Aborted => { guard.mark_broken(); - return Err(BuildError::aborted(drv_path)); + Err(BuildError::aborted(drv_path)) } Drained::Timeout(e) => { guard.mark_broken(); - return Err(BuildError::timeout(e)); + Err(BuildError::timeout(e)) } - }; - - let cpu_usec = daemon_cpu_usec(result.cpu_user, result.cpu_system); + } + } - match result.inner { + pub(super) async fn outputs( + &self, + result: BuildResultInner, + updater: &mut JobUpdater, + task_index: u32, + drv_path: &str, + log_fetch_from_store: bool, + ) -> Result<(Vec, bool), BuildError> { + match result { BuildResultInner::Success(s) => { info!(drv = %drv_path, "build succeeded"); let pairs = output_pairs_from_built_or_drv(&s.built_outputs, &self.drv); @@ -204,7 +209,7 @@ impl ParsedDerivation { products, }); } - Ok((outputs, substituted, cpu_usec)) + Ok((outputs, substituted)) } BuildResultInner::Failure(f) => { @@ -270,6 +275,15 @@ pub(super) async fn load_products(store_path: &str) -> Vec { products } +async fn build_mode(store: &LocalNixStore, task: &BuildSpec) -> anyhow::Result { + for output in task.outputs.iter().filter(|o| !o.path.is_empty()) { + if store.is_hidden(&output.path).await? { + return Ok(BuildMode::Repair); + } + } + Ok(BuildMode::Normal) +} + #[allow( clippy::too_many_arguments, reason = "arg-heavy; refactor tracked in #503" @@ -280,15 +294,16 @@ pub async fn build_derivation( task_index: u32, updater: &mut JobUpdater, abort: &mut watch::Receiver, - build_metrics: bool, - cgroup_root: &str, log_limits: crate::executor::log_limit::LogRateLimits, log_fetch_from_store: bool, - build_cores: u32, + host: BuildHost, ) -> Result, BuildError> { let parsed = ParsedDerivation::load(&task.drv_path) .await .map_err(BuildError::transient)?; + let mode = build_mode(store, task) + .await + .map_err(BuildError::transient)?; let realize = parsed.realize( store, @@ -298,38 +313,46 @@ pub async fn build_derivation( task.max_silent_secs, abort, log_limits, - log_fetch_from_store, - build_cores, + host.build_cores, + mode, ); - let net_sampler = build_metrics.then(NetworkPeakSampler::start); - let cgroup_sampler = build_metrics.then(|| CgroupSampler::start(cgroup_root, &task.drv_path)); + let running = RUNNING_BUILDS.start(); let started = std::time::Instant::now(); - let realize_result: Result<(Vec, bool, Option), BuildError> = - match task.timeout_secs.map(std::time::Duration::from_secs) { - Some(d) => match tokio::time::timeout(d, realize).await { - Ok(r) => r, - Err(_) => Err(BuildError::timeout(anyhow::anyhow!( - "build exceeded wall-clock timeout of {}s", - d.as_secs() - ))), - }, - None => realize.await, - }; - - let build_time_ms = started.elapsed().as_millis() as u64; - let peak_network_mbps = match net_sampler { - Some(s) => s.finish().await, - None => None, + let realized = match task.timeout_secs.map(std::time::Duration::from_secs) { + Some(d) => tokio::time::timeout(d, realize).await.unwrap_or_else(|_| { + Err(BuildError::timeout(anyhow::anyhow!( + "build exceeded wall-clock timeout of {}s", + d.as_secs() + ))) + }), + None => realize.await, }; - let cgroup_raw = match cgroup_sampler { - Some(s) => s.finish().await, - None => None, + + let usage = realized.as_ref().map(ResourceUsage::of).unwrap_or_default(); + let metrics = build_metrics(usage, started.elapsed().as_millis() as u64, host, &running); + drop(running); + let built = match realized { + Ok(result) => { + parsed + .outputs( + result.inner, + updater, + task_index, + &task.drv_path, + log_fetch_from_store, + ) + .await + } + Err(e) => Err(e), }; - let cpu_usec = realize_result.as_ref().ok().and_then(|(_, _, c)| *c); - let metrics = assemble_build_metrics(cgroup_raw, cpu_usec, build_time_ms, peak_network_mbps); - let (outputs, substituted, _) = realize_result?; + let (outputs, substituted) = built.map_err(|e| e.with_metrics(metrics.clone()))?; + let output_paths: Vec = outputs.iter().map(|o| o.store_path.clone()).collect(); + store + .reveal(&output_paths) + .await + .map_err(BuildError::transient)?; updater .report_build_output( task.build_id.clone(), diff --git a/backend/gradient-worker/src/executor/build_metrics.rs b/backend/gradient-worker/src/executor/build_metrics.rs index 6718df10e..1d590f3ec 100644 --- a/backend/gradient-worker/src/executor/build_metrics.rs +++ b/backend/gradient-worker/src/executor/build_metrics.rs @@ -4,68 +4,37 @@ * SPDX-License-Identifier: AGPL-3.0-only */ -use gradient_wire::messages::BuildMetrics; -use harmonia_protocol::daemon_wire::types2::Microseconds; -use std::path::{Path, PathBuf}; -use tracing::debug; +use std::sync::atomic::{AtomicU32, Ordering}; -use crate::metrics::cgroup::{BuildMetricsRaw, read_build_cgroup}; +use gradient_wire::messages::BuildMetrics; +use harmonia_protocol::daemon_wire::types2::{BuildResult, Microseconds}; const BYTES_PER_MB: u64 = 1_048_576; -const CGROUP_SAMPLE_MS: u64 = 200; -fn raw_to_build_metrics( - raw: Option, - build_time_ms: u64, - cpu_count: u32, - peak_network_mbps: Option, -) -> BuildMetrics { - let Some(raw) = raw else { - return BuildMetrics { - build_time_ms: Some(build_time_ms), - peak_network_mbps, - ..Default::default() - }; - }; +/// The daemon is reading these from the build's cgroup. A daemon without cgroups or without the +/// `build-resource-usage` feature is leaving all but the CPU times empty. +#[derive(Debug, Default, Clone, Copy, PartialEq, Eq)] +pub(super) struct ResourceUsage { + cpu_usec: Option, + memory_peak: Option, + io_read_bytes: Option, + io_write_bytes: Option, + oom_kills: Option, +} - let cpu_time_ms = raw.cpu_usage_usec.map(|u| u / 1000); - let avg_cpu_pct = match cpu_time_ms { - Some(cpu_ms) if build_time_ms > 0 && cpu_count > 0 => { - Some(cpu_ms as f32 / (build_time_ms as f32 * cpu_count as f32) * 100.0) +impl ResourceUsage { + pub(super) fn of(result: &BuildResult) -> Self { + Self { + cpu_usec: cpu_usec(result.cpu_user, result.cpu_system), + memory_peak: result.memory_peak, + io_read_bytes: result.io_read_bytes, + io_write_bytes: result.io_write_bytes, + oom_kills: result.oom_kills, } - _ => None, - }; - - BuildMetrics { - peak_ram_mb: raw.peak_ram_bytes.map(|b| b / BYTES_PER_MB), - cpu_time_ms, - avg_cpu_pct, - disk_read_bytes: Some(raw.disk_read_bytes), - disk_write_bytes: Some(raw.disk_write_bytes), - oom_killed: raw.oom_killed, - build_time_ms: Some(build_time_ms), - peak_network_mbps, } } -fn build_cgroup(root: &Path, drv_path: &str) -> Option { - let (hash, _) = Path::new(drv_path).file_name()?.to_str()?.split_once('-')?; - let prefix = format!("nix-build@{hash}-"); - std::fs::read_dir(root) - .ok()? - .flatten() - .map(|entry| entry.path()) - .find(|path| { - path.file_name() - .and_then(|name| name.to_str()) - .is_some_and(|name| name.starts_with(&prefix)) - }) -} - -pub(super) fn daemon_cpu_usec( - user: Option, - system: Option, -) -> Option { +fn cpu_usec(user: Option, system: Option) -> Option { let us = |m: Microseconds| m.0.max(0) as u64; match (user, system) { (None, None) => None, @@ -73,271 +42,181 @@ pub(super) fn daemon_cpu_usec( } } -pub(super) fn assemble_build_metrics( - sampled: Option, - cpu_usec: Option, - build_time_ms: u64, - peak_network_mbps: Option, -) -> BuildMetrics { - let cpu_count = crate::metrics::host_static().cpu_count; - let raw = match (sampled, cpu_usec) { - (None, None) => None, - (s, cpu) => { - let s = s.unwrap_or_default(); - Some(BuildMetricsRaw { - cpu_usage_usec: cpu.or(s.cpu_usage_usec), - ..s - }) - } - }; - if let Some(r) = raw.as_ref() { - let bytes = r.disk_read_bytes + r.disk_write_bytes; - if build_time_ms > 0 && bytes > 0 { - let mb_per_s = (bytes as f64 / 1_048_576.0) / (build_time_ms as f64 / 1000.0); - gradient_worker_client::throughput::DISK.observe(mb_per_s); - } - } - raw_to_build_metrics(raw, build_time_ms, cpu_count, peak_network_mbps) +#[derive(Debug, Clone, Copy)] +pub struct BuildHost { + pub build_cores: u32, + pub cpu_core_score: u32, } -/// The peak is host-level because cgroup v2 is carrying no per-build network accounting. It is -/// exact only when the build is the sole network consumer. -pub(super) struct NetworkPeakSampler { - peak: std::sync::Arc, - stop: std::sync::Arc, - handle: tokio::task::JoinHandle<()>, -} +pub(super) static RUNNING_BUILDS: RunningBuilds = RunningBuilds::new(); -impl NetworkPeakSampler { - pub(super) fn start() -> Self { - use std::sync::Arc; - use std::sync::atomic::{AtomicBool, AtomicU64, Ordering}; - let peak = Arc::new(AtomicU64::new(0)); - let stop = Arc::new(AtomicBool::new(false)); - let (p, s) = (peak.clone(), stop.clone()); - #[expect( - clippy::disallowed_methods, - reason = "stopped through the flag when the build ends" - )] - let handle = tokio::spawn(async move { - while !s.load(Ordering::Relaxed) { - if let Some(v) = gradient_worker_client::throughput::NETWORK.current() { - let bits = (v as f64).to_bits(); - let mut prev = p.load(Ordering::Relaxed); - while f64::from_bits(prev) < v as f64 { - match p.compare_exchange_weak( - prev, - bits, - Ordering::Relaxed, - Ordering::Relaxed, - ) { - Ok(_) => break, - Err(cur) => prev = cur, - } - } - } - tokio::time::sleep(std::time::Duration::from_millis(250)).await; - } - }); - Self { peak, stop, handle } +pub(super) struct RunningBuilds(AtomicU32); + +impl RunningBuilds { + const fn new() -> Self { + Self(AtomicU32::new(0)) } - pub(super) async fn finish(self) -> Option { - use std::sync::atomic::Ordering; - self.stop.store(true, Ordering::Relaxed); - let _ = self.handle.await; - match self.peak.load(Ordering::Relaxed) { - 0 => None, - b => Some(f64::from_bits(b) as f32), + pub(super) fn start(&self) -> RunningBuild<'_> { + RunningBuild { + others: self.0.fetch_add(1, Ordering::Relaxed), + builds: self, } } } -/// Nix is destroying the build cgroup as soon as the build is done. The sampler must read it while -/// the build is running and keep the last good reading. -pub(super) struct CgroupSampler { - stop: std::sync::Arc, - handle: tokio::task::JoinHandle>, +pub(super) struct RunningBuild<'a> { + others: u32, + builds: &'a RunningBuilds, } -impl CgroupSampler { - pub(super) fn start(cgroup_root: &str, drv_path: &str) -> Self { - use std::sync::Arc; - use std::sync::atomic::{AtomicBool, Ordering}; - let stop = Arc::new(AtomicBool::new(false)); - let s = stop.clone(); - let root = PathBuf::from(cgroup_root); - let drv_path = drv_path.to_owned(); - #[expect( - clippy::disallowed_methods, - reason = "stopped through the flag when the build ends" - )] - let handle = tokio::spawn(async move { - let mut cgroup: Option = None; - let mut last: Option = None; - while !s.load(Ordering::Relaxed) { - let root = root.clone(); - let drv_path = drv_path.clone(); - let known = cgroup.clone(); - let sample = tokio::task::spawn_blocking(move || { - let cgroup = known.or_else(|| build_cgroup(&root, &drv_path))?; - let raw = read_build_cgroup(&cgroup); - Some((cgroup, raw)) - }) - .await - .ok() - .flatten(); - - match sample { - Some((dir, Some(cur))) => { - if cgroup.is_none() { - debug!(cgroup = %dir.display(), "sampling build cgroup for metrics"); - } - cgroup = Some(dir); - last = Some(merge_cgroup_sample(last, cur)); - } - Some((_, None)) => break, - None => {} - } - tokio::time::sleep(std::time::Duration::from_millis(CGROUP_SAMPLE_MS)).await; - } - last - }); - Self { stop, handle } +impl Drop for RunningBuild<'_> { + fn drop(&mut self) { + self.builds.0.fetch_sub(1, Ordering::Relaxed); } +} - pub(super) async fn finish(self) -> Option { - use std::sync::atomic::Ordering; - self.stop.store(true, Ordering::Relaxed); - self.handle.await.ok().flatten() +pub(super) fn build_metrics( + usage: ResourceUsage, + build_time_ms: u64, + host: BuildHost, + running: &RunningBuild<'_>, +) -> BuildMetrics { + observe_disk_speed(usage, build_time_ms); + let cpu_count = crate::metrics::host_static().cpu_count; + BuildMetrics { + concurrent_builds: Some(running.others), + build_cores: Some(effective_cores(host.build_cores, cpu_count)), + cpu_core_score: Some(host.cpu_core_score), + ..metrics_from(usage, build_time_ms, cpu_count) } } -fn merge_cgroup_sample(prev: Option, cur: BuildMetricsRaw) -> BuildMetricsRaw { - let prev = prev.unwrap_or_default(); - BuildMetricsRaw { - peak_ram_bytes: prev.peak_ram_bytes.max(cur.peak_ram_bytes), - cpu_usage_usec: cur.cpu_usage_usec.or(prev.cpu_usage_usec), - disk_read_bytes: cur.disk_read_bytes, - disk_write_bytes: cur.disk_write_bytes, - oom_killed: prev.oom_killed || cur.oom_killed, +fn effective_cores(build_cores: u32, cpu_count: u32) -> u32 { + if build_cores == 0 { + cpu_count + } else { + build_cores.min(cpu_count) } } -#[cfg(test)] -mod tests { - use super::*; - - const DRV: &str = "/nix/store/0123456789abcdfghijklmnpqrsvwxyz-hello.drv"; +fn observe_disk_speed(usage: ResourceUsage, build_time_ms: u64) { + let bytes = usage.io_read_bytes.unwrap_or(0) + usage.io_write_bytes.unwrap_or(0); + if build_time_ms > 0 && bytes > 0 { + let mb_per_s = (bytes as f64 / BYTES_PER_MB as f64) / (build_time_ms as f64 / 1000.0); + gradient_worker_client::throughput::DISK.observe(mb_per_s); + } +} - #[test] - fn build_cgroup_matches_the_derivation_among_siblings() { - let root = tempfile::tempdir().unwrap(); - for name in [ - "nix-daemon", - "nix-build@zyxwvsrqpnmlkjihgfdcba9876543210-30001", - "nix-build@0123456789abcdfghijklmnpqrsvwxyz-30002", - ] { - std::fs::create_dir(root.path().join(name)).unwrap(); +fn metrics_from(usage: ResourceUsage, build_time_ms: u64, cpu_count: u32) -> BuildMetrics { + let cpu_time_ms = usage.cpu_usec.map(|u| u / 1000); + let avg_cpu_pct = match cpu_time_ms { + Some(cpu_ms) if build_time_ms > 0 && cpu_count > 0 => { + Some(cpu_ms as f32 / (build_time_ms as f32 * cpu_count as f32) * 100.0) } - assert_eq!( - build_cgroup(root.path(), DRV), - Some( - root.path() - .join("nix-build@0123456789abcdfghijklmnpqrsvwxyz-30002") - ), - ); - } + _ => None, + }; - #[test] - fn build_cgroup_none_before_the_build_starts() { - let root = tempfile::tempdir().unwrap(); - std::fs::create_dir(root.path().join("nix-daemon")).unwrap(); - assert!(build_cgroup(root.path(), DRV).is_none()); - assert!(build_cgroup(&root.path().join("missing"), DRV).is_none()); + BuildMetrics { + peak_ram_mb: usage.memory_peak.map(|b| b / BYTES_PER_MB), + cpu_time_ms, + avg_cpu_pct, + disk_read_bytes: usage.io_read_bytes, + disk_write_bytes: usage.io_write_bytes, + oom_killed: usage.oom_kills.is_some_and(|kills| kills > 0), + build_time_ms: Some(build_time_ms), + ..Default::default() } +} + +#[cfg(test)] +mod tests { + use super::*; #[test] - fn daemon_cpu_usec_sums_present_fields() { - assert_eq!(daemon_cpu_usec(None, None), None); - assert_eq!(daemon_cpu_usec(Some(Microseconds(700)), None), Some(700)); + fn cpu_usec_sums_present_fields() { + assert_eq!(cpu_usec(None, None), None); + assert_eq!(cpu_usec(Some(Microseconds(700)), None), Some(700)); assert_eq!( - daemon_cpu_usec(Some(Microseconds(700)), Some(Microseconds(300))), + cpu_usec(Some(Microseconds(700)), Some(Microseconds(300))), Some(1000), ); assert_eq!( - daemon_cpu_usec(Some(Microseconds(-1)), Some(Microseconds(5))), + cpu_usec(Some(Microseconds(-1)), Some(Microseconds(5))), Some(5) ); } #[test] - fn raw_to_metrics_always_sets_build_time() { - let m = raw_to_build_metrics(None, 5_000, 4, None); - assert_eq!(m.build_time_ms, Some(5_000)); - assert_eq!(m.peak_ram_mb, None); - assert_eq!(m.cpu_time_ms, None); - assert_eq!(m.avg_cpu_pct, None); - assert!(!m.oom_killed); + fn a_daemon_without_resource_usage_leaves_only_the_build_time() { + let m = metrics_from(ResourceUsage::default(), 5_000, 4); + assert_eq!( + m, + BuildMetrics { + build_time_ms: Some(5_000), + ..Default::default() + } + ); } #[test] - fn raw_to_metrics_handles_zero_divisors() { - let raw = BuildMetricsRaw { - peak_ram_bytes: Some(2 * BYTES_PER_MB), - cpu_usage_usec: Some(1_000_000), - disk_read_bytes: 10, - disk_write_bytes: 20, - oom_killed: false, + fn the_daemon_usage_becomes_the_build_metrics() { + let usage = ResourceUsage { + cpu_usec: Some(8_000_000), + memory_peak: Some(3 * BYTES_PER_MB + 1), + io_read_bytes: Some(10), + io_write_bytes: Some(20), + oom_kills: Some(1), }; - let m = raw_to_build_metrics(Some(raw), 0, 4, None); - assert_eq!(m.avg_cpu_pct, None); - let m = raw_to_build_metrics(Some(raw), 1_000, 0, None); - assert_eq!(m.avg_cpu_pct, None); - assert_eq!(m.peak_ram_mb, Some(2)); - assert_eq!(m.cpu_time_ms, Some(1_000)); - assert_eq!(m.disk_read_bytes, Some(10)); - assert_eq!(m.disk_write_bytes, Some(20)); + assert_eq!( + metrics_from(usage, 4_000, 4), + BuildMetrics { + peak_ram_mb: Some(3), + cpu_time_ms: Some(8_000), + avg_cpu_pct: Some(50.0), + disk_read_bytes: Some(10), + disk_write_bytes: Some(20), + oom_killed: true, + build_time_ms: Some(4_000), + ..Default::default() + } + ); } #[test] - fn raw_to_metrics_computes_avg_cpu_pct() { - let raw = BuildMetricsRaw { - peak_ram_bytes: None, - cpu_usage_usec: Some(8_000_000), - disk_read_bytes: 0, - disk_write_bytes: 0, - oom_killed: false, + fn no_out_of_memory_kill_is_not_an_out_of_memory_build() { + let usage = ResourceUsage { + oom_kills: Some(0), + ..Default::default() }; - let m = raw_to_build_metrics(Some(raw), 4_000, 4, Some(125.0)); - assert_eq!(m.cpu_time_ms, Some(8_000)); - assert_eq!(m.avg_cpu_pct, Some(50.0)); - assert_eq!(m.peak_network_mbps, Some(125.0)); + assert!(!metrics_from(usage, 1_000, 4).oom_killed); } - fn sample(cpu_usage_usec: Option) -> BuildMetricsRaw { - BuildMetricsRaw { - peak_ram_bytes: Some(BYTES_PER_MB), - cpu_usage_usec, - disk_read_bytes: 1, - disk_write_bytes: 2, - oom_killed: false, - } + #[test] + fn a_build_counts_the_builds_running_beside_it() { + let builds = RunningBuilds::new(); + let first = builds.start(); + let second = builds.start(); + drop(first); + let third = builds.start(); + + assert_eq!((second.others, third.others), (1, 1)); } #[test] - fn merge_keeps_the_latest_cumulative_cpu_usage() { - let merged = merge_cgroup_sample(Some(sample(Some(1_000))), sample(Some(5_000))); - assert_eq!(merged.cpu_usage_usec, Some(5_000)); - let merged = merge_cgroup_sample(Some(sample(Some(5_000))), sample(None)); - assert_eq!(merged.cpu_usage_usec, Some(5_000)); + fn zero_build_cores_is_every_core_and_a_larger_value_is_capped() { + assert_eq!(effective_cores(0, 16), 16); + assert_eq!(effective_cores(4, 16), 4); + assert_eq!(effective_cores(64, 16), 16); } #[test] - fn assemble_falls_back_to_the_sampled_cpu_without_daemon_times() { - let m = assemble_build_metrics(Some(sample(Some(3_000_000))), None, 1_000, None); - assert_eq!(m.cpu_time_ms, Some(3_000)); - let m = assemble_build_metrics(Some(sample(Some(3_000_000))), Some(4_000_000), 1_000, None); - assert_eq!(m.cpu_time_ms, Some(4_000)); + fn the_cpu_share_needs_a_build_time_and_cores() { + let usage = ResourceUsage { + cpu_usec: Some(1_000_000), + ..Default::default() + }; + assert_eq!(metrics_from(usage, 0, 4).avg_cpu_pct, None); + assert_eq!(metrics_from(usage, 1_000, 0).avg_cpu_pct, None); } } diff --git a/backend/gradient-worker/src/executor/compress.rs b/backend/gradient-worker/src/executor/compress.rs index 5adb19a26..478ca2843 100644 --- a/backend/gradient-worker/src/executor/compress.rs +++ b/backend/gradient-worker/src/executor/compress.rs @@ -13,7 +13,7 @@ use std::collections::{BTreeMap, HashMap}; use anyhow::Result; use gradient_util::store_path::nix_store_path; -use gradient_wire::messages::{BuildProgressPhase, CachedPath}; +use gradient_wire::messages::{BuildProgressPhase, BuildStage, CachedPath}; use tokio::sync::watch; use super::NarUpload; @@ -36,7 +36,7 @@ pub async fn push_outputs( return Ok(UploadedNar::default()); } - updater.report_compressing().await?; + updater.report_stage(BuildStage::Upload).await?; let paths: Vec = outputs .iter() .map(|o| nix_store_path(&o.store_path)) diff --git a/backend/gradient-worker/src/executor/failure.rs b/backend/gradient-worker/src/executor/failure.rs index 3598c1ee0..b43b46eb1 100644 --- a/backend/gradient-worker/src/executor/failure.rs +++ b/backend/gradient-worker/src/executor/failure.rs @@ -4,7 +4,7 @@ * SPDX-License-Identifier: AGPL-3.0-only */ -use gradient_wire::messages::BuildFailureKind; +use gradient_wire::messages::{BuildFailureKind, BuildMetrics}; use gradient_worker_client::connection::{Unresponsive, WriterUnavailable}; use crate::executor::eval::CorruptEvalCache; @@ -15,6 +15,7 @@ pub struct BuildError { pub kind: BuildFailureKind, pub source: anyhow::Error, pub missing_paths: Vec, + pub metrics: Option>, } impl std::fmt::Display for BuildError { @@ -30,6 +31,13 @@ impl BuildError { kind, source, missing_paths: Vec::new(), + metrics: None, + } + } + pub(super) fn with_metrics(self, metrics: BuildMetrics) -> Self { + Self { + metrics: Some(Box::new(metrics)), + ..self } } pub(crate) fn transient(e: impl Into) -> Self { @@ -52,6 +60,7 @@ impl BuildError { kind: BuildFailureKind::InputsUnavailable, source: e.into(), missing_paths, + metrics: None, } } /// An abort must never be reported as `Permanent`. @@ -179,6 +188,11 @@ pub(crate) fn wire_failure(e: &anyhow::Error) -> (BuildFailureKind, Vec) } } +pub(crate) fn failure_metrics(e: &anyhow::Error) -> Option { + e.downcast_ref::() + .and_then(|be| be.metrics.as_deref().cloned()) +} + #[cfg(test)] mod tests { use super::*; @@ -252,6 +266,20 @@ mod tests { assert_eq!(missing, vec!["/nix/store/x-y".to_owned()]); } + #[test] + fn a_failed_build_reports_the_metrics_of_its_attempt() { + let metrics = BuildMetrics { + peak_ram_mb: Some(7_000), + oom_killed: true, + ..Default::default() + }; + let e: anyhow::Error = BuildError::transient(anyhow::anyhow!("Killed")) + .with_metrics(metrics.clone()) + .into(); + assert_eq!(failure_metrics(&e), Some(metrics)); + assert_eq!(failure_metrics(&anyhow::anyhow!("eval failed")), None); + } + #[test] fn wire_failure_maps_corrupt_eval_cache_with_fingerprint() { let e = anyhow::Error::new(CorruptEvalCache { diff --git a/backend/gradient-worker/src/executor/mod.rs b/backend/gradient-worker/src/executor/mod.rs index 7f6ac1e5c..c123326ec 100644 --- a/backend/gradient-worker/src/executor/mod.rs +++ b/backend/gradient-worker/src/executor/mod.rs @@ -19,16 +19,20 @@ mod substitute; pub mod timeline; use std::sync::Arc; +use std::time::Instant; use anyhow::Result; +use gradient_util::sync::Mutex; use gradient_wire::messages::{ - BuildJob, BuildOutput, BuildProgressPhase, BuildSpec, BuildSpecKind, FlakeJob, FlakeStep, + BuildJob, BuildOutput, BuildProgressPhase, BuildSpec, BuildSpecKind, BuildStage, FlakeJob, + FlakeStep, }; use tokio::sync::watch; use tracing::instrument; use gradient_wire::messages::JobPhase; +use crate::executor::timeline::PhaseGuard; use crate::nix::gcroots::{GcRootHandle, GcRootKeeper}; use crate::nix::store::LocalNixStore; use crate::proto::progress::{Progress, Tally}; @@ -37,6 +41,7 @@ use gradient_wire::messages::CachedPath; use gradient_wire::traits::WorkerStore; use gradient_worker_client::nar; +pub use build_metrics::BuildHost; pub use eval::WorkerEvaluator; async fn query_fetched_paths( @@ -63,6 +68,7 @@ pub(crate) async fn push_paths( let mut guard = updater.phase(JobPhase::DrvClosurePush); guard.record(paths.len() as u32, 0); let (paths, sizes): (Vec, Vec>) = paths.iter().cloned().unzip(); + store.reveal(&paths).await?; let cache_entries = query_fetched_paths(updater, paths, sizes).await?; upload_all(updater, pair_with_store(cache_entries, store), None).await?; Ok(()) @@ -85,7 +91,11 @@ fn pair_with_store<'a>(entries: Vec, store: &'a LocalNixStore) -> Ve .collect() } -async fn upload_one_nar(updater: &JobUpdater, upload: NarUpload<'_>) -> Result { +async fn upload_one_nar( + updater: &JobUpdater, + upload: NarUpload<'_>, + spans: &PushSpans<'_>, +) -> Result { let cp = &upload.cached; if cp.cached { tracing::debug!(store_path = %cp.path, "skipping NAR upload - already cached"); @@ -97,13 +107,51 @@ async fn upload_one_nar(updater: &JobUpdater, upload: NarUpload<'_>) -> Result spans.granted(), + nar::UploadEvent::Read(read) => counted.at(read), + }, ) .await?; counted.transfer_done(); Ok(uploaded) } +/// The batch is getting one span pair because the uploads overlap. Per-path spans would chart as +/// nested and double count. +struct PushSpans<'a> { + updater: &'a JobUpdater, + waiting: Mutex>, + pushing: Mutex>, +} + +impl<'a> PushSpans<'a> { + fn start(updater: &'a JobUpdater) -> Self { + Self { + updater, + waiting: Mutex::new(Some(updater.phase(JobPhase::UploadWait))), + pushing: Mutex::new(None), + } + } + + fn granted(&self) { + let Some(wait) = self.waiting.lock().take() else { + return; + }; + + drop(wait); + *self.pushing.lock() = Some((self.updater.phase(JobPhase::NarPush), Instant::now())); + } + + fn finish(self, paths: usize, uploaded: nar::UploadedNar) { + if let Some((mut push, started)) = self.pushing.into_inner() { + gradient_worker_client::throughput::UPLOAD + .observe_transfer(uploaded.nar_size, started.elapsed()); + push.record(paths as u32, uploaded.file_size); + } + } +} + pub(crate) async fn upload_all( updater: &JobUpdater, uploads: Vec>, @@ -116,20 +164,18 @@ pub(crate) async fn upload_all( return Ok(nar::UploadedNar::default()); } - // The batch is getting one span because the uploads overlap. The timeline is parenting a - // span to the innermost open one. Per-path spans would chart as nested and double count. - let mut guard = updater.phase(JobPhase::NarPush); - guard.record(pending as u32, 0); + let spans = PushSpans::start(updater); let mut uploaded = nar::UploadedNar::default(); let mut running: FuturesUnordered<_> = uploads .into_iter() - .map(|upload| upload_unless_aborted(updater, upload, abort)) + .map(|upload| upload_unless_aborted(updater, upload, abort, &spans)) .collect(); while let Some(result) = running.next().await { uploaded += result?; } - guard.record(0, uploaded.file_size); + drop(running); + spans.finish(pending, uploaded); Ok(uploaded) } @@ -137,12 +183,13 @@ async fn upload_unless_aborted( updater: &JobUpdater, upload: NarUpload<'_>, abort: Option<&watch::Receiver>, + spans: &PushSpans<'_>, ) -> Result { if let Some(abort) = abort { check_abort(abort)?; } - let result = upload_one_nar(updater, upload).await; + let result = upload_one_nar(updater, upload, spans).await; if result.is_err() && let Some(abort) = abort { @@ -207,11 +254,9 @@ pub struct JobExecutor { pub(crate) evaluator: Arc, pub(crate) gcroots: GcRootKeeper, pub(crate) binpath_ssh: String, - pub(crate) build_metrics: bool, - pub(crate) build_cgroup_root: String, pub(crate) log_limits: crate::executor::log_limit::LogRateLimits, pub(crate) log_fetch_from_store: bool, - pub(crate) build_cores: u32, + pub(crate) host: BuildHost, } impl JobExecutor { @@ -224,22 +269,18 @@ impl JobExecutor { evaluator: WorkerEvaluator, gcroots: GcRootKeeper, binpath_ssh: String, - build_metrics: bool, - build_cgroup_root: String, log_limits: crate::executor::log_limit::LogRateLimits, log_fetch_from_store: bool, - build_cores: u32, + host: BuildHost, ) -> Self { Self { store: Arc::new(store), evaluator: Arc::new(evaluator), gcroots, binpath_ssh, - build_metrics, - build_cgroup_root, log_limits, log_fetch_from_store, - build_cores, + host, } } @@ -494,6 +535,7 @@ impl JobExecutor { gc_handles.push(self.gcroots.add(&build_task.drv_path).await); + updater.report_stage(BuildStage::Prefetch).await?; { let mut prefetch = updater.phase(JobPhase::Prefetch); let fetched = @@ -503,6 +545,7 @@ impl JobExecutor { prefetch.record(fetched.paths, fetched.bytes); } + updater.report_stage(BuildStage::Build).await?; let _build = updater.phase(JobPhase::Build); let built = build::build_derivation( &self.store, @@ -510,11 +553,9 @@ impl JobExecutor { index as u32, updater, &mut abort, - self.build_metrics, - &self.build_cgroup_root, self.log_limits, self.log_fetch_from_store, - self.build_cores, + self.host, ) .await?; for o in &built { diff --git a/backend/gradient-worker/src/executor/substitute.rs b/backend/gradient-worker/src/executor/substitute.rs index 3fdf405f1..94236525e 100644 --- a/backend/gradient-worker/src/executor/substitute.rs +++ b/backend/gradient-worker/src/executor/substitute.rs @@ -167,12 +167,20 @@ impl UpstreamIo for JobUpdaterIo<'_> { progress: &mut Progress, ) -> Result>> { let _fetch = self.0.phase(JobPhase::SubstituteFetch); + let started = std::time::Instant::now(); let (_, body) = download_one_presigned( gradient_worker_client::http::download_client(), upstream.clone(), progress, ) .await?; + if body.is_some() + && let Some(nar_size) = upstream.nar_size + { + gradient_worker_client::throughput::DOWNLOAD + .observe_transfer(nar_size, started.elapsed()); + } + Ok(body.map(|(bytes, _)| bytes)) } } diff --git a/backend/gradient-worker/src/metrics/cgroup.rs b/backend/gradient-worker/src/metrics/cgroup.rs deleted file mode 100644 index dd9abcbc2..000000000 --- a/backend/gradient-worker/src/metrics/cgroup.rs +++ /dev/null @@ -1,128 +0,0 @@ -/* - * SPDX-FileCopyrightText: 2026 Wavelens GmbH - * - * SPDX-License-Identifier: AGPL-3.0-only - */ - -#[derive(Debug, Clone, Copy, Default, PartialEq)] -pub struct BuildMetricsRaw { - pub peak_ram_bytes: Option, - pub cpu_usage_usec: Option, - pub disk_read_bytes: u64, - pub disk_write_bytes: u64, - pub oom_killed: bool, -} - -pub fn parse_memory_peak(s: &str) -> Option { - s.trim().parse().ok() -} - -pub fn parse_cpu_usage_usec(s: &str) -> Option { - s.lines() - .find(|l| l.starts_with("usage_usec "))? - .split_ascii_whitespace() - .nth(1)? - .parse() - .ok() -} - -pub fn parse_io_stat(s: &str) -> (u64, u64) { - s.lines().fold((0, 0), |(rb, wb), line| { - let mut r = rb; - let mut w = wb; - for token in line.split_ascii_whitespace().skip(1) { - if let Some(v) = token - .strip_prefix("rbytes=") - .and_then(|n| n.parse::().ok()) - { - r += v; - } else if let Some(v) = token - .strip_prefix("wbytes=") - .and_then(|n| n.parse::().ok()) - { - w += v; - } - } - (r, w) - }) -} - -pub fn parse_oom_kill(s: &str) -> bool { - s.lines() - .find(|l| l.starts_with("oom_kill ")) - .and_then(|l| l.split_ascii_whitespace().nth(1)) - .and_then(|n| n.parse::().ok()) - .map(|n| n > 0) - .unwrap_or(false) -} - -pub fn read_build_cgroup(dir: &std::path::Path) -> Option { - if !dir.exists() { - return None; - } - let read = |name| std::fs::read_to_string(dir.join(name)).ok(); - let io = read("io.stat").map(|s| parse_io_stat(&s)).unwrap_or((0, 0)); - Some(BuildMetricsRaw { - peak_ram_bytes: read("memory.peak").and_then(|s| parse_memory_peak(&s)), - cpu_usage_usec: read("cpu.stat").and_then(|s| parse_cpu_usage_usec(&s)), - disk_read_bytes: io.0, - disk_write_bytes: io.1, - oom_killed: read("memory.events") - .map(|s| parse_oom_kill(&s)) - .unwrap_or(false), - }) -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn cpu_stat_usage() { - let s = "usage_usec 1234567\nuser_usec 1000000\nsystem_usec 234567\n"; - assert_eq!(parse_cpu_usage_usec(s), Some(1_234_567)); - } - - #[test] - fn io_stat_sums_devices() { - let s = "8:0 rbytes=1000 wbytes=2000 rios=1 wios=2\n8:16 rbytes=500 wbytes=500\n"; - assert_eq!(parse_io_stat(s), (1500, 2500)); - } - - #[test] - fn memory_events_oom() { - assert!(parse_oom_kill("low 0\nhigh 0\noom 1\noom_kill 3\n")); - assert!(!parse_oom_kill("low 0\noom_kill 0\n")); - } - - #[test] - fn memory_peak_value() { - assert_eq!(parse_memory_peak("4194304\n"), Some(4_194_304)); - } - - #[test] - fn read_missing_dir_is_none() { - assert!(read_build_cgroup(std::path::Path::new("/nonexistent/cgroup/xyz")).is_none()); - } - - #[test] - fn read_build_cgroup_from_tempdir() { - let dir = tempfile::tempdir().unwrap(); - let p = dir.path(); - std::fs::write(p.join("memory.peak"), "4194304\n").unwrap(); - std::fs::write( - p.join("cpu.stat"), - "usage_usec 1234567\nuser_usec 1000000\n", - ) - .unwrap(); - std::fs::write(p.join("io.stat"), "8:0 rbytes=1000 wbytes=2000\n").unwrap(); - std::fs::write(p.join("memory.events"), "oom_kill 0\n").unwrap(); - - let m = read_build_cgroup(p).unwrap(); - assert_eq!(m.peak_ram_bytes, Some(4_194_304)); - assert_eq!(m.cpu_usage_usec, Some(1_234_567)); - assert_eq!(m.disk_read_bytes, 1000); - assert_eq!(m.disk_write_bytes, 2000); - assert!(!m.oom_killed); - } -} diff --git a/backend/gradient-worker/src/metrics/mod.rs b/backend/gradient-worker/src/metrics/mod.rs index b257011d9..31e4b334c 100644 --- a/backend/gradient-worker/src/metrics/mod.rs +++ b/backend/gradient-worker/src/metrics/mod.rs @@ -4,8 +4,7 @@ * SPDX-License-Identifier: AGPL-3.0-only */ -pub mod cgroup; - +use std::sync::OnceLock; use std::time::Instant; use sysinfo::{CpuRefreshKind, MemoryRefreshKind, RefreshKind, System}; @@ -59,6 +58,11 @@ const SCORE_MIN: u32 = 1; const SCORE_MAX: u32 = 100_000; pub fn cpu_core_score() -> u32 { + static SCORE: OnceLock = OnceLock::new(); + *SCORE.get_or_init(benchmark_cpu_core_score) +} + +fn benchmark_cpu_core_score() -> u32 { let start = Instant::now(); let mut hash: u64 = FNV_OFFSET; for i in 0..BENCH_ITERATIONS { diff --git a/backend/gradient-worker/src/nix/adopt.rs b/backend/gradient-worker/src/nix/adopt.rs new file mode 100644 index 000000000..ae9726d73 --- /dev/null +++ b/backend/gradient-worker/src/nix/adopt.rs @@ -0,0 +1,93 @@ +/* + * SPDX-FileCopyrightText: 2026 Wavelens GmbH + * + * SPDX-License-Identifier: AGPL-3.0-only + */ + +use anyhow::{Context as _, Result}; +use futures::StreamExt as _; +use gradient_util::nix_hash::{nix32_encode, normalize_nar_hash}; +use gradient_wire::types::CachedPath; +use sha2::{Digest as _, Sha256}; + +use super::store::LocalNixStore; + +#[derive(Debug, PartialEq, Eq)] +pub(crate) enum Adoption { + Reveal, + Download, +} + +pub(crate) fn adoption(local: Option<&str>, server: Option<&str>) -> Adoption { + match (local, server) { + (Some(local), Some(server)) if normalize_nar_hash(local) == normalize_nar_hash(server) => { + Adoption::Reveal + } + _ => Adoption::Download, + } +} + +async fn local_nar_hash(store_path: &str) -> Result { + let mut stream = harmonia_file_nar::NarByteStream::new(store_path.to_owned().into()); + let mut hasher = Sha256::new(); + while let Some(chunk) = stream.next().await { + hasher.update(&chunk.context("NAR stream error")?); + } + Ok(format!("sha256:{}", nix32_encode(&hasher.finalize()))) +} + +pub(crate) async fn adopt_hidden( + store: &LocalNixStore, + entries: Vec, +) -> Result<(Vec, Vec)> { + let mut adopted = Vec::new(); + let mut download = Vec::new(); + for entry in entries { + let local = match store.is_hidden(&entry.path).await? { + true => Some(local_nar_hash(&entry.path).await?), + false => None, + }; + match adoption(local.as_deref(), entry.nar_hash.as_deref()) { + Adoption::Reveal => adopted.push(entry), + Adoption::Download => download.push(entry), + } + } + let paths: Vec = adopted.iter().map(|e| e.path.clone()).collect(); + store.reveal(&paths).await?; + Ok((adopted, download)) +} + +#[cfg(test)] +mod tests { + use super::*; + + const HASH: &str = "sha256:1b8m03r63zqhnjf7l5wnldhh7c134ap5vpj0850ymkq1iyzicy5s"; + + #[test] + fn a_hidden_path_matching_the_server_hash_is_revealed() { + assert_eq!(adoption(Some(HASH), Some(HASH)), Adoption::Reveal); + } + + #[test] + fn a_hidden_path_with_different_content_is_downloaded() { + let other = "sha256:0000000000000000000000000000000000000000000000000000"; + assert_eq!(adoption(Some(other), Some(HASH)), Adoption::Download); + } + + #[test] + fn a_hidden_path_without_a_server_hash_is_downloaded() { + assert_eq!(adoption(Some(HASH), None), Adoption::Download); + } + + #[test] + fn a_path_missing_on_disk_is_downloaded() { + assert_eq!(adoption(None, Some(HASH)), Adoption::Download); + } + + #[test] + fn the_server_hash_matches_in_sri_form_too() { + let sri = "sha256-47DEQpj8HBSa+/TImW+5JCeuQeRkm5NMpJWZG3hSuFU="; + let nix32 = gradient_util::nix_hash::normalize_nar_hash(sri); + assert_eq!(adoption(Some(&nix32), Some(sri)), Adoption::Reveal); + } +} diff --git a/backend/gradient-worker/src/nix/mod.rs b/backend/gradient-worker/src/nix/mod.rs index ef3e7705e..315383c06 100644 --- a/backend/gradient-worker/src/nix/mod.rs +++ b/backend/gradient-worker/src/nix/mod.rs @@ -9,6 +9,8 @@ pub use gradient_eval::{eval_worker, wildcard_walk}; +pub mod adopt; pub mod gcroots; pub mod log; pub mod store; +pub mod visibility; diff --git a/backend/gradient-worker/src/nix/store.rs b/backend/gradient-worker/src/nix/store.rs index 35bf3b4de..43cd6e9b4 100644 --- a/backend/gradient-worker/src/nix/store.rs +++ b/backend/gradient-worker/src/nix/store.rs @@ -28,6 +28,8 @@ use tracing::{debug, warn}; use gradient_wire::traits::WorkerStore; use gradient_worker_client::nar::{PathMeta, PathMetaSource}; +use super::visibility::PathVisibility; + const POOL_ACQUIRE_TIMEOUT: Duration = Duration::from_secs(600); const POOL_CONNECT_TIMEOUT: Duration = Duration::from_secs(600); @@ -45,6 +47,7 @@ const DEFAULT_DAEMON_SOCKET: &str = "/nix/var/nix/daemon-socket/socket"; #[derive(Clone)] pub struct LocalNixStore { pool: ConnectionPool, + visibility: PathVisibility, } impl LocalNixStore { @@ -55,9 +58,26 @@ impl LocalNixStore { pub fn connect_at(socket_path: &str, pool_size: usize) -> Result { Ok(Self { pool: ConnectionPool::new(socket_path, build_pool_config(pool_size)), + visibility: PathVisibility::default(), }) } + pub fn visibility(&self) -> &PathVisibility { + &self.visibility + } + + pub async fn has_path(&self, store_path: &str) -> Result { + Ok(self.visibility.allows(store_path) && self.is_on_disk(store_path).await?) + } + + pub async fn is_hidden(&self, store_path: &str) -> Result { + Ok(!self.visibility.allows(store_path) && self.is_on_disk(store_path).await?) + } + + pub async fn reveal(&self, store_paths: &[String]) -> Result<()> { + self.visibility.reveal(store_paths).await + } + pub async fn acquire(&self) -> Result { tokio::time::timeout(POOL_ACQUIRE_TIMEOUT, self.pool.acquire()) .await @@ -72,7 +92,7 @@ impl LocalNixStore { /// `is_valid_path` is the authoritative check for a parent the daemon will accept. /// `query_path_info` can still report metadata after a GC race or an interrupted import. /// That false positive would make the prefetch walk skip a path the daemon then rejects. - pub async fn has_path(&self, store_path: &str) -> Result { + async fn is_on_disk(&self, store_path: &str) -> Result { let hash_name = strip_store_prefix(store_path); let sp = StorePath::from_base_path(hash_name) .map_err(|e| anyhow::anyhow!("invalid store path {store_path}: {e}"))?; @@ -115,16 +135,19 @@ impl LocalNixStore { } pub async fn import_nar(&self, info: &ValidPathInfo, nar: &[u8]) -> Result<()> { + let path = info.info.store_dir.display(&info.path).to_string(); + let repair = !self.visibility.allows(&path); let mut guard = self.acquire().await?; guard .execute(|client| async move { - let logs = client.add_to_store_nar(info, nar, false, true); + let logs = client.add_to_store_nar(info, nar, repair, true); let mut logs = pin!(logs); while let Some(_msg) = logs.next().await {} logs.await }) .await - .map_err(|e| anyhow::anyhow!("daemon add_to_store_nar({}) failed: {e}", info.path)) + .map_err(|e| anyhow::anyhow!("daemon add_to_store_nar({}) failed: {e}", info.path))?; + self.reveal(&[path]).await } fn content_addressed(name: &str, nar: &[u8]) -> Result { diff --git a/backend/gradient-worker/src/nix/visibility.rs b/backend/gradient-worker/src/nix/visibility.rs new file mode 100644 index 000000000..ee7625d8d --- /dev/null +++ b/backend/gradient-worker/src/nix/visibility.rs @@ -0,0 +1,113 @@ +/* + * SPDX-FileCopyrightText: 2026 Wavelens GmbH + * + * SPDX-License-Identifier: AGPL-3.0-only + */ + +use std::collections::HashSet; +use std::sync::Arc; + +use anyhow::Result; +use gradient_util::store_path::strip_store_prefix; +use gradient_util::sync::Mutex; +use gradient_wire::messages::ClientMessage; +use gradient_worker_client::connection::ProtoWriter; + +#[derive(Clone, Default)] +pub struct PathVisibility { + inner: Arc, +} + +#[derive(Default)] +struct Inner { + // `None` until the first handover: a worker without one sees every path. + list: Mutex>>, + reports: Mutex>, +} + +impl PathVisibility { + pub fn allows(&self, store_path: &str) -> bool { + self.inner + .list + .lock() + .as_ref() + .is_none_or(|list| list.contains(strip_store_prefix(store_path))) + } + + pub fn start_list(&self) { + *self.inner.list.lock() = Some(HashSet::new()); + } + + pub fn extend>(&self, paths: I) { + if let Some(list) = self.inner.list.lock().as_mut() { + list.extend(paths); + } + } + + pub fn report_to(&self, writer: ProtoWriter) { + *self.inner.reports.lock() = Some(writer); + } + + pub async fn reveal(&self, store_paths: &[String]) -> Result<()> { + let added: Vec = { + let mut list = self.inner.list.lock(); + let Some(list) = list.as_mut() else { + return Ok(()); + }; + store_paths + .iter() + .map(|p| strip_store_prefix(p).to_owned()) + .filter(|p| list.insert(p.clone())) + .collect() + }; + let writer = self.inner.reports.lock().clone(); + match writer { + Some(writer) if !added.is_empty() => { + writer + .send(ClientMessage::PathsAdded { paths: added }) + .await + } + _ => Ok(()), + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn without_a_list_every_path_is_visible() { + let visibility = PathVisibility::default(); + assert!(visibility.allows("/nix/store/aaaa-hello")); + } + + #[test] + fn a_started_list_hides_every_path_not_on_it() { + let visibility = PathVisibility::default(); + visibility.start_list(); + visibility.extend(["aaaa-hello".to_owned()]); + assert!(visibility.allows("/nix/store/aaaa-hello")); + assert!(!visibility.allows("/nix/store/bbbb-openssl")); + } + + #[test] + fn starting_a_new_list_forgets_the_previous_user() { + let visibility = PathVisibility::default(); + visibility.start_list(); + visibility.extend(["aaaa-hello".to_owned()]); + visibility.start_list(); + assert!(!visibility.allows("/nix/store/aaaa-hello")); + } + + #[tokio::test] + async fn a_revealed_path_becomes_visible() { + let visibility = PathVisibility::default(); + visibility.start_list(); + visibility + .reveal(&["/nix/store/bbbb-openssl".to_owned()]) + .await + .unwrap(); + assert!(visibility.allows("/nix/store/bbbb-openssl")); + } +} diff --git a/backend/gradient-worker/src/proto/job.rs b/backend/gradient-worker/src/proto/job.rs index 69fa1ac27..7f033bb1e 100644 --- a/backend/gradient-worker/src/proto/job.rs +++ b/backend/gradient-worker/src/proto/job.rs @@ -28,7 +28,9 @@ use crate::proto::progress::{ BuildProgressSink, EvalProgressSender, Progress, Tally, count_delivery, }; use gradient_wire::traits::{EvalProgressSink, JobReporter}; -use gradient_wire::types::{BuildProgressPhase, GrantTarget, UploadMetadata, UploadObject}; +use gradient_wire::types::{ + BuildProgressPhase, BuildStage, GrantTarget, PROTO_BUILD_STAGES, UploadMetadata, UploadObject, +}; use gradient_worker_client::connection::ProtoWriter; use gradient_worker_client::nar_recv::{NarPayload, NarReceiver, NarUnavailable}; use gradient_worker_client::upload::UploadClient; @@ -385,8 +387,12 @@ impl JobUpdater { }) } - pub async fn report_compressing(&self) -> Result<()> { - self.send_update(JobUpdateKind::Compressing).await + pub async fn report_stage(&self, stage: BuildStage) -> Result<()> { + if self.writer.version() < PROTO_BUILD_STAGES { + return Ok(()); + } + + self.send_update(JobUpdateKind::Stage(stage)).await } pub async fn send_eval_message( @@ -548,10 +554,6 @@ impl JobReporter for JobUpdater { .await } - async fn report_compressing(&mut self) -> Result<()> { - self.send_update(JobUpdateKind::Compressing).await - } - async fn send_log_chunk(&mut self, task_index: u32, data: Vec) -> Result<()> { self.writer .send(ClientMessage::LogChunk { diff --git a/backend/gradient-worker/src/proto/prefetch.rs b/backend/gradient-worker/src/proto/prefetch.rs index 10b543ae1..fc3c3539b 100644 --- a/backend/gradient-worker/src/proto/prefetch.rs +++ b/backend/gradient-worker/src/proto/prefetch.rs @@ -16,6 +16,7 @@ use gradient_wire::messages::{BuildSpec, CachedPath, EvalMessageLevel, QueryMode use gradient_wire::types::{BuildProgressPhase, JobPhase}; use tracing::{debug, error, warn}; +use crate::nix::adopt::adopt_hidden; use crate::nix::store::LocalNixStore; use crate::proto::compression::drv_closure_seeds_from_compressed_nar; use crate::proto::job::JobUpdater; @@ -110,7 +111,6 @@ pub(crate) async fn download_one_presigned( let mut backoff = PRESIGNED_RETRY_BASE; for attempt in 1..=PRESIGNED_DOWNLOAD_MAX_ATTEMPTS { - let started = std::time::Instant::now(); let attempt_err = match http.get(&url).send().await { Ok(resp) => { let status = resp.status().as_u16(); @@ -144,8 +144,6 @@ pub(crate) async fn download_one_presigned( return Ok((path, None)); } - gradient_worker_client::throughput::NETWORK - .observe_transfer(bytes.len() as u64, started.elapsed()); return Ok((path, Some((bytes, cp)))); } } @@ -446,6 +444,8 @@ impl<'a> InputPrefetcher<'a> { } let (by_url, by_request) = self.query_and_split(to_query).await?; + let (adopted_by_url, by_url) = adopt_hidden(self.store, by_url).await?; + let (adopted_by_request, by_request) = adopt_hidden(self.store, by_request).await?; progress.expect( download_size(by_url.iter().chain(&by_request)), (by_url.len() + by_request.len()) as u32, @@ -465,7 +465,9 @@ impl<'a> InputPrefetcher<'a> { } let mut refs: HashSet = batch .iter() - .flat_map(|(_, _, meta)| meta.references.clone().unwrap_or_default()) + .map(|(_, _, meta)| meta) + .chain(adopted_by_url.iter().chain(&adopted_by_request)) + .flat_map(|meta| meta.references.clone().unwrap_or_default()) .filter(|r| !queried.contains(r)) .collect(); @@ -549,8 +551,11 @@ impl<'a> InputPrefetcher<'a> { tally: &Tally, ) -> Result> { let mut fetch = self.updater.phase(JobPhase::NarFetch); + let started = std::time::Instant::now(); let mut batch = self.fetch_by_request(by_request, tally).await?; batch.extend(download_by_url(by_url, tally).await?); + let nar_bytes = batch.iter().filter_map(|(_, _, cp)| cp.nar_size).sum(); + gradient_worker_client::throughput::DOWNLOAD.observe_transfer(nar_bytes, started.elapsed()); fetch.record(batch.len() as u32, payload_bytes(&batch).await); Ok(batch) } @@ -851,33 +856,6 @@ mod tests { assert_eq!(bytes, b"NAR-BYTES"); } - #[tokio::test] - async fn a_presigned_download_feeds_the_network_throughput() { - use wiremock::matchers::method; - use wiremock::{Mock, MockServer, ResponseTemplate}; - - let objects = MockServer::start().await; - Mock::given(method("GET")) - .respond_with(ResponseTemplate::new(200).set_body_bytes(b"NAR-BYTES".to_vec())) - .mount(&objects) - .await; - let cp = cached( - "/nix/store/aaaa-measured", - Some(&format!("{}/nar/abc.nar", objects.uri())), - ); - - let http = gradient_util::http::build_download_client().expect("download client"); - download_one_presigned(&http, cp, &mut Progress::silent()) - .await - .expect("download"); - - assert!( - gradient_worker_client::throughput::NETWORK - .current() - .is_some() - ); - } - #[tokio::test] async fn redirect_swallowed_as_empty_body_is_not_a_download() { use wiremock::matchers::{method, path as path_matcher}; diff --git a/backend/gradient-worker/src/worker/handover.rs b/backend/gradient-worker/src/worker/handover.rs new file mode 100644 index 000000000..63250602b --- /dev/null +++ b/backend/gradient-worker/src/worker/handover.rs @@ -0,0 +1,59 @@ +/* + * SPDX-FileCopyrightText: 2026 Wavelens GmbH + * + * SPDX-License-Identifier: AGPL-3.0-only + */ + +use std::path::{Path, PathBuf}; + +use anyhow::{Context as _, Result}; + +use crate::config::WorkerConfig; +use crate::executor::JobExecutor; + +fn user_state_dirs(config: &WorkerConfig, xdg_cache_home: Option<&str>) -> Vec { + let mut dirs = vec![ + PathBuf::from(config.eval_cache_dir()), + config.nar_partial_dir(), + ]; + dirs.extend(xdg_cache_home.map(|cache| Path::new(cache).join("nix"))); + dirs +} + +pub(super) async fn forget_previous_user( + executor: &JobExecutor, + config: &WorkerConfig, +) -> Result<()> { + executor.evaluator.resolver().release_idle().await; + let xdg_cache_home = std::env::var("XDG_CACHE_HOME").ok(); + for dir in user_state_dirs(config, xdg_cache_home.as_deref()) { + match tokio::fs::remove_dir_all(&dir).await { + Ok(()) => {} + Err(e) if e.kind() == std::io::ErrorKind::NotFound => {} + Err(e) => return Err(e).with_context(|| format!("wiping {}", dir.display())), + } + tokio::fs::create_dir_all(&dir) + .await + .with_context(|| format!("recreating {}", dir.display()))?; + } + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn a_handover_wipes_the_eval_cache_partial_nars_and_the_nix_fetcher_cache() { + let config = WorkerConfig::default(); + let dirs = user_state_dirs(&config, Some("/var/lib/gradient-worker/www/.cache")); + assert_eq!( + dirs, + [ + PathBuf::from("/var/lib/gradient-worker/eval-cache"), + PathBuf::from("/var/lib/gradient-worker/nar-partial"), + PathBuf::from("/var/lib/gradient-worker/www/.cache/nix"), + ] + ); + } +} diff --git a/backend/gradient-worker/src/worker/message_loop.rs b/backend/gradient-worker/src/worker/message_loop.rs index 9acf46451..e28495951 100644 --- a/backend/gradient-worker/src/worker/message_loop.rs +++ b/backend/gradient-worker/src/worker/message_loop.rs @@ -284,6 +284,23 @@ impl MessageLoopState { } } + async fn on_handover(&mut self, index: u32, paths: Vec, is_final: bool) -> Result<()> { + let visibility = self.executor.store.visibility().clone(); + if index == 0 { + for job in self.jobs.running.values() { + let _ = job.abort.send(true); + } + super::handover::forget_previous_user(&self.executor, &self.config).await?; + visibility.start_list(); + } + visibility.extend(paths); + if is_final { + visibility.report_to(self.writer.clone()); + self.writer.send(ClientMessage::HandoverDone).await?; + } + Ok(()) + } + async fn route(&mut self, msg: ServerMessage) -> Result<()> { match msg { ServerMessage::JobListChunk { @@ -378,6 +395,13 @@ impl MessageLoopState { | ServerMessage::NarAbort { .. } => { warn!("a NAR frame reached the control dispatch"); } + ServerMessage::Handover { + index, + paths, + is_final, + } => { + self.on_handover(index, paths, is_final).await?; + } } Ok(()) } @@ -406,7 +430,15 @@ impl MessageLoopState { let completed_kind = job.kind; let assignment_id = job.assignment_id.get(); let dropped_spans = job.timeline.dropped(); - let TimelineSnapshot { spans, elapsed_ms } = job.timeline.snapshot(); + let TimelineSnapshot { + mut spans, + elapsed_ms, + } = job.timeline.snapshot(); + let version = self.writer.version(); + for span in &mut spans { + span.phase = span.phase.known_to(version); + } + if dropped_spans > 0 { debug!(%job_id, dropped_spans, "phase timeline hit its span cap"); } @@ -426,6 +458,7 @@ impl MessageLoopState { Err(e) => { let error_chain = format!("{e:#}"); let (kind, missing_paths) = crate::executor::failure::wire_failure(&e); + let metrics = crate::executor::failure::failure_metrics(&e); error!(%job_id, error = %error_chain, ?kind, phases = spans.len(), "job failed"); self.writer .send(ClientMessage::JobFailed { @@ -436,6 +469,7 @@ impl MessageLoopState { missing_paths, spans, elapsed_ms, + metrics, }) .await?; } @@ -733,6 +767,7 @@ impl MessageLoopState { missing_paths: Vec::new(), spans: Vec::new(), elapsed_ms: 0, + metrics: None, }) .await?; } @@ -847,7 +882,8 @@ fn send_live_metrics(writer: &ProtoWriter) { cpu_usage_pct: m.cpu_usage_pct, ram_free_mb: m.ram_free_mb, disk_speed_mbps: gradient_worker_client::throughput::DISK.current(), - network_speed_mbps: gradient_worker_client::throughput::NETWORK.current(), + upload_speed_mbps: gradient_worker_client::throughput::UPLOAD.current(), + download_speed_mbps: gradient_worker_client::throughput::DOWNLOAD.current(), }) .await { diff --git a/backend/gradient-worker/src/worker/mod.rs b/backend/gradient-worker/src/worker/mod.rs index ef6e3991e..090961112 100644 --- a/backend/gradient-worker/src/worker/mod.rs +++ b/backend/gradient-worker/src/worker/mod.rs @@ -5,6 +5,7 @@ */ mod cluster; +mod handover; mod id; mod message_loop; mod scoring; @@ -193,23 +194,16 @@ impl Worker { evaluator, gcroots, config.ssh_bin.clone(), - config.build.metrics, - config.build.cgroup_root.clone(), crate::executor::log_limit::LogRateLimits { burst_bytes_per_min: config.log.burst_bytes_per_min, sustained_bytes_per_hour: config.log.sustained_bytes_per_hour, }, config.log.fetch_from_store, - config.build_cores(), + crate::executor::BuildHost { + build_cores: config.build_cores(), + cpu_core_score: config.cpu_core_score(), + }, ); - if config.build.metrics { - tracing::info!( - cgroup_root = %config.build.cgroup_root, - "build metrics enabled: CPU from daemon build result, peak RAM/disk sampled from the build cgroup" - ); - } else { - tracing::debug!("build metrics disabled; reporting wall-clock build time only"); - } Ok((executor, JobScorer::new())) } } @@ -259,10 +253,7 @@ async fn perform_setup( None => detect_system_features(&config.nix_bin).await, }; let host = crate::metrics::host_static(); - let cpu_core_score = config - .system - .cpu_core_score - .unwrap_or_else(crate::metrics::cpu_core_score); + let cpu_core_score = config.cpu_core_score(); info!( ?architectures, ?system_features, diff --git a/backend/gradient-worker/src/worker_pool/resolver.rs b/backend/gradient-worker/src/worker_pool/resolver.rs index 48f0522fa..e7b7b6247 100644 --- a/backend/gradient-worker/src/worker_pool/resolver.rs +++ b/backend/gradient-worker/src/worker_pool/resolver.rs @@ -296,6 +296,10 @@ impl WorkerPoolResolver { self.pool.shutdown().await; } + pub async fn release_idle(&self) { + self.pool.release_idle().await; + } + pub async fn fingerprint( &self, repository: String, diff --git a/docs/gradient-api.yaml b/docs/gradient-api.yaml index a1c292ea1..0708a467a 100644 --- a/docs/gradient-api.yaml +++ b/docs/gradient-api.yaml @@ -9059,10 +9059,10 @@ paths: get: tags: [board] summary: Most expensive builds by resource - description: Top 20 derivations by a captured per-build resource (peak RAM, CPU time, total disk bytes, or host network peak) from each derivation's latest `derivation_metric` row in the window, skipping zero values. Project-scoped; `worker_name` is shown only when the worker is registered by a project the caller can see. + description: Top 20 derivations by a captured per-build resource (peak RAM, CPU time or total disk bytes) from each derivation's latest `derivation_metric` row in the window, skipping zero values. Project-scoped; `worker_name` is shown only when the worker is registered by a project the caller can see. operationId: getBoardExpensiveByResource parameters: - - { name: metric, in: query, required: true, schema: { type: string, enum: [ram, cpu, disk, network] } } + - { name: metric, in: query, required: true, schema: { type: string, enum: [ram, cpu, disk] } } - { name: window_days, in: query, schema: { type: integer } } responses: '200': @@ -9078,7 +9078,7 @@ paths: project: { type: string, format: uuid } name: { type: string } value: { type: number } - unit: { type: string, enum: [MB, ms, bytes, Mbps] } + unit: { type: string, enum: [MB, ms, bytes] } worker: { type: string } worker_name: { type: string, nullable: true } '404': @@ -9316,7 +9316,7 @@ paths: get: tags: [board] summary: Network & API statistics - description: NAR egress series, per-worker latest network/disk speeds (project-scoped), and per-route HTTP latency/throughput (superuser-only). + description: NAR egress series, per-worker latest upload, download and disk speeds (project-scoped), and per-route HTTP latency/throughput (superuser-only). operationId: getBoardNetwork parameters: - { name: window_hours, in: query, required: false, schema: { type: integer, default: 24 } } @@ -9452,7 +9452,7 @@ paths: get: tags: [board] summary: Per-worker metrics - description: Live-metric sample time-series (CPU, RAM, disk, network, load), connect/disconnect history, and total assigned-job count for one worker. Project members only, and only for a worker registered in the project or enabled for it as a base worker; the telemetry covers everything the worker did, not only this project's jobs. + description: Live-metric sample time-series (CPU, RAM, disk, upload, download, load), connect/disconnect history, and total assigned-job count for one worker. Project members only, and only for a worker registered in the project or enabled for it as a base worker; the telemetry covers everything the worker did, not only this project's jobs. operationId: getProjectWorkerMetrics parameters: - { name: project, in: path, required: true, schema: { type: string } } @@ -12456,7 +12456,7 @@ components: type: object required: [id, build_id, derivation_path, eval, build_status, has_artefacts, outputs, architecture, deps, deps_total, created_at, prioritized] properties: - prioritized: { type: boolean, description: "The entry point's build is handed out ahead of unprioritized work." } + prioritized: { type: boolean, description: "The build gets a QoS bonus: a user prioritized it or a build that depends on it, or it belongs to a running evaluation that is prioritized or runs a build request." } id: { type: string, format: uuid } build_id: { type: string, format: uuid } derivation_path: { type: string, description: "Prefix-free `.drv` store path (`-.drv`, no `/nix/store/`)" } @@ -12533,7 +12533,7 @@ components: type: object required: [id, commit, status, wildcard, total_builds, builds, errors, warnings, created_at, updated_at, prioritized] properties: - prioritized: { type: boolean, description: "The evaluation's whole build tree is handed out ahead of unprioritized work." } + prioritized: { type: boolean, description: "The evaluation gets a QoS bonus while it runs: a user prioritized it or it runs a build request." } id: type: string format: uuid @@ -12800,7 +12800,7 @@ components: type: object required: [id, repository, commit, wildcard, status, created_at, updated_at, prioritized] properties: - prioritized: { type: boolean, description: "The evaluation's whole build tree is handed out ahead of unprioritized work. Cleared when the evaluation fails or is aborted." } + prioritized: { type: boolean, description: "The evaluation gets a QoS bonus while it runs: a user prioritized it or it runs a build request." } id: type: string format: uuid @@ -13092,7 +13092,7 @@ components: type: object required: [id, name, status, has_artefacts, updated_at, build_time_ms, dispatched_job, depth, prioritized] properties: - prioritized: { type: boolean, description: "The build is handed out ahead of unprioritized work. Cleared when it fails permanently, dependency-fails, times out, is aborted or is skipped." } + prioritized: { type: boolean, description: "The build gets a QoS bonus: a user prioritized it or a build that depends on it, or it belongs to a running evaluation that is prioritized or runs a build request." } id: type: string format: uuid @@ -13380,7 +13380,7 @@ components: type: object required: [id, evaluation, status, derivation_path, architecture, output, created_at, updated_at, prioritized] properties: - prioritized: { type: boolean, description: "The build is handed out ahead of unprioritized work. Cleared when it fails permanently, dependency-fails, times out, is aborted or is skipped." } + prioritized: { type: boolean, description: "The build gets a QoS bonus: a user prioritized it or a build that depends on it, or it belongs to a running evaluation that is prioritized or runs a build request." } id: type: string format: uuid @@ -14913,7 +14913,9 @@ components: ram_free_mb: { type: integer, nullable: true, description: Null until the worker's first live-metrics heartbeat. } cpu_usage_pct: { type: number, nullable: true, description: Null until the worker's first live-metrics heartbeat. } disk_speed_mbps: { type: number, nullable: true } - network_speed_mbps: { type: number, nullable: true } + upload_speed_mbps: { type: number, nullable: true, description: Output NAR megabits per second of upload batches from the first grant on. } + download_speed_mbps: { type: number, nullable: true, description: NAR megabits per second of input download batches and substitutions. } + running_builds: { type: integer, description: Jobs of this worker in the build stage. } JobContext: type: object @@ -14941,12 +14943,15 @@ components: is_fixed_output: { type: boolean, description: Build-only. } history: type: object - description: Build-only history prediction from prior builds of this derivation. + description: Build-only history prediction from prior builds of this derivation. A value is null when no prior build measured it. properties: - peak_ram_mb: { type: integer } - avg_cpu_time_ms: { type: integer } - build_time_ms: { type: integer } - avg_disk_bytes: { type: integer } + peak_ram_mb: { type: integer, nullable: true } + avg_cpu_time_ms: { type: integer, nullable: true } + build_time_ms: { type: integer, nullable: true } + uncontended_build_time_ms: { type: integer, nullable: true, description: Mean build time with the slowdown of concurrent builds on the worker removed. Mean run time of the task for an evaluation. } + build_core_score: { type: integer, nullable: true, description: Mean cpu_core_score of the workers behind the build times. } + avg_disk_bytes: { type: integer, nullable: true } + output_nar_size: { type: integer, nullable: true, description: Mean summed NAR size of the outputs of prior builds. } oom_rate: { type: number } samples: { type: integer } derivations: @@ -14974,7 +14979,6 @@ components: cpu_time_ms: { $ref: '#/components/schemas/Windowed' } avg_cpu_pct: { $ref: '#/components/schemas/Windowed' } disk_bytes: { $ref: '#/components/schemas/Windowed' } - network_mbps: { $ref: '#/components/schemas/Windowed' } oom_rate: { $ref: '#/components/schemas/Windowed' } closure_size: { $ref: '#/components/schemas/Windowed' } nar_size_mb: { $ref: '#/components/schemas/Windowed' } @@ -14986,6 +14990,16 @@ components: total_workers: { type: integer } idle_workers: { type: integer } cpu_core_score_mean: { type: number, nullable: true, description: Mean cpu_core_score of the connected workers that reported one. } + upload_speed_mean_mbps: { type: number, nullable: true, description: Mean upload_speed_mbps of the connected workers that measured one. } + download_speed_mean_mbps: { type: number, nullable: true, description: Mean download_speed_mbps of the connected workers that measured one. } + downloads_in_flight: { type: integer, description: Build jobs of every worker in the prefetch stage. } + uploads_in_flight: { type: integer, description: Build jobs of every worker in the upload stage. } + storage_read_mbps: { type: number, nullable: true, description: Best total stored megabits per second of overlapping NAR downloads in the last hour. } + storage_write_mbps: { type: number, nullable: true, description: Best total stored megabits per second of overlapping NAR uploads in the last hour. } + compression_ratio: { type: number, nullable: true, description: Stored bytes per NAR byte of the uploads in the last hour. } + per_path_secs: { type: number, nullable: true, description: Seconds a missing path costs beyond its bytes, fitted over the prefetches of the last 24 hours. } + download_slots: { type: integer, description: The configured nar.maxConcurrentDownloads. } + upload_slots: { type: integer, description: The configured upload.concurrency. } DispatchedJobDetail: type: object @@ -15109,6 +15123,7 @@ components: - compress - nar_push - cache_query_wait + - upload_wait start_ms: { type: integer, format: int64 } end_ms: { type: integer, format: int64 } paths: diff --git a/docs/src/contributors/eval-metrics.md b/docs/src/contributors/eval-metrics.md index caeb9b271..515992d8a 100644 --- a/docs/src/contributors/eval-metrics.md +++ b/docs/src/contributors/eval-metrics.md @@ -8,7 +8,7 @@ flowchart LR worker -->|EvalStats| server[Server] server --> tables[(evaluation_metric,
evaluation_attr_cost,
flake_output_node)] tables --> board[Job Board: Evals] - tables --> rule[ResourceFitRule: p95 RAM per task] + tables --> rule[EstimatedTimeRule: p95 RAM per task] ``` ## Tables @@ -28,7 +28,7 @@ flowchart LR ## RAM Routing -`ResourceFitRule` (see [Scoring](scheduler/scoring.md)) can read a per-task rolling p95 of `peak_rss_mb` over the last 24 h. Tasks whose evaluations needed much RAM go to big-RAM workers. The prediction can update as evaluations finish, with no manual thresholds. +`EstimatedTimeRule` (see [Scoring](scheduler/scoring.md)) can read a per-task rolling p95 of `peak_rss_mb` over the last 24 h. Predicted RAM above a worker's free memory can add the expected time of a repeated evaluation. Tasks whose evaluations needed much RAM go to big-RAM workers. The prediction can update as evaluations finish, with no manual thresholds. ## Endpoints diff --git a/docs/src/contributors/proto/capabilities-and-dispatch.md b/docs/src/contributors/proto/capabilities-and-dispatch.md index 351b5b28e..775893eed 100644 --- a/docs/src/contributors/proto/capabilities-and-dispatch.md +++ b/docs/src/contributors/proto/capabilities-and-dispatch.md @@ -37,7 +37,7 @@ Workers with the `build` capability send `WorkerCapabilities` after the handshak ## Metrics and Liveness -- The `WorkerMetrics` fields (`cpu_usage_pct`, `ram_free_mb`, `disk_speed_mbps`, `network_speed_mbps`) travel with the 10 s heartbeat. +- The `WorkerMetrics` fields (`cpu_usage_pct`, `ram_free_mb`, `disk_speed_mbps`, `upload_speed_mbps`, `download_speed_mbps`) travel with the 10 s heartbeat. - Workers without metrics get unknown values in scoring. - The server will mark a worker as seen on every message. - The `worker_liveness_pass` will drop workers silent for `proto.workerHeartbeatTimeoutSecs` (120 s, `0` disabling the check). diff --git a/docs/src/contributors/proto/jobs.md b/docs/src/contributors/proto/jobs.md index f95a7b60b..b11d2abcf 100644 --- a/docs/src/contributors/proto/jobs.md +++ b/docs/src/contributors/proto/jobs.md @@ -86,11 +86,12 @@ Build jobs carry exactly one `BuildSpec`, meaning one shared build (`derivation_ | `InputUpdateResult`, `InputUpdateExpansion` | Flake update candidate lock and bumped inputs | | `Building { build_id }` | Build turning `Building`. An already aborted build is getting `AbortJob` instead | | `BuildOutput` | Output sizes, build products, metrics, the `substituted` flag | -| `Compressing` | No change | +| `Stage(Prefetch / Build / Upload)` | Live stage of the job in the worker pool, for the [estimated time](../../reference/scheduler-policies.md#estimated-time). Sent to servers on protocol 28 and later | +| `Compressing` | No change. Workers no longer send it | An `EvalProgress` message will carry one download row per flake input while fetching and the live thunk count while evaluating. The eval worker will download the inputs itself, with one download in flight per second-level domain. -`JobCompleted` and `JobFailed` carry the phase timeline shown on the [Job Board](../../ui/job-board.md#job-inspection). The server will drop reports from a stale `assignment_id`. +`JobCompleted` and `JobFailed` carry the phase timeline shown on the [Job Board](../../ui/job-board.md#job-inspection). A server before protocol 28 will receive `NarPush` in place of the `UploadWait` phase. Build metrics also hold the number of concurrent builds on the worker, the cores of the build and the worker's CPU score. The server will drop reports from a stale `assignment_id`. ## Failures diff --git a/docs/src/contributors/proto/messages.md b/docs/src/contributors/proto/messages.md index 002f38875..eee34db0b 100644 --- a/docs/src/contributors/proto/messages.md +++ b/docs/src/contributors/proto/messages.md @@ -32,6 +32,7 @@ Every message on `/proto`, from `backend/gradient-wire/src/messages`. IDs (`job_ | `CacheError` | Cache state unknown. The worker is retrying | `query_id`, `message` | | `UploadGrant` | Upload admission: skip, passthrough (with resume offset), presigned PUT or multipart | `request_id`, `target` | | `UploadCommitted` | Upload outcome: ok, retry or rejected | `request_id`, `outcome` | +| `Handover` | Path list of the next user of a shared worker, in chunks. Index 0 starts a new list and wipes per-user caches. Paths off the list stay hidden from jobs | `index`, `paths`, `is_final` | ## Worker -> Server @@ -42,7 +43,7 @@ Every message on `/proto`, from `backend/gradient-wire/src/messages`. IDs (`job_ | `ReauthRequest` | Asking for a new `AuthChallenge` | - | | `Reject` | Declining a server-dialed session before `InitConnection` (`400`, `401`) | `code`, `reason` | | `WorkerCapabilities` | Systems, features, slots, CPU, RAM, core score, zone, endpoint | `architectures`, `system_features`, `max_concurrent_builds`, `zone`, `endpoint`, ... | -| `WorkerMetrics` | Load heartbeat | `cpu_usage_pct`, `ram_free_mb`, `disk_speed_mbps`, `network_speed_mbps` | +| `WorkerMetrics` | Load heartbeat | `cpu_usage_pct`, `ram_free_mb`, `disk_speed_mbps`, `upload_speed_mbps`, `download_speed_mbps` | | `RequestJobList` | Asking for the full candidate list | - | | `RequestJobChunk` | Score deltas | `scores`, `is_final` | | `RequestJob` | One free slot of a kind. Repeated every 10 s while idle | `kind` (`Flake` or `Build`) | @@ -50,7 +51,7 @@ Every message on `/proto`, from `backend/gradient-wire/src/messages`. IDs (`job_ | `ClusterSignal` | Control message to one member (`to`) or every other member (`to` unset) of a started attempt | `attempt`, `to`, `payload` | | `JobUpdate` | Progress of a job | `job_id`, `assignment_id`, `update` | | `JobCompleted` | Job done, with the phase timeline | `job_id`, `assignment_id`, `spans` | -| `JobFailed` | Job failed | `job_id`, `assignment_id`, `error`, `kind`, `missing_paths`, `spans` | +| `JobFailed` | Job failed, with the metrics of a failed build | `job_id`, `assignment_id`, `error`, `kind`, `missing_paths`, `spans`, `metrics` | | `BuildProgress` | Bytes and paths of a build's prefetch, download or upload | `job_id`, `assignment_id`, `build_id`, `phase`, `bytes_done`, `bytes_total`, `paths_done`, `paths_total` | | `EvalProgress` | Flake input downloads or live thunks of an eval job, at most once per second | `job_id`, `assignment_id`, `progress` | | `Draining` | Worker draining | - | @@ -65,6 +66,8 @@ Every message on `/proto`, from `backend/gradient-wire/src/messages`. IDs (`job_ | `UploadChunk` (bulk) | Passthrough upload bytes | `request_id`, `data`, `offset`, `is_final` | | `UploadFinished` | Upload done, with NAR metadata | `request_id`, `metadata` | | `UploadCancel` | Cancelling an upload | `request_id` | +| `HandoverDone` | Last `Handover` chunk applied | - | +| `PathsAdded` | Paths verified, imported or built for the current user of a shared worker | `paths` | ## Cache Query Modes diff --git a/docs/src/contributors/proto/transfer.md b/docs/src/contributors/proto/transfer.md index 9b74d210d..1b31a4aea 100644 --- a/docs/src/contributors/proto/transfer.md +++ b/docs/src/contributors/proto/transfer.md @@ -25,7 +25,7 @@ sequenceDiagram | `Multipart` | S3, NAR over 1 GiB | Presigned parts of at least 64 MiB | - **Admission** is server-wide and fair across sessions. - - The limits are `upload.concurrency` (16) large uploads and `upload.bytesBudget` (8 GiB) at once. + - The limits are `upload.concurrency` (8) large uploads and `upload.bytesBudget` (8 GiB) at once. - Small uploads (at most 1 MiB of NAR, `SMALL_UPLOAD_BYTES`) get a window of their own, `SMALL_UPLOADS_IN_FLIGHT` (128). - The cost of a small upload is its two round trips. Each `EvalResult` batch of an evaluation will wait on the push of its own `.drv` files. - Small uploads go ahead of larger ones in their session. The server will turn first to sessions with a waiting small upload. @@ -57,7 +57,7 @@ Workers prefetch every input missing from the local store ahead of the build. | Condition | Delivered As | |---|---| | S3 store, confirmed NAR above `nar.smallBytes` (1 MiB) | Presigned GET URL | -| Anything else | Stream over `/proto`, at most `nar.maxConcurrentServes` (8) paths per connection | +| Anything else | Stream over `/proto`, at most `nar.maxConcurrentServes` (8) paths per connection and `nar.maxConcurrentDownloads` (16) across all connections | - Small NARs come from an in-memory hot cache, `nar.hotCacheBytes` (512 MiB). - A `NarUnavailable` answer will also remove the stale cache row on the server. diff --git a/docs/src/contributors/scheduler/scoring.md b/docs/src/contributors/scheduler/scoring.md index 63bfef1e8..3d66fca2b 100644 --- a/docs/src/contributors/scheduler/scoring.md +++ b/docs/src/contributors/scheduler/scoring.md @@ -33,8 +33,8 @@ flowchart LR |---|---|---| | `JobContext` | `ScoredJob` (kind, architecture, `prefer_local_build`, `is_fixed_output`, `pname`, closure size, history), `missing_count`, `missing_nar_size`, `outputs_present`, `dependency_count`, `queued_at`, `ready_at`, `project_work_share`, `prioritized`, `rescore_count`, `now` | `JobTracker::score_candidates` in `gradient-scheduler/src/jobs.rs` | | `WorkerContext` | `architectures`, `system_features`, `fetch`, `metrics` | `worker_context_of` from the worker's `WorkerCaps` | -| `WorkerMetricsView` | `cpu_count`, `cpu_core_score`, `ram_total_mb`, `ram_free_mb`, `cpu_usage_pct`, `disk_speed_mbps`, `network_speed_mbps` | `WorkerCapabilities` (static) and the 10 s `WorkerMetrics` heartbeat (live) | -| `InstanceContext` | 13 `Windowed` averages, `active_builds`, `pending_builds`, `total_workers`, `idle_workers`, `cpu_core_score_mean` | `instance_metrics_pass`, see below | +| `WorkerMetricsView` | `cpu_count`, `cpu_core_score`, `ram_total_mb`, `ram_free_mb`, `cpu_usage_pct`, `disk_speed_mbps`, `upload_speed_mbps`, `download_speed_mbps`, `running_builds` | `WorkerCapabilities` (static), the 10 s `WorkerMetrics` heartbeat and the build stages (live) | +| `InstanceContext` | 12 `Windowed` averages, `active_builds`, `pending_builds`, `total_workers`, `idle_workers`, `cpu_core_score_mean`, `upload_speed_mean_mbps`, `download_speed_mean_mbps`, `storage_read_mbps`, `storage_write_mbps`, `compression_ratio`, `per_path_secs`, `download_slots`, `upload_slots`, `downloads_in_flight`, `uploads_in_flight` | `instance_metrics_pass`, see below. The two in-flight counts come from the build stages at assignment time | - `missing_count`, `missing_nar_size` and `outputs_present` are per worker. The worker must score each offered candidate against its store and send a `CandidateScore` (see [Offers](../proto/capabilities-and-dispatch.md#offers)). The values are `None` until that worker reported. - `dependency_count` is the number of direct input derivations (`derivation_dependency` rows), not the number of builds needing the derivation. @@ -43,6 +43,18 @@ flowchart LR - The caller must pass `now` in. Rules never read the wall clock. - `JobContext::build_history` will return an empty prediction when `outputs_present` is set. A worker holding every output will build nothing. +## Build Stages + +Workers on protocol 28 report `JobUpdateKind::Stage` on entering `Prefetch`, `Build` or `Upload` in a build job. The worker pool can store the stage next to each assigned job. + +| Count | Meaning | +|---|---| +| `running_builds` | Jobs of this worker in `Build` | +| `downloads_in_flight` | Jobs of every worker in `Prefetch`, presigned downloads included | +| `uploads_in_flight` | Jobs of every worker in `Upload`, jobs waiting for their grant included | + +A released job will drop out of the counts with its slot. + ## History and Closure Size `ScoredJob` can hold only owned values. Scoring itself will compute nothing. `load_sizes_and_histories` in `loops/build.rs` can materialize both on the pending job, only when `policy.uses_history()` is true. @@ -51,9 +63,16 @@ flowchart LR |---|---| | Closure size | `derivation.closure_size`, else one batched `transitive_closure_sizes` walk. The graph writer will persist computed sizes | | Build history | `history::predict`: latest 20 `derivation_metric` rows with the same `history_name` (`pname`, else `name`) and architecture, within [`retentionDays`](../../reference/configuration.md#general). One query per distinct pair | +| Output size | `derivation_output.nar_size` of the derivations behind those rows, summed per derivation. One more query per pair with history | | Evaluation history | `compute_eval_history`: per-task p95 of `evaluation_metric.peak_rss_mb` over 24 h | -`HistoryPrediction` must carry the p95 peak RAM, mean CPU time, mean build time, mean disk bytes, OOM rate and a `samples` count. Rules treat `samples == 0` as no history and add nothing. +- Only real builds can write a `derivation_metric` row. A substituted output will write none. +- A failed build will write a row only after an out-of-memory kill. + +`HistoryPrediction` must carry the p95 peak RAM, mean CPU time, mean build time, mean disk bytes, mean output NAR size, OOM rate and a `samples` count. The mean build time without contention and the mean CPU score of the building workers are part of it too. A value will stay `None` when no build in the window measured it. Rules add nothing for a `None` value. Evaluations carry the mean run time of their task in the build time fields. + +- Each `derivation_metric` row records `concurrent_builds`, `build_cores` and `cpu_core_score` from the worker's `BuildMetrics`. +- `contention_factor` can turn each row into a build time without contention, dividing by `1 + 0.04 * concurrent_builds`. ## Instance Windows @@ -61,9 +80,14 @@ flowchart LR | Source table | Windows | |---|---| -| `derivation_metric` | `peak_ram_mb`, `cpu_time_ms`, `avg_cpu_pct`, `disk_bytes`, `network_mbps`, `build_time_ms`, `closure_size`, `oom_rate`, `completed` | +| `derivation_metric` | `peak_ram_mb`, `cpu_time_ms`, `avg_cpu_pct`, `disk_bytes`, `build_time_ms`, `closure_size`, `oom_rate`, `completed` | | `dispatched_job` (builds with `ready_at`) | `wait_secs`, `nar_size_mb`, `missing_paths`, `dependency_cnt` | +- `STORAGE_PEAK_THROUGHPUT` can read the `NarFetch` and `NarPush` spans of the last hour. Spans contribute their rate between their start and end on the server clock. The highest sum of rates will be `storage_read_mbps` or `storage_write_mbps`. +- `PREFETCH_TIME_FIT` can fit the `Prefetch` seconds of the last 24 hours over a constant, megabytes and paths. Least squares on these sums yield `per_path_secs` from 100 spans on. +- The same fit on C3D2 gave the fallback of 0.18 s per path. The data were 16938 prefetches in 3 days, with 3.6 s fixed and 29 MB/s. +- `EVAL_HISTORY_DURATION` can average `worker_elapsed_ms` of completed eval jobs per task over 7 days. Tasks without runs use the mean of every run. +- `STORED_TO_NAR_RATIO` can divide the `NarPush` bytes by the NAR bytes of their parent `Compress` span. Spans record stored bytes. The ratio can convert the throughput into NAR bytes. - Each `Windowed` value can hold 5 min, 1 h and 24 h averages. `None` will stand for no samples. A measured zero will stay zero. - Rules read `w1h_or(fallback)` or `w24h_or(fallback)` for instance-relative thresholds. - A failed query will leave its windows empty. The in-memory counts always survive. @@ -79,12 +103,18 @@ flowchart LR | Signal | Measured in | Formula | |---|---|---| -| `network_speed_mbps` | Passthrough NAR upload (`nar.rs`), NAR receive (`nar_recv.rs`), presigned PUT (`object_put.rs`) and presigned download (`download_one_presigned`) | bits / elapsed seconds / 10^6 | -| `disk_speed_mbps` | `build_metrics.rs` after each build | cgroup `disk_read_bytes + disk_write_bytes` in MiB / build seconds | +| `upload_speed_mbps` | Each NAR upload batch (`upload_all`), passthrough, presigned PUT and multipart alike | NAR bits / seconds from the first grant to the end of the batch, packing and compression included / 10^6 | +| `download_speed_mbps` | Each NAR fetch round (`fetch_round`) and each substitution from an upstream cache | NAR bits / transfer seconds / 10^6 | +| `disk_speed_mbps` | `build_metrics.rs` after each build | Daemon `io_read_bytes + io_write_bytes` in MiB / build seconds | | `cpu_core_score` | Startup micro-benchmark, or `GRADIENT_WORKER_SYSTEM_CPU_CORE_SCORE` | Static, sent with `WorkerCapabilities` | -- Network and disk are EWMAs (`alpha = 0.3`) in `gradient-worker-client/src/throughput.rs`, `None` until the first sample. -- `cpu_core_score_mean` is the mean over connected workers with a non-zero score. +- Upload, download and disk are EWMAs (`alpha = 0.3`) in `gradient-worker-client/src/throughput.rs`, `None` until the first sample. +- Batches under 1 MiB stay out of the upload and download speeds. Connection setup would dominate their time. +- Eval-cache blobs feed neither speed. +- `cpu_core_score_mean`, `upload_speed_mean_mbps` and `download_speed_mean_mbps` are means over connected workers with a measured value. +- The wait for the server's upload grant has its own `UploadWait` span. The `NarPush` span can only start at the first grant. +- Substitutions measure the worker's own link, the same link as a cache download. The storage share in `EstimatedTimeRule` models the server side. +- `EstimatedTimeRule` can use the fleet mean speed for a worker without a measurement. ## Adding a Rule diff --git a/docs/src/reference/configuration.md b/docs/src/reference/configuration.md index ee82feed9..2468732a7 100644 --- a/docs/src/reference/configuration.md +++ b/docs/src/reference/configuration.md @@ -157,6 +157,7 @@ Declarative entities under `services.gradient.state` are in the [state reference | Option | Type | Default | Env | Description | |---|---|---|---|---| | `nar.hotCacheBytes` | int | `536870912` | `GRADIENT_NAR_HOT_CACHE_BYTES` | Capacity in bytes of the in-memory NAR cache. | +| `nar.maxConcurrentDownloads` | int | `16` | `GRADIENT_NAR_MAX_CONCURRENT_DOWNLOADS` | NAR downloads from storage that may run at once across all connections, keeping the storage near its best total throughput. | | `nar.maxConcurrentServes` | int | `8` | `GRADIENT_NAR_MAX_CONCURRENT_SERVES` | NAR serving tasks that may run at once per worker connection, bounding memory and storage fan-out for large batches. | | `nar.maxUploadSize` | int | `536870912` | `GRADIENT_NAR_MAX_UPLOAD_SIZE` | Maximum size in bytes of a NAR uploaded to the cache upload endpoint. | | `nar.partialTtlSecs` | int | `86400` | `GRADIENT_NAR_PARTIAL_TTL_SECS` | Seconds since the last write of an unfinished upload staged under ``, after which the next [deep GC](../contributors/internals/nar-storage.md#deep-gc) is removing the upload. `0` is keeping every unfinished upload. | @@ -298,7 +299,7 @@ Declarative entities under `services.gradient.state` are in the [state reference | Option | Type | Default | Env | Description | |---|---|---|---|---| | `upload.bytesBudget` | int | `8589934592` | `GRADIENT_UPLOAD_BYTES_BUDGET` | Total size in bytes of admitted uploads. | -| `upload.concurrency` | int | `16` | `GRADIENT_UPLOAD_CONCURRENCY` | Uploads over 1 MiB (NARs and eval cache blobs) admitted at once across all workers and REST clients. Smaller uploads have a window of 128 of their own. An upload is holding its permit until the object is in storage. | +| `upload.concurrency` | int | `8` | `GRADIENT_UPLOAD_CONCURRENCY` | Uploads over 1 MiB (NARs and eval cache blobs) admitted at once across all workers and REST clients. Smaller uploads have a window of 128 of their own. An upload is holding its permit until the object is in storage. | | `upload.leaseIdleSecs` | int | `300` | `GRADIENT_UPLOAD_LEASE_IDLE_SECS` | Seconds a granted worker upload may go without data before the server is reclaiming its permit and telling the worker to retry. | | `upload.restWaitSecs` | int | `30` | `GRADIENT_UPLOAD_REST_WAIT_SECS` | Seconds a NAR upload to the cache upload endpoint is waiting for a permit before the server is answering with 503 and `Retry-After`. | @@ -326,10 +327,9 @@ Declarative entities under `services.gradient.state` are in the [state reference | Option | Type | Default | Env | Description | |---|---|---|---|---| -| `worker.build.cgroupRoot` | string | `"/sys/fs/cgroup/system.slice/nix-daemon.service"` | `GRADIENT_WORKER_BUILD_CGROUP_ROOT` | Cgroup of the Nix daemon. The daemon is creating each build's cgroup under this cgroup when `worker.build.metrics` is enabled. | | `worker.build.maxConcurrent` | int | `32` | `GRADIENT_WORKER_BUILD_MAX_CONCURRENT` | Maximum simultaneous builds. | | `worker.build.maxCores` | null or (int) | `null` | `GRADIENT_WORKER_BUILD_MAX_CORES` | CPU cores a single build may use, passed as `--cores`. | -| `worker.build.metrics` | bool | `false` | `GRADIENT_WORKER_BUILD_METRICS` | Whether to record per-build peak memory, CPU time and disk I/O. | +| `worker.build.metrics` | bool | `false` | - | Whether to record per-build peak memory, CPU time, disk I/O and out-of-memory kills. | ## `worker.capabilities` diff --git a/docs/src/reference/scheduler-policies.md b/docs/src/reference/scheduler-policies.md index 376d27e46..476d1135e 100644 --- a/docs/src/reference/scheduler-policies.md +++ b/docs/src/reference/scheduler-policies.md @@ -11,45 +11,71 @@ services.gradient.scheduler.scoringPolicy = "resource-aware"; # (1)! | Policy | Rules | Pick when | |---|---|---| | `resource-aware` (default) | All rules below | Workers reporting metrics, with heavy builds meant for machines that fit them | -| `simple` | The first table only | Workers without metrics, or placement by cache warmth and wait time alone | +| `simple` | The first table only | Workers without metrics, or placement by estimated time and wait time alone | An unknown name will fall back to `resource-aware`. ## Negative Scores -The worker will get its highest-scoring job with a total of at least 0 and no veto from any rule. A vetoed or negative job will wait for a better fit. The worker will idle this round without any eligible job. Bonus rules are never going below zero. Only penalties and vetoes can hold a job back. +The worker will get its highest-scoring job with a total of at least 0 and no veto from any rule. A vetoed or negative job will wait for a better fit. The worker will idle this round without any eligible job. Only penalties and vetoes can hold a job back. ## Rules in Both Policies | Rule | Kind | Effect | |---|---|---| -| `MissingPathsRule` | Bonus, up to 200 | Worker already holding most of the inputs | -| `MissingNarSizeRule` | Bonus, up to 500 | Little data to download before the build | -| `RealisedOutputsRule` | Bonus, 2500 | Worker already holding every output and only uploading them | -| `DependencyCountRule` | Bonus, up to 50 | Builds with many direct inputs | -| `WaitTimeRule` | Bonus, growing | Long-waiting jobs rising, against starvation. Counted from the moment dependencies finished | -| `BuiltinDeprioritizeRule` | Bonus, 50 or 100 | Real builds before `builtin` downloads. `builtin` jobs still reaching workers without systems | +| `EstimatedTimeRule` | Bonus, up to 3600 | Jobs expected to finish soonest on this worker, see [Estimated Time](#estimated-time) | +| `WaitTimeRule` | Bonus, up to 4000 | Long-waiting jobs rising, against starvation. Counted from the moment dependencies finished | | `QosRule` | Bonus, 5000 and 1000 | Prioritized jobs beating every other job. Jobs of a build request ([`gradient build`](../guides/build-before-push.md) or [SSH](../guides/build-over-ssh.md)) gaining another 1000 | | `RescoreWaitRule` | Veto | Holding a build until a worker reported its missing data size. Lifted after 4 rounds | -| `ReserveFetchWorkersRule` | Penalty | Keeping fetch-capable workers free for fetching while capacity is short | +| `ReserveFetchWorkersRule` | Penalty, up to 300 | Keeping fetch-capable workers free for fetching while capacity is short | +| `TransferLimitRule` | Penalty, 4000 | Holding large transfers while the transfer slots are full, see [Transfer Limits](#transfer-limits) | The **Prioritize** entry in the task or evaluation menu can set the `QosRule` flag on a build and its dependencies, or on a whole evaluation. The flag will clear on a failed or aborted build or evaluation. ## Rules in `resource-aware` Only -The memory predictions (`ResourceFitRule`, the out-of-memory check) are requiring `services.gradient.worker.build.metrics` on the workers. They also need earlier builds of the same package. The CPU and memory saturation check will use live worker load. +The saturation check can use live worker load and the predicted peak memory. The prediction needs earlier builds of the same package. | Rule | Kind | Effect | |---|---|---| -| `ResourceFitRule` | Penalty | Predicted peak memory above the worker's free memory, for builds and evaluations | -| `ResourceSaturationRule` | Penalty, up to -10000 | Worker above 80% CPU (90% for `builtin` jobs) or below 10% free memory, or a likely out-of-memory build | -| `PreferLocalBuildRule` | Bonus | `preferLocalBuild` derivations on a worker holding most of the closure | -| `NetworkAffinityRule` | Bonus | Fixed-output downloads on workers with fast network | -| `DiskAffinityRule` | Bonus | Disk-heavy builds on workers with fast disks | -| `CpuAffinityRule` | Bonus or penalty, up to 1200 | Long builds on faster cores than the fleet average, away from slower ones | +| `ResourceSaturationRule` | Penalty, up to -17200 | Worker above 80% CPU (90% for `builtin` jobs) or below 10% free memory, or a likely out-of-memory build | | `FairShareRule` | Penalty, disabled | Would slow projects holding a large share of running work | -Workers are measuring network and disk speed from their own NAR transfers and builds. The affinity rules are adding nothing until the first transfer. +## Estimated Time + +`EstimatedTimeRule` can sum the seconds a job should take on the worker. Each second costs one point below the cap of 3600. + +| Part | Estimate | +|---|---| +| Download | Missing NAR size over the slower of the worker's download speed and its share of the storage read throughput | +| Paths | Missing paths times the per-path time of recent prefetches | +| Build | Build time of earlier builds of the package, scaled by their CPU score over the worker's, plus 4% per build already running there | +| Out of memory | Build time again, weighted by the chance of an out-of-memory kill | +| Upload | Output size of earlier builds over the slower of the worker's upload speed and its share of the storage write throughput | + +- A worker already holding every output only needs the upload. +- An evaluation can take the mean run time of its task over the last 7 days, or of every task. Downloads and paths stay out. +- The chance of an out-of-memory kill is the package's kill rate plus the share of predicted peak memory above free memory. The chance can reach at most 1. +- The storage share is the best total throughput of the last hour, split across the transfers in flight plus this one. +- The storage throughput will be 150 MB/s for reads and 140 MB/s for writes before the first measurement. +- A least-squares fit of the prefetch time over its megabytes and paths in the last 24 hours can yield the per-path time. Every path can cost 0.18 s before 100 prefetches exist. +- A package without history can use the instance's mean build time. +- The CPU score ratio must stay between 0.5 and 2. +- Workers measure upload, download and disk speed from their own NAR transfers, substitutions and builds. Transfers under 1 MiB stay out. + +The cap of 3600 lies below the 4000 of `WaitTimeRule`. A long-waiting job can always overtake a shorter one. A prioritized job with its 5000 can outrank any estimate gap. The penalty of `ResourceSaturationRule` includes the cap and can still keep a build off a saturated worker. + +## Transfer Limits + +`TransferLimitRule` can hold a build below 0 while the slots of the server are full. + +| Slots | Full at | Held build | +|---|---|---| +| Downloads | Builds in prefetch reaching [`nar.maxConcurrentDownloads`](configuration.md#nar) | Missing NAR size above the mean of the last hour | +| Uploads | Builds in upload reaching [`upload.concurrency`](configuration.md#upload) | Expected output above 1 MiB | + +- A build with local inputs or a small output can take the free worker slot instead. +- The hold of 4000 is as large as the cap of `WaitTimeRule`. A held build will reach 0 once its wait bonus makes up for the hold. ## Custom Policies diff --git a/docs/src/roadmap.md b/docs/src/roadmap.md index e2a14d013..d39d5bc3e 100644 --- a/docs/src/roadmap.md +++ b/docs/src/roadmap.md @@ -24,13 +24,21 @@ The first release with a stability pledge. Deep garbage collection will run slowly in the background at all times. Checkpoints let each pass resume where the last one stopped. -- :material-server-network: **Cluster Jobs** +- :material-server-network: **Cluster Jobs (beta)** Jobs that run on many workers at once. Gradient will allocate all members together. Jobs needing a fast interconnect get all members from one [zone](concepts/workers.md#zones). - :material-console-network: **SSH Builds** - An [`ssh-ng://` store](guides/build-over-ssh.md) for every project. `nixos-rebuild --build-host` and `nix copy` talk straight to the CI workers and caches. + An [ssh-ng:// store](guides/build-over-ssh.md) for every project. `nixos-rebuild --build-host` and `nix copy` talk straight to the CI workers and caches. + +- :material-palette: **Gradient.CI Servers** + + Gradient Remote Worker Pay-As-You-Go Service from [Gradient.CI Servers](https://servers.gradient.ci). This Service will help use to continue the development of Gradient. + +- :material-palette: **Gradient Teams** + + Organizational structures for better team management with SSO. - :material-palette: **Corporate Design** @@ -44,13 +52,13 @@ The first release with a stability pledge. ## v2.1.0 -Faster evaluation of large flakes. +Jobs beyond the Nix sandbox.
-- :material-graph: **Multi-Node Evaluations** +- :material-play-network: **Runner Workers** - One evaluation split across many workers. Large flakes like nixpkgs or fleets of NixOS hosts finish in a fraction of the time. + Workers that build outside the sandbox. Integration testing with network access, real hardware or deployment credentials.
@@ -60,13 +68,13 @@ Faster evaluation of large flakes. ## v2.2.0 -Jobs beyond the Nix sandbox. +Faster evaluation of large flakes.
-- :material-play-network: **Runner Workers** +- :material-graph: **Multi-Node Evaluations** - Workers that build outside the sandbox. Integration testing with network access, real hardware or deployment credentials. + One evaluation split across many workers. Large flakes like nixpkgs or fleets of NixOS hosts finish in a fraction of the time.
diff --git a/docs/src/ui/job-board.md b/docs/src/ui/job-board.md index 431c36c10..020078b7c 100644 --- a/docs/src/ui/job-board.md +++ b/docs/src/ui/job-board.md @@ -14,8 +14,8 @@ Activity of the scheduler and the workers, right now and over time. Live jobs, t | Workers | Fleet size, load by capability, system and feature, slot use per worker | Spot the missing kind of worker | | Cache | Stored size, traffic, growth, latency per upstream cache | | | Storage | NAR storage latency and errors per operation (file or S3), writer lane fill and send stalls, NAR delivery queue and failures. Every chart on one time axis over the whole window, minute resolution up to 6 h. Superusers only | Pick the time window | -| Network | NAR egress, worker network and disk speed, HTTP latency per route | | -| Jobs | The costliest builds by wall time, peak RAM, CPU time, disk I/O and network | Pick the time window | +| Network | NAR egress, worker upload, download and disk speed, HTTP latency per route | | +| Jobs | The costliest builds by wall time, peak RAM, CPU time and disk I/O | Pick the time window | | Evals | The costliest evaluations by time, peak memory, thunks, function calls and allocations | Pick the time window | | System Health | Server runtime, metric pipeline lag, route stats. Superusers only | **Run Deep GC**. **Enable Draining** before stopping the server | @@ -61,7 +61,7 @@ Jobs opened from **Live Jobs** show five parts. - The server's marks: **Queued**, **Ready**, **Assigned**, **Finished**. - **Worker tail**: the worker's time after its last phase, for finished jobs. - **Transit**: the rest of the time between **Assigned** and **Finished** of a finished job, spent on the network and in queues on both ends. -- The worker timeline: nested phases (fetch, evaluate, build, compress, NAR push) with duration, share and bytes moved. +- The worker timeline: nested phases (fetch, evaluate, build, compress, upload wait, NAR push) with duration, share and bytes moved. - The score breakdown: each scoring rule's contribution to the winning worker. The timeline shows where a slow job spent its time. The score will explain the worker choice for the job. @@ -94,7 +94,7 @@ The timeline shows where a slow job spent its time. The score will explain the w Load charts set the running jobs of one kind against the slots of the workers taking that kind. A chart near 100% will name the worker type to add: evaluation, build, a system such as `aarch64-linux`, or a feature such as `kvm`. -**Project -> Workers -> Metrics** has the per-worker CPU, memory, disk and network history. The connection and disconnect history will stay there for [`retentionDays`](../reference/configuration.md#general) (90 days by default). +**Project -> Workers -> Metrics** has the per-worker CPU, memory, disk, upload and download history. The connection and disconnect history will stay there for [`retentionDays`](../reference/configuration.md#general) (90 days by default). ## Visibility diff --git a/frontend/src/app/core/services/board.service.ts b/frontend/src/app/core/services/board.service.ts index 0e7787e5a..389dab668 100644 --- a/frontend/src/app/core/services/board.service.ts +++ b/frontend/src/app/core/services/board.service.ts @@ -77,14 +77,15 @@ export interface WorkerContextView { ram_free_mb: number | null; cpu_usage_pct: number | null; disk_speed_mbps: number | null; - network_speed_mbps: number | null; + upload_speed_mbps: number | null; + download_speed_mbps: number | null; } export interface DerivationRef { build_id: string; drv_path: string; pname: string | null; } export interface JobHistoryView { - peak_ram_mb: number; avg_cpu_time_ms: number; build_time_ms: number; - avg_disk_bytes: number; oom_rate: number; samples: number; + peak_ram_mb: number | null; avg_cpu_time_ms: number | null; build_time_ms: number | null; + avg_disk_bytes: number | null; output_nar_size: number | null; oom_rate: number; samples: number; } export interface JobContextView { @@ -114,7 +115,6 @@ export interface InstanceContextView { cpu_time_ms: Windowed; avg_cpu_pct: Windowed; disk_bytes: Windowed; - network_mbps: Windowed; oom_rate: Windowed; closure_size: Windowed; nar_size_mb: Windowed; @@ -309,7 +309,8 @@ export interface HttpRouteStat { export interface WorkerNet { worker_id: string | null; - network_speed_mbps: number | null; + upload_speed_mbps: number | null; + download_speed_mbps: number | null; disk_speed_mbps: number | null; } @@ -521,7 +522,7 @@ export class BoardService { } getExpensiveByResource( - metric: 'ram' | 'cpu' | 'disk' | 'network', + metric: 'ram' | 'cpu' | 'disk', windowDays = 30 ): Observable { return this.api.get( diff --git a/frontend/src/app/core/services/workers.service.ts b/frontend/src/app/core/services/workers.service.ts index 906bbe415..4831ba40d 100644 --- a/frontend/src/app/core/services/workers.service.ts +++ b/frontend/src/app/core/services/workers.service.ts @@ -21,7 +21,8 @@ export interface WorkerSamplePoint { ram_free_mb: number | null; ram_total_mb: number | null; disk_speed_mbps: number | null; - network_speed_mbps: number | null; + upload_speed_mbps: number | null; + download_speed_mbps: number | null; assigned_jobs: number; max_concurrent_builds: number; state: number; diff --git a/frontend/src/app/features/board/expensive-jobs/expensive-jobs.component.scss b/frontend/src/app/features/board/expensive-jobs/expensive-jobs.component.scss index 0d1509453..8d5b1fb4a 100644 --- a/frontend/src/app/features/board/expensive-jobs/expensive-jobs.component.scss +++ b/frontend/src/app/features/board/expensive-jobs/expensive-jobs.component.scss @@ -12,7 +12,6 @@ .controls { display: flex; gap: 1.5rem; margin-bottom: 1rem; color: var(--gr-text-secondary); align-items: center; } .controls label { display: inline-flex; align-items: center; gap: 0.35rem; cursor: pointer; } select { background: var(--gr-surface-raised); color: var(--gr-text-primary); border: 1px solid var(--gr-border-subtle); border-radius: $border-radius-sm; padding: 0.25rem; } -.note { color: var(--gr-text-muted); font-size: $font-size-xs; margin: 0 0 0.75rem; } .muted { color: var(--gr-text-muted); } h2 { color: var(--gr-text-primary); font-size: $font-size-lg; margin: 1.5rem 0 0.75rem; } diff --git a/frontend/src/app/features/board/expensive-jobs/expensive-jobs.component.ts b/frontend/src/app/features/board/expensive-jobs/expensive-jobs.component.ts index 644e1df53..ff3c9bcef 100644 --- a/frontend/src/app/features/board/expensive-jobs/expensive-jobs.component.ts +++ b/frontend/src/app/features/board/expensive-jobs/expensive-jobs.component.ts @@ -17,7 +17,7 @@ import { MetricChartComponent } from '@shared/ui'; import { firstLoad } from '../first-load'; import { formatDuration, formatQuantity } from '@shared/text'; -type Tab = 'time' | 'ram' | 'cpu' | 'disk' | 'network'; +type Tab = 'time' | 'ram' | 'cpu' | 'disk'; @Component({ selector: 'app-board-expensive-jobs', @@ -55,16 +55,13 @@ type Tab = 'time' | 'ram' | 'cpu' | 'disk' | 'network'; } @else { - @if (tab() === 'network') { -

Network is a host-level peak measured during each build's window (cgroup v2 has no per-build network accounting); exact only when the build is the host's sole network user.

- } #Derivation{{ valueHeader() }}Worker @for (r of resources(); track r.derivation; let i = $index) { {{ i + 1 }}{{ r.name }}{{ quantity(r.value, r.unit) }}{{ r.worker_name ?? (r.worker || '-') }} } @empty { - No per-build metrics recorded in this window (needs cgroup metrics enabled on workers). + No per-build metrics recorded in this window (needs build metrics enabled on workers). } @@ -101,7 +98,6 @@ export class BoardExpensiveJobsComponent implements OnInit { { key: 'ram', label: 'Peak RAM' }, { key: 'cpu', label: 'CPU time' }, { key: 'disk', label: 'Disk I/O' }, - { key: 'network', label: 'Network' }, ]; readonly duration = formatDuration; @@ -122,7 +118,7 @@ export class BoardExpensiveJobsComponent implements OnInit { this.board.getExpensive(this.windowDays).pipe(this.first.track()).subscribe((b) => this.builds.set(b)); } else { this.board - .getExpensiveByResource(this.tab() as 'ram' | 'cpu' | 'disk' | 'network', this.windowDays) + .getExpensiveByResource(this.tab() as 'ram' | 'cpu' | 'disk', this.windowDays) .pipe(this.first.track()) .subscribe((r) => this.resources.set(r)); } diff --git a/frontend/src/app/features/board/job-detail/job-detail.component.spec.ts b/frontend/src/app/features/board/job-detail/job-detail.component.spec.ts index 8685b55a1..61153b9c1 100644 --- a/frontend/src/app/features/board/job-detail/job-detail.component.spec.ts +++ b/frontend/src/app/features/board/job-detail/job-detail.component.spec.ts @@ -49,7 +49,8 @@ const DETAIL: AssignedJobDetail = { ram_free_mb: 16000, cpu_usage_pct: 42, disk_speed_mbps: null, - network_speed_mbps: 940, + upload_speed_mbps: 940, + download_speed_mbps: null, }, job_context: { kind: 'Build', @@ -71,6 +72,7 @@ const DETAIL: AssignedJobDetail = { avg_cpu_time_ms: 1200, build_time_ms: 4200, avg_disk_bytes: 5000, + output_nar_size: null, oom_rate: 0, samples: 4, }, @@ -83,7 +85,6 @@ const DETAIL: AssignedJobDetail = { cpu_time_ms: { w5m: 1, w1h: 2, w24h: 3 }, avg_cpu_pct: { w5m: 1, w1h: 2, w24h: 3 }, disk_bytes: { w5m: 1, w1h: 2, w24h: 3 }, - network_mbps: { w5m: 1, w1h: 2, w24h: 3 }, oom_rate: { w5m: 0, w1h: 0, w24h: 0 }, closure_size: { w5m: 1, w1h: 2, w24h: 3 }, nar_size_mb: { w5m: 1, w1h: 2, w24h: 3 }, diff --git a/frontend/src/app/features/board/job-detail/job-detail.component.ts b/frontend/src/app/features/board/job-detail/job-detail.component.ts index c9bd24287..672a6e56a 100644 --- a/frontend/src/app/features/board/job-detail/job-detail.component.ts +++ b/frontend/src/app/features/board/job-detail/job-detail.component.ts @@ -152,7 +152,8 @@ interface RuleRow { RAM total{{ formatMegabytes(j.worker_context.ram_total_mb) }} RAM free{{ formatMegabytes(j.worker_context.ram_free_mb) }} Disk speed{{ j.worker_context.disk_speed_mbps != null ? (j.worker_context.disk_speed_mbps | number) + ' MB/s' : '-' }} - Network speed{{ j.worker_context.network_speed_mbps != null ? (j.worker_context.network_speed_mbps | number) + ' Mbps' : '-' }} + Upload speed{{ j.worker_context.upload_speed_mbps != null ? (j.worker_context.upload_speed_mbps | number) + ' Mbps' : '-' }} + Download speed{{ j.worker_context.download_speed_mbps != null ? (j.worker_context.download_speed_mbps | number) + ' Mbps' : '-' }} @@ -192,6 +193,7 @@ interface RuleRow { Avg CPU time{{ formatDuration(h.avg_cpu_time_ms) }} Build time{{ formatDuration(h.build_time_ms) }} Avg disk usage{{ formatBytes(h.avg_disk_bytes) }} + Output size{{ formatBytes(h.output_nar_size) }} OOM rate{{ h.oom_rate | number: '1.0-3' }} Samples{{ h.samples }} @@ -451,7 +453,6 @@ export class BoardJobDetailComponent implements OnInit { { name: 'CPU time (ms)', w: inst.cpu_time_ms }, { name: 'Avg CPU (%)', w: inst.avg_cpu_pct }, { name: 'Disk bytes', w: inst.disk_bytes }, - { name: 'Network (Mbps)', w: inst.network_mbps }, { name: 'OOM rate', w: inst.oom_rate }, { name: 'Closure size', w: inst.closure_size }, { name: 'NAR size (MB)', w: inst.nar_size_mb }, diff --git a/frontend/src/app/features/board/job-detail/job-timeline.component.ts b/frontend/src/app/features/board/job-detail/job-timeline.component.ts index 1e6af4f9f..56414827b 100644 --- a/frontend/src/app/features/board/job-detail/job-timeline.component.ts +++ b/frontend/src/app/features/board/job-detail/job-timeline.component.ts @@ -80,6 +80,7 @@ const PHASE_LABELS: Record = { download: 'Download', build: 'Build', compress: 'Compress', + upload_wait: 'Upload wait', nar_push: 'NAR push', cache_query_wait: 'Cache-query wait', }; diff --git a/frontend/src/app/features/board/network/network.component.ts b/frontend/src/app/features/board/network/network.component.ts index 607cf0aba..6e464bbf6 100644 --- a/frontend/src/app/features/board/network/network.component.ts +++ b/frontend/src/app/features/board/network/network.component.ts @@ -32,11 +32,11 @@ type HttpSortKey = keyof Pick @@ -134,8 +134,9 @@ export class BoardNetworkComponent implements OnInit { workerCats = computed(() => (this.stats()?.workers ?? []).map((w) => (w.worker_id ?? '-').slice(0, 12)) ); - netSeries = computed(() => [ - { name: 'network', data: (this.stats()?.workers ?? []).map((w) => w.network_speed_mbps ?? 0) }, + transferSeries = computed(() => [ + { name: 'upload', data: (this.stats()?.workers ?? []).map((w) => w.upload_speed_mbps ?? 0) }, + { name: 'download', data: (this.stats()?.workers ?? []).map((w) => w.download_speed_mbps ?? 0) }, ]); diskSeries = computed(() => [ { name: 'disk', data: (this.stats()?.workers ?? []).map((w) => w.disk_speed_mbps ?? 0) }, diff --git a/frontend/src/app/features/evaluations/evaluation-log/evaluation-log.component.html b/frontend/src/app/features/evaluations/evaluation-log/evaluation-log.component.html index d503982af..d41a75f87 100644 --- a/frontend/src/app/features/evaluations/evaluation-log/evaluation-log.component.html +++ b/frontend/src/app/features/evaluations/evaluation-log/evaluation-log.component.html @@ -38,7 +38,7 @@

Evaluation not found

- + {{ completedBuildsCount() }}/{{ totalBuildsCount() }}
@@ -146,7 +146,7 @@

Evaluation not found

- +
Duration diff --git a/frontend/src/app/features/projects/workers/worker-metrics/worker-metrics.component.ts b/frontend/src/app/features/projects/workers/worker-metrics/worker-metrics.component.ts index f65a3869d..b9982bc8b 100644 --- a/frontend/src/app/features/projects/workers/worker-metrics/worker-metrics.component.ts +++ b/frontend/src/app/features/projects/workers/worker-metrics/worker-metrics.component.ts @@ -51,7 +51,7 @@ import { formatMegabytes, formatPercent, formatQuantity } from '@shared/text'; - + @@ -95,7 +95,10 @@ export class WorkerMetricsComponent implements OnInit { times = computed(() => this.samples().map((s) => s.at.slice(11, 16))); cpuSeries = computed(() => [{ name: 'cpu', data: this.samples().map((s) => s.cpu_usage_pct ?? 0) }]); ramSeries = computed(() => [{ name: 'ram free', data: this.samples().map((s) => s.ram_free_mb ?? 0) }]); - netSeries = computed(() => [{ name: 'network', data: this.samples().map((s) => s.network_speed_mbps ?? 0) }]); + transferSeries = computed(() => [ + { name: 'upload', data: this.samples().map((s) => s.upload_speed_mbps ?? 0) }, + { name: 'download', data: this.samples().map((s) => s.download_speed_mbps ?? 0) }, + ]); diskSeries = computed(() => [{ name: 'disk', data: this.samples().map((s) => s.disk_speed_mbps ?? 0) }]); loadSeries = computed(() => [{ name: 'assigned', data: this.samples().map((s) => s.assigned_jobs) }]); diff --git a/frontend/src/app/features/tasks/task-detail/task-detail.component.html b/frontend/src/app/features/tasks/task-detail/task-detail.component.html index b566448f8..8b74331b6 100644 --- a/frontend/src/app/features/tasks/task-detail/task-detail.component.html +++ b/frontend/src/app/features/tasks/task-detail/task-detail.component.html @@ -74,7 +74,7 @@

{{ proj.display_name }}

@for (e of evaluations(); track e.id) {
- + {{ evalTitle(e) }}
{{ triggerLabel(e) }}{{ evalDuration(e) }}
@@ -89,12 +89,12 @@

{{ proj.display_name }}

- +

{{ evalTitle(sel) }}

@if (isRunning(sel.status)) { - + }
@@ -139,7 +139,7 @@

{{ evalTitle(sel) }}

(contextmenu)="openPkgContextMenu($event, ep, sel.id, pkgMenu)"> - + {{ attrLabel(ep.eval) }} {{ ep.architecture }} diff --git a/frontend/src/app/shared/ui/eval-status-badge/eval-status-badge.component.ts b/frontend/src/app/shared/ui/eval-status-badge/eval-status-badge.component.ts index d3ceab855..34ce6a8e3 100644 --- a/frontend/src/app/shared/ui/eval-status-badge/eval-status-badge.component.ts +++ b/frontend/src/app/shared/ui/eval-status-badge/eval-status-badge.component.ts @@ -15,7 +15,7 @@ import { StatusIconComponent } from '../status-icon/status-icon.component'; imports: [StatusIconComponent], template: ` - + {{ label() }} `, @@ -24,6 +24,7 @@ import { StatusIconComponent } from '../status-icon/status-icon.component'; }) export class EvalStatusBadgeComponent { status = input.required(); + prioritized = input(false); phase = computed(() => evaluationPhase(this.status())); diff --git a/frontend/src/app/shared/ui/status-icon/status-icon.component.spec.ts b/frontend/src/app/shared/ui/status-icon/status-icon.component.spec.ts index 37b95573a..59c566a7a 100644 --- a/frontend/src/app/shared/ui/status-icon/status-icon.component.spec.ts +++ b/frontend/src/app/shared/ui/status-icon/status-icon.component.spec.ts @@ -107,6 +107,25 @@ describe('StatusIconComponent', () => { expect(spin.updatePlaybackRate).toHaveBeenLastCalledWith(1); }); + it('spins a prioritized run 2.5 times as fast and follows the flag', () => { + const fixture = TestBed.createComponent(StatusIconComponent); + fixture.componentRef.setInput('phase', 'running'); + fixture.componentRef.setInput('prioritized', true); + fixture.detectChanges(); + expect(spin.updatePlaybackRate).toHaveBeenLastCalledWith(2.5); + fixture.componentRef.setInput('prioritized', false); + fixture.detectChanges(); + expect(spin.updatePlaybackRate).toHaveBeenLastCalledWith(1); + }); + + it('keeps a prioritized queue at its own pace', () => { + const fixture = TestBed.createComponent(StatusIconComponent); + fixture.componentRef.setInput('phase', 'queued'); + fixture.componentRef.setInput('prioritized', true); + fixture.detectChanges(); + expect(spin.updatePlaybackRate).toHaveBeenLastCalledWith(0.25); + }); + it('finishes the current lap instead of snapping when the run ends', () => { const fixture = render('running'); change(fixture, 'success'); diff --git a/frontend/src/app/shared/ui/status-icon/status-icon.component.ts b/frontend/src/app/shared/ui/status-icon/status-icon.component.ts index fd83b2c23..e4e2ff43e 100644 --- a/frontend/src/app/shared/ui/status-icon/status-icon.component.ts +++ b/frontend/src/app/shared/ui/status-icon/status-icon.component.ts @@ -23,6 +23,7 @@ export type StatusIconSize = 'sm' | 'md'; const SPIN_RATE: Partial> = { queued: 0.25, running: 1 }; const SPIN_KEYFRAMES: Keyframe[] = [{ transform: 'rotate(0deg)' }, { transform: 'rotate(360deg)' }]; const SPIN_LAP_MS = 1000; +const PRIORITIZED_RUN_BOOST = 2.5; function motionAllowed(): boolean { return typeof Element.prototype.animate === 'function' @@ -65,6 +66,7 @@ export class StatusIconComponent { phase = input.required(); size = input('md'); label = input(); + prioritized = input(false); private readonly spinner = viewChild.required>('spinner'); private readonly motion = motionAllowed(); @@ -78,15 +80,15 @@ export class StatusIconComponent { protected readonly animate = computed(() => this.motion && this.changed()); constructor() { - afterRenderEffect(() => this.syncSpin(this.phase())); + afterRenderEffect(() => this.syncSpin(this.phase(), this.prioritized())); inject(DestroyRef).onDestroy(() => this.spin?.cancel()); } - private syncSpin(phase: StatusPhase): void { + private syncSpin(phase: StatusPhase, prioritized: boolean): void { if (!this.motion) return; const rate = SPIN_RATE[phase]; if (rate === undefined) this.finishLap(); - else this.spinning().updatePlaybackRate(rate); + else this.spinning().updatePlaybackRate(phase === 'running' && prioritized ? rate * PRIORITIZED_RUN_BOOST : rate); } private spinning(): Animation { diff --git a/nix/modules/gradient-worker.nix b/nix/modules/gradient-worker.nix index de43bec82..e2640a72b 100644 --- a/nix/modules/gradient-worker.nix +++ b/nix/modules/gradient-worker.nix @@ -329,20 +329,11 @@ in { type = lib.types.bool; default = false; description = '' - Whether to record per-build peak memory, CPU time and disk I/O. Enabling it is turning on - Nix's experimental `cgroups` feature and `use-cgroups` and delegating cgroup controllers to - {file}`nix-daemon.service`. Peak memory and disk I/O need Gradient's Nix fork on the - daemon, and {option}`nix.package` is defaulting to its package. Wall-clock time is always - recorded. - ''; - }; - - cgroupRoot = lib.mkOption { - type = lib.types.str; - default = "/sys/fs/cgroup/system.slice/nix-daemon.service"; - description = '' - Cgroup of the Nix daemon. The daemon is creating each build's cgroup under this cgroup when - {option}`services.gradient.worker.build.metrics` is enabled. + Whether to record per-build peak memory, CPU time, disk I/O and out-of-memory kills. Enabling + it is turning on Nix's experimental `cgroups` feature and `use-cgroups` and delegating cgroup + controllers to {file}`nix-daemon.service`. The daemon is reporting the values in each build + result. Peak memory, disk I/O and out-of-memory kills need Gradient's Nix fork on the daemon, + and {option}`nix.package` is defaulting to its package. Wall-clock time is always recorded. ''; }; }; @@ -552,8 +543,6 @@ in { GRADIENT_WORKER_EVAL_METRICS = lib.boolToString cfg.eval.metrics; GRADIENT_WORKER_EVAL_CACHE_SHARE = lib.boolToString cfg.eval.cache.share; GRADIENT_WORKER_BUILD_MAX_CONCURRENT = toString cfg.build.maxConcurrent; - GRADIENT_WORKER_BUILD_METRICS = lib.boolToString cfg.build.metrics; - GRADIENT_WORKER_BUILD_CGROUP_ROOT = cfg.build.cgroupRoot; GRADIENT_WORKER_NAR_MAX_CONCURRENT_UPLOADS = toString cfg.nar.maxConcurrentUploads; GRADIENT_WORKER_NAR_PARTIAL_TTL_SECS = toString cfg.nar.partialTtlSecs; GRADIENT_WORKER_LOG_LEVEL_DEFAULT = cfg.log.level.default; diff --git a/nix/modules/gradient.nix b/nix/modules/gradient.nix index 908c2371d..e7e7fa9c2 100644 --- a/nix/modules/gradient.nix +++ b/nix/modules/gradient.nix @@ -83,6 +83,7 @@ in { (lib.mkRemovedOptionModule [ "services" "gradient" "nar" "maxBufferBytes" ] "replaced by services.gradient.upload.concurrency and services.gradient.upload.bytesBudget") (lib.mkRemovedOptionModule [ "services" "gradient" "scheduler" "recordCandidates" ] "runner-up candidates were never recorded") (lib.mkRemovedOptionModule [ "services" "gradient" "oidc" "iconUrl" ] "the login page never showed the icon") + (lib.mkRemovedOptionModule [ "services" "gradient" "worker" "build" "cgroupRoot" ] "the Nix daemon reports build resource usage in the build result") (lib.mkRenamedOptionModule [ "services" "gradient" "scheduler" "dispatchRetentionDays" ] [ "services" "gradient" "retentionDays" ]) ]; @@ -529,7 +530,7 @@ in { upload = { concurrency = lib.mkOption { type = lib.types.ints.positive; - default = 16; + default = 8; description = '' Uploads over 1 MiB (NARs and eval cache blobs) admitted at once across all workers and REST clients. Smaller uploads have a window of 128 of their own. An upload is holding its @@ -626,6 +627,15 @@ in { ''; }; + maxConcurrentDownloads = lib.mkOption { + type = lib.types.ints.positive; + default = 16; + description = '' + NAR downloads from storage that may run at once across all connections, keeping the + storage near its best total throughput. + ''; + }; + partialTtlSecs = lib.mkOption { type = lib.types.ints.unsigned; default = 86400; @@ -1438,6 +1448,7 @@ in { GRADIENT_NAR_STORAGE_OPEN_TIMEOUT_SECS = toString cfg.nar.storageOpenTimeoutSecs; GRADIENT_NAR_SEND_CHUNK_TIMEOUT_SECS = toString cfg.nar.sendChunkTimeoutSecs; GRADIENT_NAR_MAX_CONCURRENT_SERVES = toString cfg.nar.maxConcurrentServes; + GRADIENT_NAR_MAX_CONCURRENT_DOWNLOADS = toString cfg.nar.maxConcurrentDownloads; GRADIENT_NAR_PARTIAL_TTL_SECS = toString cfg.nar.partialTtlSecs; GRADIENT_CACHE_UPSTREAM_QUERY_CONCURRENCY = toString cfg.cache.upstreamQueryConcurrency; GRADIENT_CACHE_MAX_STORAGE_GB = toString cfg.cache.maxStorageGb; diff --git a/nix/patches/nix/0017-libstore-keep-build-statistics-in-DerivationTrampoli.patch b/nix/patches/nix/0017-libstore-keep-build-statistics-in-DerivationTrampoli.patch new file mode 100644 index 000000000..cb89bf0e7 --- /dev/null +++ b/nix/patches/nix/0017-libstore-keep-build-statistics-in-DerivationTrampoli.patch @@ -0,0 +1,55 @@ +From ef85fe76c5212dfb9d4d3fe71935331c8963c1cb Mon Sep 17 00:00:00 2001 +From: Dennis Wuitz +Date: Sun, 4 Oct 2026 11:05:44 +0200 +Subject: [PATCH 17/18] libstore: keep build statistics in + DerivationTrampolineGoal results + +The trampoline goal rebuilt its result from the status and outputs +only, so timesBuilt, startTime, stopTime, cpuUser and cpuSystem never +reached clients of Worker::buildDerivation, e.g. daemon clients using +BuildDerivation. +--- + src/libstore/build/derivation-trampoline-goal.cc | 12 +++++++++--- + 1 file changed, 9 insertions(+), 3 deletions(-) + +diff --git a/src/libstore/build/derivation-trampoline-goal.cc b/src/libstore/build/derivation-trampoline-goal.cc +index 14864ed..1c34798 100644 +--- a/src/libstore/build/derivation-trampoline-goal.cc ++++ b/src/libstore/build/derivation-trampoline-goal.cc +@@ -184,6 +184,7 @@ Goal::Co DerivationTrampolineGoal::haveDerivation(StorePath drvPath, Derivation + /* Report the exit status of *some* failing goal. This might not be strictly + correct, since multiple subgoals can fail independently, but this should be + a good enough heuristic without --keep-going. */ ++ buildResult = g->buildResult; + co_return doneFailure(exitCode, *failure); + } + +@@ -198,8 +199,6 @@ Goal::Co DerivationTrampolineGoal::haveDerivation(StorePath drvPath, Derivation + for (const auto & success : successes) + std::ranges::copy(success.builtOutputs, std::inserter(outputs, outputs.end())); + +- auto statuses = successes | std::views::transform(&BuildResult::Success::status); +- + /* Aggregate the status code. If some outputs we already valid, but we had + to build/substitute the other ones, report it as the smallest common + denominator. */ +@@ -224,8 +223,15 @@ Goal::Co DerivationTrampolineGoal::haveDerivation(StorePath drvPath, Derivation + return toPriority(a) < toPriority(b); + }; + ++ /* Keep the statistics (build times, CPU usage) of the goal that did the ++ most work, `Worker::buildDerivation` reports them to daemon clients. */ ++ const auto & mostWork = *std::ranges::min_element(concreteDrvGoals, compareSuccesses, [](const GoalPtr & goal) { ++ return goal->buildResult.tryGetSuccess()->status; ++ }); ++ buildResult = mostWork->buildResult; ++ + co_return doneSuccess({ +- .status = std::ranges::min(statuses, compareSuccesses), ++ .status = mostWork->buildResult.tryGetSuccess()->status, + .builtOutputs = std::move(outputs), + }); + } +-- +2.55.0 + diff --git a/nix/patches/nix/0018-libstore-report-cgroup-memory-io-and-oom-stats-in-Bu.patch b/nix/patches/nix/0018-libstore-report-cgroup-memory-io-and-oom-stats-in-Bu.patch new file mode 100644 index 000000000..8ce79dc98 --- /dev/null +++ b/nix/patches/nix/0018-libstore-report-cgroup-memory-io-and-oom-stats-in-Bu.patch @@ -0,0 +1,322 @@ +From 6033fa639dd6e21500e89410b55bd122c8cb4df4 Mon Sep 17 00:00:00 2001 +From: Dennis Wuitz +Date: Sun, 4 Oct 2026 11:05:44 +0200 +Subject: [PATCH 18/18] libstore: report cgroup memory, io and oom stats in + BuildResult + +Read memory.peak, the rbytes/wbytes sums of io.stat and the oom_kill +count of memory.events from the build cgroup next to cpu.stat. When the +build-resource-usage worker protocol feature is negotiated, BuildResult +carries memoryPeak, ioReadBytes, ioWriteBytes and oomKills after +cpuSystem, each as an optional 64-bit number. +--- + .../json/schema/build-result-v1.yaml | 28 +++++++++ + src/libstore/build-result.cc | 15 +++++ + .../include/nix/store/build-result.hh | 7 +++ + .../include/nix/store/worker-protocol.hh | 8 +++ + .../linux/build/linux-derivation-builder.cc | 4 ++ + src/libstore/worker-protocol.cc | 37 ++++++++++++ + src/libutil/linux/cgroup.cc | 59 ++++++++++++++----- + src/libutil/linux/include/nix/util/cgroup.hh | 1 + + 8 files changed, 143 insertions(+), 16 deletions(-) + +diff --git a/doc/manual/source/protocols/json/schema/build-result-v1.yaml b/doc/manual/source/protocols/json/schema/build-result-v1.yaml +index d0d8d8a..696f956 100644 +--- a/doc/manual/source/protocols/json/schema/build-result-v1.yaml ++++ b/doc/manual/source/protocols/json/schema/build-result-v1.yaml +@@ -49,6 +49,34 @@ properties: + description: | + System CPU time the build took, in microseconds. + ++ memoryPeak: ++ type: integer ++ minimum: 0 ++ title: Peak memory ++ description: | ++ Peak memory usage of the build, in bytes. ++ ++ ioReadBytes: ++ type: integer ++ minimum: 0 ++ title: Bytes read ++ description: | ++ Bytes the build read from block devices. ++ ++ ioWriteBytes: ++ type: integer ++ minimum: 0 ++ title: Bytes written ++ description: | ++ Bytes the build wrote to block devices. ++ ++ oomKills: ++ type: integer ++ minimum: 0 ++ title: Out-of-memory kill count ++ description: | ++ How many processes of the build the kernel killed for running out of memory. ++ + "$defs": + success: + type: object +diff --git a/src/libstore/build-result.cc b/src/libstore/build-result.cc +index c9d78e5..021971f 100644 +--- a/src/libstore/build-result.cc ++++ b/src/libstore/build-result.cc +@@ -127,6 +127,15 @@ static BuildResult::Failure::Status failureStatusFromString(std::string_view str + throw Error("unknown built result failure status '%s'", str); + } + ++using ResourceUsageField = std::optional BuildResult::*; ++ ++static constexpr std::array, 4> resourceUsageFields{{ ++ {"memoryPeak", &BuildResult::memoryPeak}, ++ {"ioReadBytes", &BuildResult::ioReadBytes}, ++ {"ioWriteBytes", &BuildResult::ioWriteBytes}, ++ {"oomKills", &BuildResult::oomKills}, ++}}; ++ + bool BuildError::operator==(const BuildError & other) const noexcept + { + return status == other.status && isNonDeterministic == other.isNonDeterministic && message() == other.message(); +@@ -162,6 +171,9 @@ void adl_serializer::to_json(json & res, const BuildResult & br) + if (br.cpuSystem.has_value()) { + res["cpuSystem"] = br.cpuSystem->count(); + } ++ for (auto [name, field] : resourceUsageFields) ++ if (auto & value = br.*field) ++ res[std::string{name}] = *value; + + // Handle success or failure variant + std::visit( +@@ -198,6 +210,9 @@ BuildResult adl_serializer::from_json(const json & _json) + if (auto cpuSystem = optionalValueAt(json, "cpuSystem")) { + br.cpuSystem = std::chrono::microseconds(getUnsigned(*cpuSystem)); + } ++ for (auto [name, field] : resourceUsageFields) ++ if (auto value = optionalValueAt(json, name)) ++ br.*field = getUnsigned(*value); + + // Determine success or failure based on success field + bool success = getBoolean(valueAt(json, "success")); +diff --git a/src/libstore/include/nix/store/build-result.hh b/src/libstore/include/nix/store/build-result.hh +index 808fcc8..5be5749 100644 +--- a/src/libstore/include/nix/store/build-result.hh ++++ b/src/libstore/include/nix/store/build-result.hh +@@ -193,6 +193,13 @@ struct BuildResult + */ + std::optional cpuUser, cpuSystem; + ++ /** ++ * Peak memory usage and bytes read from and written to block ++ * devices by the build, and how many of its processes were killed ++ * for running out of memory. ++ */ ++ std::optional memoryPeak, ioReadBytes, ioWriteBytes, oomKills; ++ + bool operator==(const BuildResult &) const noexcept; + std::strong_ordering operator<=>(const BuildResult &) const noexcept; + }; +diff --git a/src/libstore/include/nix/store/worker-protocol.hh b/src/libstore/include/nix/store/worker-protocol.hh +index d09d304..55ae493 100644 +--- a/src/libstore/include/nix/store/worker-protocol.hh ++++ b/src/libstore/include/nix/store/worker-protocol.hh +@@ -135,6 +135,12 @@ struct WorkerProto + */ + static constexpr std::string_view featureDisableSetOptions = "disable-set-options"; + ++ /** ++ * Feature for transmitting the peak memory, block I/O and ++ * out-of-memory kill count of a build in `BuildResult`. ++ */ ++ static constexpr std::string_view featureBuildResourceUsage = "build-resource-usage"; ++ + /** + * A unidirectional read connection, to be used by the read half of the + * canonical serializers below. +@@ -349,6 +355,8 @@ DECLARE_WORKER_SERIALISER(std::optional); + template<> + DECLARE_WORKER_SERIALISER(std::optional); + template<> ++DECLARE_WORKER_SERIALISER(std::optional); ++template<> + DECLARE_WORKER_SERIALISER(WorkerProto::ClientHandshakeInfo); + + template<> +diff --git a/src/libstore/linux/build/linux-derivation-builder.cc b/src/libstore/linux/build/linux-derivation-builder.cc +index d8376da..2262779 100644 +--- a/src/libstore/linux/build/linux-derivation-builder.cc ++++ b/src/libstore/linux/build/linux-derivation-builder.cc +@@ -949,6 +949,10 @@ void ChrootLinuxDerivationBuilder::killSandbox(bool getStats) + if (getStats) { + buildResult.cpuUser = stats.cpuUser; + buildResult.cpuSystem = stats.cpuSystem; ++ buildResult.memoryPeak = stats.memoryPeak; ++ buildResult.ioReadBytes = stats.ioReadBytes; ++ buildResult.ioWriteBytes = stats.ioWriteBytes; ++ buildResult.oomKills = stats.oomKills; + } + return; + } +diff --git a/src/libstore/worker-protocol.cc b/src/libstore/worker-protocol.cc +index 37aa4e3..c82c37c 100644 +--- a/src/libstore/worker-protocol.cc ++++ b/src/libstore/worker-protocol.cc +@@ -29,6 +29,7 @@ const WorkerProto::Version WorkerProto::latest = { + WorkerProto::featureRealisationWithPath, + }, + std::string{WorkerProto::featureDeleteDeadSpecificReferrers}, ++ std::string{WorkerProto::featureBuildResourceUsage}, + }, + }; + +@@ -191,6 +192,30 @@ void WorkerProto::Serialise>::write( + } + } + ++std::optional ++WorkerProto::Serialise>::read(const StoreDirConfig & store, WorkerProto::ReadConn conn) ++{ ++ auto tag = readNum(conn.from); ++ switch (tag) { ++ case 0: ++ return std::nullopt; ++ case 1: ++ return readNum(conn.from); ++ default: ++ throw Error("Invalid optional tag from remote"); ++ } ++} ++ ++void WorkerProto::Serialise>::write( ++ const StoreDirConfig & store, WorkerProto::WriteConn conn, const std::optional & optValue) ++{ ++ if (!optValue.has_value()) { ++ conn.to << uint8_t{0}; ++ } else { ++ conn.to << uint8_t{1} << *optValue; ++ } ++} ++ + DerivedPath WorkerProto::Serialise::read(const StoreDirConfig & store, WorkerProto::ReadConn conn) + { + auto s = readString(conn.from); +@@ -264,6 +289,12 @@ BuildResult WorkerProto::Serialise::read(const StoreDirConfig & sto + res.cpuUser = WorkerProto::Serialise>::read(store, conn); + res.cpuSystem = WorkerProto::Serialise>::read(store, conn); + } ++ if (conn.version.features.contains(WorkerProto::featureBuildResourceUsage)) { ++ res.memoryPeak = WorkerProto::Serialise>::read(store, conn); ++ res.ioReadBytes = WorkerProto::Serialise>::read(store, conn); ++ res.ioWriteBytes = WorkerProto::Serialise>::read(store, conn); ++ res.oomKills = WorkerProto::Serialise>::read(store, conn); ++ } + + if (conn.version.features.contains(WorkerProto::featureRealisationWithPath)) { + success.builtOutputs = WorkerProto::Serialise>::read(store, conn); +@@ -315,6 +346,12 @@ void WorkerProto::Serialise::write( + WorkerProto::write(store, conn, res.cpuUser); + WorkerProto::write(store, conn, res.cpuSystem); + } ++ if (conn.version.features.contains(WorkerProto::featureBuildResourceUsage)) { ++ WorkerProto::write(store, conn, res.memoryPeak); ++ WorkerProto::write(store, conn, res.ioReadBytes); ++ WorkerProto::write(store, conn, res.ioWriteBytes); ++ WorkerProto::write(store, conn, res.oomKills); ++ } + + if (conn.version.features.contains(WorkerProto::featureRealisationWithPath)) { + WorkerProto::write(store, conn, builtOutputs); +diff --git a/src/libutil/linux/cgroup.cc b/src/libutil/linux/cgroup.cc +index 9ebc3ab..5a2fac9 100644 +--- a/src/libutil/linux/cgroup.cc ++++ b/src/libutil/linux/cgroup.cc +@@ -49,30 +49,57 @@ StringMap getCgroups(const std::filesystem::path & cgroupFile) + return cgroups; + } + ++static std::optional readCgroupFile(const std::filesystem::path & path) ++{ ++ if (!pathExists(path)) ++ return std::nullopt; ++ return readFile(path); ++} ++ ++/* For "flat keyed" and "nested keyed" files, see "Format" in the ++ kernel's cgroup-v2 documentation. */ ++static std::optional getFlatKeyedValue(std::string_view contents, std::string_view key) ++{ ++ for (auto & line : tokenizeString>(contents, "\n")) { ++ auto fields = tokenizeString>(line); ++ if (fields.size() == 2 && fields[0] == key) ++ return string2Int(fields[1]); ++ } ++ return std::nullopt; ++} ++ ++static uint64_t sumNestedKeyedValues(std::string_view contents, std::string_view key) ++{ ++ auto prefix = std::string(key) + "="; ++ uint64_t sum = 0; ++ for (auto & field : tokenizeString>(contents)) ++ if (hasPrefix(field, prefix)) ++ if (auto n = string2Int(field.substr(prefix.size()))) ++ sum += *n; ++ return sum; ++} ++ + CgroupStats getCgroupStats(const std::filesystem::path & cgroup) + { + CgroupStats stats; + +- auto cpustatPath = cgroup / "cpu.stat"; ++ if (auto cpuStat = readCgroupFile(cgroup / "cpu.stat")) { ++ auto toMicroseconds = [](uint64_t n) { return std::chrono::microseconds(n); }; ++ stats.cpuUser = getFlatKeyedValue(*cpuStat, "user_usec").transform(toMicroseconds); ++ stats.cpuSystem = getFlatKeyedValue(*cpuStat, "system_usec").transform(toMicroseconds); ++ } + +- if (pathExists(cpustatPath)) { +- for (auto & line : tokenizeString>(readFile(cpustatPath), "\n")) { +- std::string_view userPrefix = "user_usec "; +- if (hasPrefix(line, userPrefix)) { +- auto n = string2Int(line.substr(userPrefix.size())); +- if (n) +- stats.cpuUser = std::chrono::microseconds(*n); +- } ++ if (auto memoryPeak = readCgroupFile(cgroup / "memory.peak")) ++ stats.memoryPeak = string2Int(trim(*memoryPeak)); + +- std::string_view systemPrefix = "system_usec "; +- if (hasPrefix(line, systemPrefix)) { +- auto n = string2Int(line.substr(systemPrefix.size())); +- if (n) +- stats.cpuSystem = std::chrono::microseconds(*n); +- } +- } ++ if (auto ioStat = readCgroupFile(cgroup / "io.stat")) { ++ stats.ioReadBytes = sumNestedKeyedValues(*ioStat, "rbytes"); ++ stats.ioWriteBytes = sumNestedKeyedValues(*ioStat, "wbytes"); + } + ++ if (auto memoryEvents = readCgroupFile(cgroup / "memory.events")) ++ stats.oomKills = getFlatKeyedValue(*memoryEvents, "oom_kill"); ++ + return stats; + } + +diff --git a/src/libutil/linux/include/nix/util/cgroup.hh b/src/libutil/linux/include/nix/util/cgroup.hh +index 7b50ffe..ced9c77 100644 +--- a/src/libutil/linux/include/nix/util/cgroup.hh ++++ b/src/libutil/linux/include/nix/util/cgroup.hh +@@ -17,6 +17,7 @@ StringMap getCgroups(const std::filesystem::path & cgroupFile); + struct CgroupStats + { + std::optional cpuUser, cpuSystem; ++ std::optional memoryPeak, ioReadBytes, ioWriteBytes, oomKills; + }; + + /** +-- +2.55.0 +