Compare commits

..

No commits in common. "main" and "release-46a4d4e84222" have entirely different histories.

53 changed files with 364 additions and 7305 deletions

1
.gitignore vendored
View file

@ -1,5 +1,4 @@
/target/
/web/target/
/.clusterflux/
**/.clusterflux/
/vscode-extension/node_modules/

24
Cargo.lock generated
View file

@ -290,20 +290,6 @@ dependencies = [
"wasmparser 0.245.1",
]
[[package]]
name = "clusterflux-client"
version = "0.1.0"
dependencies = [
"base64",
"clusterflux-control",
"clusterflux-coordinator",
"clusterflux-core",
"serde",
"serde_json",
"thiserror 1.0.69",
"tokio",
]
[[package]]
name = "clusterflux-control"
version = "0.1.0"
@ -1726,6 +1712,16 @@ dependencies = [
"windows-sys 0.52.0",
]
[[package]]
name = "runtime-conformance"
version = "0.1.0"
dependencies = [
"clusterflux-sdk",
"futures-executor",
"serde",
"serde_json",
]
[[package]]
name = "rustc-hash"
version = "2.1.3"

View file

@ -1,9 +1,7 @@
[workspace]
resolver = "2"
exclude = ["web"]
members = [
"crates/clusterflux-cli",
"crates/clusterflux-client",
"crates/clusterflux-control",
"crates/clusterflux-coordinator",
"crates/clusterflux-core",

View file

@ -97,9 +97,6 @@ cargo install --path crates/clusterflux-coordinator --bin clusterflux-coordinato
cargo install --path crates/clusterflux-dap --bin clusterflux-debug-dap
~~~
On NixOS or another system with Nix, the equivalent package is available with
`nix profile install .#clusterflux-tools`.
Rootless Podman is required on Linux nodes that build or run a declared
Containerfile environment. Install VS Code when you want the graphical debug
workflow.

View file

@ -12,7 +12,6 @@ use crate::client::{
};
use crate::config::StoredCliSession;
use crate::errors::cli_error_summary_for_category;
use crate::process::hydrate_process_scope;
use crate::process_events::{
artifact_download_grant_disclosures, artifact_download_session_summary,
artifact_export_plan_summary, artifact_response_machine_error, artifact_summaries,
@ -28,10 +27,9 @@ pub(crate) fn artifact_list_report(args: ArtifactListArgs) -> Result<Value> {
}
pub(crate) fn artifact_list_report_with_session(
mut args: ArtifactListArgs,
args: ArtifactListArgs,
stored_session: Option<&StoredCliSession>,
) -> Result<Value> {
hydrate_process_scope(&mut args.scope, stored_session);
let events = list_task_events_if_available_with_session(
args.scope.coordinator.as_deref(),
&args.scope,
@ -55,10 +53,9 @@ pub(crate) fn artifact_download_report(args: ArtifactDownloadArgs) -> Result<Val
}
pub(crate) fn artifact_download_report_with_session(
mut args: ArtifactDownloadArgs,
args: ArtifactDownloadArgs,
stored_session: Option<&StoredCliSession>,
) -> Result<Value> {
hydrate_process_scope(&mut args.scope, stored_session);
if let Some(coordinator) = &args.scope.coordinator {
let mut session = JsonLineSession::connect(coordinator)?;
let response = session.request(authenticated_or_local_trusted_request(
@ -146,10 +143,9 @@ pub(crate) fn artifact_export_report(args: ArtifactExportArgs) -> Result<Value>
}
pub(crate) fn artifact_export_report_with_session(
mut args: ArtifactExportArgs,
args: ArtifactExportArgs,
stored_session: Option<&StoredCliSession>,
) -> Result<Value> {
hydrate_process_scope(&mut args.scope, stored_session);
if let Some(coordinator) = &args.scope.coordinator {
let mut session = JsonLineSession::connect(coordinator)?;
let response = session.request(authenticated_or_local_trusted_request(

View file

@ -3,7 +3,6 @@ use serde_json::{json, Value};
use crate::client::list_task_events_if_available_with_session;
use crate::config::StoredCliSession;
use crate::process::hydrate_process_scope;
use crate::process_events::log_entries;
use crate::LogsArgs;
@ -13,10 +12,9 @@ pub(crate) fn logs_report(args: LogsArgs) -> Result<Value> {
}
pub(crate) fn logs_report_with_session(
mut args: LogsArgs,
args: LogsArgs,
stored_session: Option<&StoredCliSession>,
) -> Result<Value> {
hydrate_process_scope(&mut args.scope, stored_session);
let events = list_task_events_if_available_with_session(
args.scope.coordinator.as_deref(),
&args.scope,

View file

@ -15,10 +15,7 @@ use crate::{
ProcessListArgs, ProcessRestartArgs, ProcessStatusArgs,
};
pub(crate) fn hydrate_process_scope(
scope: &mut CliScopeArgs,
stored_session: Option<&StoredCliSession>,
) {
fn hydrate_process_scope(scope: &mut CliScopeArgs, stored_session: Option<&StoredCliSession>) {
if scope.coordinator.is_none() {
scope.coordinator = stored_session
.filter(|session| session.session_secret.is_some())

View file

@ -3004,14 +3004,6 @@ fn user_control_commands_use_authenticated_envelope_with_stored_cli_session() {
"restart_task",
br#"{"type":"task_restart","process":"vp","task":"task-a","actor":"user-session","accepted":false,"clean_boundary_available":false,"active_task":false,"completed_event_observed":false,"requires_whole_process_restart":true,"message":"restart requires checkpoint","charged_debug_read_bytes":1024,"used_debug_read_bytes":1024,"audit_event":{"tenant":"tenant-session","project":"project-session","process":"vp","task":"task-a","actor":"user-session","operation":"restart_task","allowed":true,"reason":"restart requires checkpoint","charged_debug_read_bytes":1024,"used_debug_read_bytes":1024}}"#.as_slice(),
),
(
"list_task_events",
br#"{"type":"task_events","events":[{"process":"vp","task":"task-a","terminal_state":"completed","stdout_tail":"compiled\n","stderr_tail":""}]}"#.as_slice(),
),
(
"list_task_events",
br#"{"type":"task_events","events":[{"process":"vp","task":"task-a","terminal_state":"completed","artifact_path":"/vfs/artifacts/app.txt","artifact_digest":"sha256:app","artifact_size_bytes":3}]}"#.as_slice(),
),
(
"create_artifact_download_link",
br#"{"type":"artifact_download_link","link":{"artifact":"app.txt","source":{"RetainedNode":"node-a"},"url_path":"/artifacts/tenant-session/project-session/vp/app.txt","scoped_token_digest":"sha256:token","expires_at_epoch_seconds":60,"tenant":"tenant-session","project":"project-session","process":"vp","actor":{"User":"user-session"},"max_bytes":2048,"policy_context_digest":"sha256:policy"}}"#.as_slice(),
@ -3116,26 +3108,9 @@ fn user_control_commands_use_authenticated_envelope_with_stored_cli_session() {
Some(&session),
)
.unwrap();
let logs = logs_report_with_session(
LogsArgs {
scope: scope.clone(),
process: Some("vp".to_owned()),
task: Some("task-a".to_owned()),
},
Some(&session),
)
.unwrap();
let artifacts = artifact_list_report_with_session(
ArtifactListArgs {
scope: scope.clone(),
process: Some("vp".to_owned()),
},
Some(&session),
)
.unwrap();
let download = artifact_download_report_with_session(
artifact_download_report_with_session(
ArtifactDownloadArgs {
scope,
scope: coordinator_scope.clone(),
artifact: "app.txt".to_owned(),
to: None,
max_bytes: 2048,
@ -3143,9 +3118,6 @@ fn user_control_commands_use_authenticated_envelope_with_stored_cli_session() {
Some(&session),
)
.unwrap();
assert_eq!(logs["log_entries"][0]["stdout_tail"], "compiled\n");
assert_eq!(artifacts["artifacts"][0]["artifact"], "app.txt");
assert_eq!(download["coordinator"], session.coordinator);
debug_attach_report_with_dap_and_session(
DebugAttachArgs {
scope: coordinator_scope,

View file

@ -1,18 +0,0 @@
[package]
name = "clusterflux-client"
version = "0.1.0"
edition.workspace = true
license.workspace = true
repository.workspace = true
[dependencies]
base64.workspace = true
clusterflux-control = { path = "../clusterflux-control" }
clusterflux-core = { path = "../clusterflux-core" }
serde.workspace = true
serde_json.workspace = true
thiserror.workspace = true
tokio = { workspace = true, features = ["sync", "time"] }
[dev-dependencies]
clusterflux-coordinator = { path = "../clusterflux-coordinator" }

View file

@ -1,68 +0,0 @@
# clusterflux-client
`clusterflux-client` is the public, typed Rust boundary for a Clusterflux web
backend. It talks to the same versioned control and login endpoints as the CLI.
Callers do not construct JSON envelopes or place session secrets into request
objects.
The crate intentionally contains no coordinator implementation, persistence
types, scheduler policy, or website-specific business rules. A web backend only
needs this crate to authenticate and work with account status, projects, Agent
keys, node enrollment and liveness, current/recent processes, task attempts,
recent logs, artifacts, quota state, process control, task recovery, and Debug
Epoch state. It does not expose workflow launch or whole-process replay.
```rust,no_run
use clusterflux_client::{ClusterfluxClient, SessionCredential};
# async fn example() -> Result<(), clusterflux_client::ClientError> {
let credential = SessionCredential::from_secret(
std::env::var("CLUSTERFLUX_SESSION_SECRET").expect("server-side session secret"),
);
let client = ClusterfluxClient::connect("https://clusterflux.lesstuff.com")?
.with_session_credential(&credential);
let account = client.account_status().await?;
let nodes = client.list_nodes().await?;
let processes = client.list_processes(None, 20).await?;
# let _ = (account, nodes, processes);
# Ok(())
# }
```
## Login boundary
`begin_browser_login` starts the hosted Authentik flow. The identity callback
returns only to the fixed configured website URL with a short-lived handoff.
The website backend calls `exchange_browser_login_handoff` once and stores the
returned `SessionCredential` only in encrypted server-side session storage.
The credential has redacted `Debug` output and deliberately does not implement
Serde serialization. Logout calls the existing session-revocation operation.
The browser must not receive or persist the Clusterflux session credential,
provider tokens, authorization code, PKCE verifier, or CLI polling secret.
## Errors, bounds, and transport
API failures are returned as `ClientError::Api(ApiError)`. Decisions should use
the stable `code`, `category`, `retryable`, and `request_id` fields, while
`message` is for people. The client rejects an error whose request ID does not
match the originating request.
Paginated list methods require an explicit bounded page size. Process pages are
limited to 100 entries; node, artifact, and recent-log pages to 200.
`list_nodes()` is a bounded first-page convenience; `list_nodes_page()` exposes
its cursor. Recent logs are ephemeral and can report truncation or sequence
loss. Artifact bytes use `ArtifactDownload::next_chunk`, which validates
offsets and decodes bounded chunks without materializing the complete artifact.
`ControlTransport` has bounded connect and I/O timeouts and performs blocking
network work outside the async executor. Dropping a request future cancels the
callers wait; any already-started blocking I/O remains bounded by its timeout.
`MockTransport` supplies deterministic responses and records exact envelopes
for application tests.
`CLIENT_API_VERSION` is the one supported control protocol version. The
checked-in `web_operations.json` contract fixture records every website
operation and its stable error shape, and the crates contract suite checks the
fixture.

View file

@ -1,924 +0,0 @@
mod protocol;
mod transport;
mod types;
use std::sync::atomic::{AtomicU64, Ordering};
use std::sync::{Arc, Mutex};
use base64::{engine::general_purpose::STANDARD as BASE64_STANDARD, Engine as _};
use clusterflux_control::{CONTROL_API_PATH, LOGIN_API_PATH};
use clusterflux_core::{coordinator_wire_request, ApiError, COORDINATOR_PROTOCOL_VERSION};
use protocol::{AuthenticatedRequest, LoginRequest, WireResponse};
use serde_json::json;
use thiserror::Error;
use transport::{ClientTransport, TransportRequest};
pub use clusterflux_core::{
AgentId, ApiErrorCategory, ApiErrorCode, ArtifactId, Authorization, Capability, CredentialKind,
Digest, DownloadLink, EnvironmentBackend, LimitKind, NodeCapabilities, NodeId, Os, ProcessId,
ProjectId, ResourceLimits, TaskDefinitionId, TaskFailurePolicy, TaskInstanceId, TenantId,
UserId, VfsPath,
};
pub use transport::{
ClientTransportError, ControlTransport, MockTransport, TransportFuture, TransportResponse,
};
pub use types::*;
pub const CLIENT_API_VERSION: u64 = COORDINATOR_PROTOCOL_VERSION;
#[derive(Debug, Error)]
pub enum ClientError {
#[error("Clusterflux API error: {0}")]
Api(ApiError),
#[error(transparent)]
Transport(#[from] ClientTransportError),
#[error("Clusterflux client protocol error: {0}")]
Protocol(String),
}
#[derive(Clone)]
pub struct ClusterfluxClient {
transport: Arc<dyn ClientTransport>,
session_secret: Arc<Mutex<Option<String>>>,
next_request: Arc<AtomicU64>,
}
impl ClusterfluxClient {
pub fn connect(endpoint: impl Into<String>) -> Result<Self, ClientError> {
Ok(Self::with_transport(ControlTransport::new(endpoint)?))
}
pub fn with_transport(transport: impl ClientTransport) -> Self {
Self {
transport: Arc::new(transport),
session_secret: Arc::new(Mutex::new(None)),
next_request: Arc::new(AtomicU64::new(1)),
}
}
pub fn with_session_credential(mut self, credential: &SessionCredential) -> Self {
self.session_secret = Arc::new(Mutex::new(Some(credential.0.clone())));
self
}
pub fn is_session_configured(&self) -> bool {
self.session_secret
.lock()
.map(|secret| secret.is_some())
.unwrap_or(false)
}
pub async fn begin_browser_login(&self) -> Result<BrowserLoginStart, ClientError> {
match self
.send_login(LoginRequest::BeginWebBrowserLogin {})
.await?
{
WireResponse::WebBrowserLoginStarted {
transaction_id,
authorization_url,
expires_at_epoch_seconds,
} => Ok(BrowserLoginStart {
transaction_id,
authorization_url,
expires_at_epoch_seconds,
}),
_ => Err(unexpected_response("web_browser_login_started")),
}
}
pub async fn exchange_browser_login_handoff(
&self,
transaction_id: impl Into<String>,
handoff_code: impl Into<String>,
) -> Result<BrowserSession, ClientError> {
match self
.send_login(LoginRequest::ExchangeWebLoginHandoff {
transaction_id: transaction_id.into(),
handoff_code: handoff_code.into(),
})
.await?
{
WireResponse::WebBrowserSession { session } => Ok(BrowserSession {
tenant: session.tenant,
project: session.project,
user: session.user,
credential: SessionCredential(session.session_secret),
expires_at_epoch_seconds: session.expires_at_epoch_seconds,
}),
_ => Err(unexpected_response("web_browser_session")),
}
}
pub async fn cancel_browser_login(
&self,
transaction_id: impl Into<String>,
) -> Result<(), ClientError> {
match self
.send_login(LoginRequest::CancelWebBrowserLogin {
transaction_id: transaction_id.into(),
})
.await?
{
WireResponse::WebBrowserLoginCancelled {} => Ok(()),
_ => Err(unexpected_response("web_browser_login_cancelled")),
}
}
pub async fn account_status(&self) -> Result<AccountStatus, ClientError> {
match self
.send_authenticated(AuthenticatedRequest::AuthStatus)
.await?
{
WireResponse::AuthStatus {
tenant,
project,
actor,
authenticated,
account_status,
suspended,
disabled,
deleted,
manual_review,
sanitized_reason,
next_actions,
} => Ok(AccountStatus {
tenant,
project,
actor,
authenticated,
account_status,
suspended,
disabled,
deleted,
manual_review,
sanitized_reason,
next_actions,
}),
_ => Err(unexpected_response("auth_status")),
}
}
pub async fn logout(&self) -> Result<(), ClientError> {
match self
.send_authenticated(AuthenticatedRequest::RevokeCliSession)
.await?
{
WireResponse::CliSessionRevoked {} => {
*self.session_secret.lock().map_err(|_| {
ClientError::Protocol("session credential lock was poisoned".to_owned())
})? = None;
Ok(())
}
_ => Err(unexpected_response("cli_session_revoked")),
}
}
pub async fn list_projects(&self) -> Result<Vec<Project>, ClientError> {
match self
.send_authenticated(AuthenticatedRequest::ListProjects)
.await?
{
WireResponse::Projects { projects } => Ok(projects),
_ => Err(unexpected_response("projects")),
}
}
pub async fn create_project(
&self,
project: ProjectId,
name: impl Into<String>,
) -> Result<Project, ClientError> {
match self
.send_authenticated(AuthenticatedRequest::CreateProject {
project,
name: name.into(),
})
.await?
{
WireResponse::ProjectCreated { project } => Ok(project),
_ => Err(unexpected_response("project_created")),
}
}
pub async fn select_project(&self, project: ProjectId) -> Result<Project, ClientError> {
match self
.send_authenticated(AuthenticatedRequest::SelectProject { project })
.await?
{
WireResponse::ProjectSelected { project } => Ok(project),
_ => Err(unexpected_response("project_selected")),
}
}
pub async fn register_agent_public_key(
&self,
agent: AgentId,
public_key: impl Into<String>,
) -> Result<AgentPublicKey, ClientError> {
self.agent_key_mutation(AuthenticatedRequest::RegisterAgentPublicKey {
agent,
public_key: public_key.into(),
})
.await
}
pub async fn list_agent_public_keys(&self) -> Result<Vec<AgentPublicKey>, ClientError> {
match self
.send_authenticated(AuthenticatedRequest::ListAgentPublicKeys)
.await?
{
WireResponse::AgentPublicKeys { records } => Ok(records),
_ => Err(unexpected_response("agent_public_keys")),
}
}
pub async fn rotate_agent_public_key(
&self,
agent: AgentId,
public_key: impl Into<String>,
) -> Result<AgentPublicKey, ClientError> {
self.agent_key_mutation(AuthenticatedRequest::RotateAgentPublicKey {
agent,
public_key: public_key.into(),
})
.await
}
pub async fn revoke_agent_public_key(
&self,
agent: AgentId,
) -> Result<AgentPublicKey, ClientError> {
self.agent_key_mutation(AuthenticatedRequest::RevokeAgentPublicKey { agent })
.await
}
async fn agent_key_mutation(
&self,
request: AuthenticatedRequest,
) -> Result<AgentPublicKey, ClientError> {
match self.send_authenticated(request).await? {
WireResponse::AgentPublicKey { record } => Ok(record),
_ => Err(unexpected_response("agent_public_key")),
}
}
pub async fn create_node_enrollment_grant(
&self,
ttl_seconds: u64,
) -> Result<NodeEnrollmentGrant, ClientError> {
match self
.send_authenticated(AuthenticatedRequest::CreateNodeEnrollmentGrant { ttl_seconds })
.await?
{
WireResponse::NodeEnrollmentGrantCreated {
tenant,
project,
grant,
scope,
expires_at_epoch_seconds,
} => Ok(NodeEnrollmentGrant {
tenant,
project,
grant,
scope,
expires_at_epoch_seconds,
}),
_ => Err(unexpected_response("node_enrollment_grant_created")),
}
}
pub async fn list_nodes(&self) -> Result<Vec<NodeSummary>, ClientError> {
Ok(self.list_nodes_page(None, 200).await?.nodes)
}
pub async fn list_nodes_page(
&self,
cursor: Option<String>,
limit: u32,
) -> Result<NodePage, ClientError> {
match self
.send_authenticated(AuthenticatedRequest::ListNodeSummaries { cursor, limit })
.await?
{
WireResponse::NodeSummaries { nodes, next_cursor } => {
Ok(NodePage { nodes, next_cursor })
}
_ => Err(unexpected_response("node_summaries")),
}
}
pub async fn revoke_node(&self, node: NodeId) -> Result<NodeRevocation, ClientError> {
match self
.send_authenticated(AuthenticatedRequest::RevokeNodeCredential { node })
.await?
{
WireResponse::NodeCredentialRevoked {
node,
tenant,
project,
actor,
descriptor_removed,
queued_assignments_removed,
} => Ok(NodeRevocation {
node,
tenant,
project,
actor,
descriptor_removed,
queued_assignments_removed,
}),
_ => Err(unexpected_response("node_credential_revoked")),
}
}
pub async fn list_processes(
&self,
cursor: Option<String>,
limit: u32,
) -> Result<ProcessPage, ClientError> {
match self
.send_authenticated(AuthenticatedRequest::ListProcessSummaries { cursor, limit })
.await?
{
WireResponse::ProcessSummaries {
processes,
next_cursor,
} => Ok(ProcessPage {
processes,
next_cursor,
}),
_ => Err(unexpected_response("process_summaries")),
}
}
pub async fn cancel_process(
&self,
process: ProcessId,
) -> Result<ProcessCancellation, ClientError> {
match self
.send_authenticated(AuthenticatedRequest::CancelProcess { process })
.await?
{
WireResponse::ProcessCancellationRequested {
process,
cancelled_tasks,
affected_nodes,
} => Ok(ProcessCancellation {
process,
affected_tasks: cancelled_tasks,
affected_nodes,
aborted: false,
}),
_ => Err(unexpected_response("process_cancellation_requested")),
}
}
pub async fn abort_process(
&self,
process: ProcessId,
) -> Result<ProcessCancellation, ClientError> {
match self
.send_authenticated(AuthenticatedRequest::AbortProcess {
process,
launch_attempt: None,
})
.await?
{
WireResponse::ProcessAborted {
process,
aborted_tasks,
affected_nodes,
} => Ok(ProcessCancellation {
process,
affected_tasks: aborted_tasks,
affected_nodes,
aborted: true,
}),
_ => Err(unexpected_response("process_aborted")),
}
}
pub async fn quota_status(&self) -> Result<QuotaStatus, ClientError> {
match self
.send_authenticated(AuthenticatedRequest::QuotaStatus)
.await?
{
WireResponse::QuotaStatus {
tenant,
project,
actor,
policy_label,
limits,
window_seconds,
usage,
window_started_epoch_seconds,
} => Ok(QuotaStatus {
tenant,
project,
actor,
policy_label,
limits,
window_seconds,
usage,
window_started_epoch_seconds,
}),
_ => Err(unexpected_response("quota_status")),
}
}
pub async fn list_task_events(
&self,
process: Option<ProcessId>,
) -> Result<Vec<TaskCompletionEvent>, ClientError> {
match self
.send_authenticated(AuthenticatedRequest::ListTaskEvents { process })
.await?
{
WireResponse::TaskEvents { events } => Ok(events),
_ => Err(unexpected_response("task_events")),
}
}
pub async fn list_task_snapshots(
&self,
process: ProcessId,
) -> Result<Vec<TaskAttemptSnapshot>, ClientError> {
match self
.send_authenticated(AuthenticatedRequest::ListTaskSnapshots { process })
.await?
{
WireResponse::TaskSnapshots { snapshots } => Ok(snapshots),
_ => Err(unexpected_response("task_snapshots")),
}
}
pub async fn list_recent_logs(
&self,
process: ProcessId,
task: Option<TaskInstanceId>,
after_sequence: Option<u64>,
limit: u32,
) -> Result<RecentLogPage, ClientError> {
match self
.send_authenticated(AuthenticatedRequest::ListRecentLogs {
process,
task,
after_sequence,
limit,
})
.await?
{
WireResponse::RecentLogs {
entries,
next_sequence,
history_truncated,
} => Ok(RecentLogPage {
entries,
next_sequence,
history_truncated,
}),
_ => Err(unexpected_response("recent_logs")),
}
}
pub async fn restart_task(
&self,
process: ProcessId,
task: TaskInstanceId,
replacement_bundle: Option<TaskReplacementBundle>,
) -> Result<TaskRestart, ClientError> {
match self
.send_authenticated(AuthenticatedRequest::RestartTask {
process,
task,
replacement_bundle,
})
.await?
{
WireResponse::TaskRestart {
process,
task,
restarted_task_instance,
restarted_attempt_id,
actor,
accepted,
clean_boundary_available,
active_task,
completed_event_observed,
requires_whole_process_restart,
message,
audit_event,
} => Ok(TaskRestart {
process,
task,
restarted_task_instance,
restarted_attempt_id,
actor,
accepted,
clean_boundary_available,
active_task,
completed_event_observed,
requires_whole_process_restart,
message,
audit_event,
}),
_ => Err(unexpected_response("task_restart")),
}
}
pub async fn resolve_task_failure(
&self,
process: ProcessId,
task: TaskInstanceId,
resolution: TaskFailureResolution,
) -> Result<TaskFailureResolutionResult, ClientError> {
match self
.send_authenticated(AuthenticatedRequest::ResolveTaskFailure {
process,
task,
resolution,
})
.await?
{
WireResponse::TaskFailureResolved {
process,
task,
attempt_id,
resolution,
} => Ok(TaskFailureResolutionResult {
process,
task,
attempt_id,
resolution,
}),
_ => Err(unexpected_response("task_failure_resolved")),
}
}
pub async fn debug_attach(&self, process: ProcessId) -> Result<DebugAttach, ClientError> {
match self
.send_authenticated(AuthenticatedRequest::DebugAttach { process })
.await?
{
WireResponse::DebugAttach {
process,
actor,
authorization,
audit_event,
} => Ok(DebugAttach {
process,
actor,
authorization,
audit_event,
}),
_ => Err(unexpected_response("debug_attach")),
}
}
pub async fn create_debug_epoch(
&self,
process: ProcessId,
stopped_task: TaskInstanceId,
reason: impl Into<String>,
) -> Result<DebugEpochControl, ClientError> {
match self
.send_authenticated(AuthenticatedRequest::CreateDebugEpoch {
process,
stopped_task,
reason: reason.into(),
})
.await?
{
WireResponse::DebugEpoch {
process,
actor,
epoch,
command,
affected_tasks,
all_stop_requested,
audit_event,
} => Ok(DebugEpochControl {
process,
actor,
epoch,
command,
affected_tasks,
all_stop_requested,
audit_event,
}),
_ => Err(unexpected_response("debug_epoch")),
}
}
pub async fn resume_debug_epoch(
&self,
process: ProcessId,
epoch: u64,
) -> Result<DebugEpochControl, ClientError> {
match self
.send_authenticated(AuthenticatedRequest::ResumeDebugEpoch { process, epoch })
.await?
{
WireResponse::DebugEpoch {
process,
actor,
epoch,
command,
affected_tasks,
all_stop_requested,
audit_event,
} => Ok(DebugEpochControl {
process,
actor,
epoch,
command,
affected_tasks,
all_stop_requested,
audit_event,
}),
_ => Err(unexpected_response("debug_epoch")),
}
}
pub async fn inspect_debug_epoch(
&self,
process: ProcessId,
epoch: u64,
) -> Result<DebugEpochStatus, ClientError> {
match self
.send_authenticated(AuthenticatedRequest::InspectDebugEpoch { process, epoch })
.await?
{
WireResponse::DebugEpochStatus {
process,
actor,
epoch,
command,
expected_tasks,
acknowledgements,
fully_frozen,
partially_frozen,
fully_resumed,
failed,
failure_messages,
audit_event,
} => Ok(DebugEpochStatus {
process,
actor,
epoch,
command,
expected_tasks,
acknowledgements,
fully_frozen,
partially_frozen,
fully_resumed,
failed,
failure_messages,
audit_event,
}),
_ => Err(unexpected_response("debug_epoch_status")),
}
}
pub async fn list_artifacts(
&self,
process: Option<ProcessId>,
cursor: Option<String>,
limit: u32,
) -> Result<ArtifactPage, ClientError> {
match self
.send_authenticated(AuthenticatedRequest::ListArtifacts {
process,
cursor,
limit,
})
.await?
{
WireResponse::Artifacts {
artifacts,
next_cursor,
} => Ok(ArtifactPage {
artifacts,
next_cursor,
}),
_ => Err(unexpected_response("artifacts")),
}
}
pub async fn get_artifact(&self, artifact: ArtifactId) -> Result<ArtifactSummary, ClientError> {
match self
.send_authenticated(AuthenticatedRequest::GetArtifact { artifact })
.await?
{
WireResponse::Artifact { artifact } => Ok(artifact),
_ => Err(unexpected_response("artifact")),
}
}
pub async fn begin_artifact_download(
&self,
artifact: ArtifactId,
max_bytes: u64,
ttl_seconds: u64,
chunk_bytes: u64,
) -> Result<ArtifactDownload, ClientError> {
match self
.send_authenticated(AuthenticatedRequest::CreateArtifactDownloadLink {
artifact: artifact.clone(),
max_bytes,
ttl_seconds,
})
.await?
{
WireResponse::ArtifactDownloadLink { link } => Ok(ArtifactDownload {
client: self.clone(),
artifact,
max_bytes,
chunk_bytes,
token_digest: link.scoped_token_digest.clone(),
link,
expected_offset: 0,
finished: false,
}),
_ => Err(unexpected_response("artifact_download_link")),
}
}
async fn send_authenticated(
&self,
request: AuthenticatedRequest,
) -> Result<WireResponse, ClientError> {
let session_secret = self
.session_secret
.lock()
.map_err(|_| ClientError::Protocol("session credential lock was poisoned".to_owned()))?
.clone()
.ok_or_else(|| {
ClientError::Api(ApiError::from_message(
"client",
"no authenticated Clusterflux session is configured",
))
})?;
self.send(
CONTROL_API_PATH,
json!({
"type": "authenticated",
"session_secret": session_secret,
"request": request,
}),
)
.await
}
async fn send_login(&self, request: LoginRequest) -> Result<WireResponse, ClientError> {
let payload = serde_json::to_value(request)
.map_err(|error| ClientError::Protocol(error.to_string()))?;
self.send(LOGIN_API_PATH, payload).await
}
async fn send(
&self,
api_path: &str,
payload: serde_json::Value,
) -> Result<WireResponse, ClientError> {
let request_number = self.next_request.fetch_add(1, Ordering::Relaxed);
let request_id = format!("client-{request_number}");
let envelope = coordinator_wire_request(&request_id, payload);
let body = serde_json::to_vec(&envelope)
.map_err(|error| ClientError::Protocol(error.to_string()))?;
let response = self
.transport
.send(TransportRequest {
api_path: api_path.to_owned(),
body,
})
.await?;
let response: WireResponse = serde_json::from_slice(&response.body).map_err(|error| {
ClientError::Protocol(format!("decode typed API response: {error}"))
})?;
match response {
WireResponse::Error {
code,
category,
message,
retryable,
request_id: response_request_id,
} => {
if response_request_id != request_id {
return Err(ClientError::Protocol(format!(
"error response request_id {response_request_id} does not match {request_id}"
)));
}
Err(ClientError::Api(ApiError::new(
code,
category,
message,
retryable,
response_request_id,
)))
}
response => Ok(response),
}
}
}
pub struct ArtifactDownload {
client: ClusterfluxClient,
artifact: ArtifactId,
max_bytes: u64,
chunk_bytes: u64,
token_digest: Digest,
link: clusterflux_core::DownloadLink,
expected_offset: u64,
finished: bool,
}
impl ArtifactDownload {
pub fn link(&self) -> &clusterflux_core::DownloadLink {
&self.link
}
pub async fn next_chunk(&mut self) -> Result<ArtifactDownloadPoll, ClientError> {
if self.finished {
return Ok(ArtifactDownloadPoll::Chunk {
offset: self.expected_offset,
bytes: Vec::new(),
eof: true,
});
}
match self
.client
.send_authenticated(AuthenticatedRequest::OpenArtifactDownloadStream {
artifact: self.artifact.clone(),
max_bytes: self.max_bytes,
token_digest: self.token_digest.clone(),
chunk_bytes: self.chunk_bytes,
})
.await?
{
WireResponse::ArtifactDownloadStream {
content_bytes_available,
content_offset,
content_eof,
content_base64,
..
} => {
if !content_bytes_available {
return Ok(ArtifactDownloadPoll::Pending);
}
let offset = content_offset.ok_or_else(|| {
ClientError::Protocol(
"artifact response contained bytes without an offset".to_owned(),
)
})?;
if offset != self.expected_offset {
return Err(ClientError::Protocol(format!(
"artifact chunk offset {offset} does not match expected {}",
self.expected_offset
)));
}
let bytes = BASE64_STANDARD
.decode(content_base64.unwrap_or_default())
.map_err(|error| {
ClientError::Protocol(format!(
"artifact response contains invalid base64: {error}"
))
})?;
self.expected_offset = self.expected_offset.saturating_add(bytes.len() as u64);
self.finished = content_eof;
Ok(ArtifactDownloadPoll::Chunk {
offset,
bytes,
eof: content_eof,
})
}
_ => Err(unexpected_response("artifact_download_stream")),
}
}
pub async fn cancel(mut self) -> Result<(), ClientError> {
if self.finished {
return Ok(());
}
match self
.client
.send_authenticated(AuthenticatedRequest::RevokeArtifactDownloadLink {
artifact: self.artifact.clone(),
token_digest: self.token_digest.clone(),
})
.await?
{
WireResponse::ArtifactDownloadLinkRevoked {} => {
self.finished = true;
Ok(())
}
_ => Err(unexpected_response("artifact_download_link_revoked")),
}
}
}
fn unexpected_response(expected: &str) -> ClientError {
ClientError::Protocol(format!(
"expected typed response {expected}, received another response variant"
))
}

View file

@ -1,403 +0,0 @@
use std::collections::BTreeMap;
use clusterflux_core::{
AgentId, ApiErrorCategory, ApiErrorCode, ArtifactId, Authorization, Digest, DownloadLink,
LimitKind, NodeId, ProcessId, ProjectId, ResourceLimits, TaskInstanceId, TenantId, UserId,
};
use serde::{Deserialize, Serialize};
use crate::types::*;
#[derive(Clone, Debug, Serialize, Deserialize)]
#[serde(tag = "type", rename_all = "snake_case")]
pub(crate) enum AuthenticatedRequest {
AuthStatus,
RevokeCliSession,
CreateProject {
project: ProjectId,
name: String,
},
SelectProject {
project: ProjectId,
},
ListProjects,
RegisterAgentPublicKey {
agent: AgentId,
public_key: String,
},
ListAgentPublicKeys,
RotateAgentPublicKey {
agent: AgentId,
public_key: String,
},
RevokeAgentPublicKey {
agent: AgentId,
},
CreateNodeEnrollmentGrant {
ttl_seconds: u64,
},
ListNodeSummaries {
cursor: Option<String>,
limit: u32,
},
RevokeNodeCredential {
node: NodeId,
},
ListProcessSummaries {
cursor: Option<String>,
limit: u32,
},
CancelProcess {
process: ProcessId,
},
AbortProcess {
process: ProcessId,
launch_attempt: Option<String>,
},
QuotaStatus,
ListTaskEvents {
process: Option<ProcessId>,
},
ListTaskSnapshots {
process: ProcessId,
},
ListRecentLogs {
process: ProcessId,
task: Option<TaskInstanceId>,
after_sequence: Option<u64>,
limit: u32,
},
RestartTask {
process: ProcessId,
task: TaskInstanceId,
replacement_bundle: Option<TaskReplacementBundle>,
},
ResolveTaskFailure {
process: ProcessId,
task: TaskInstanceId,
resolution: TaskFailureResolution,
},
DebugAttach {
process: ProcessId,
},
CreateDebugEpoch {
process: ProcessId,
stopped_task: TaskInstanceId,
reason: String,
},
ResumeDebugEpoch {
process: ProcessId,
epoch: u64,
},
InspectDebugEpoch {
process: ProcessId,
epoch: u64,
},
ListArtifacts {
process: Option<ProcessId>,
cursor: Option<String>,
limit: u32,
},
GetArtifact {
artifact: ArtifactId,
},
CreateArtifactDownloadLink {
artifact: ArtifactId,
max_bytes: u64,
ttl_seconds: u64,
},
OpenArtifactDownloadStream {
artifact: ArtifactId,
max_bytes: u64,
token_digest: Digest,
chunk_bytes: u64,
},
RevokeArtifactDownloadLink {
artifact: ArtifactId,
token_digest: Digest,
},
}
#[derive(Clone, Serialize, Deserialize)]
#[serde(tag = "type", rename_all = "snake_case")]
pub(crate) enum LoginRequest {
BeginWebBrowserLogin {},
CancelWebBrowserLogin {
transaction_id: String,
},
ExchangeWebLoginHandoff {
transaction_id: String,
handoff_code: String,
},
}
#[derive(Clone, Deserialize)]
pub(crate) struct WireBrowserSession {
pub tenant: TenantId,
pub project: ProjectId,
pub user: UserId,
pub session_secret: String,
pub expires_at_epoch_seconds: u64,
}
#[allow(clippy::large_enum_variant)]
#[derive(Clone, Deserialize)]
#[serde(tag = "type", rename_all = "snake_case")]
pub(crate) enum WireResponse {
AuthStatus {
tenant: TenantId,
project: ProjectId,
actor: UserId,
authenticated: bool,
account_status: String,
suspended: bool,
disabled: bool,
deleted: bool,
manual_review: bool,
sanitized_reason: Option<String>,
next_actions: Vec<String>,
},
CliSessionRevoked {},
ProjectCreated {
project: Project,
},
ProjectSelected {
project: Project,
},
Projects {
projects: Vec<Project>,
},
AgentPublicKey {
record: AgentPublicKey,
},
AgentPublicKeys {
records: Vec<AgentPublicKey>,
},
NodeEnrollmentGrantCreated {
tenant: TenantId,
project: ProjectId,
grant: String,
scope: String,
expires_at_epoch_seconds: u64,
},
NodeSummaries {
nodes: Vec<NodeSummary>,
next_cursor: Option<String>,
},
NodeCredentialRevoked {
node: NodeId,
tenant: TenantId,
project: ProjectId,
actor: UserId,
descriptor_removed: bool,
queued_assignments_removed: usize,
},
ProcessSummaries {
processes: Vec<ProcessSummary>,
next_cursor: Option<String>,
},
ProcessCancellationRequested {
process: ProcessId,
cancelled_tasks: Vec<TaskCancellationTarget>,
affected_nodes: Vec<NodeId>,
},
ProcessAborted {
process: ProcessId,
aborted_tasks: Vec<TaskCancellationTarget>,
affected_nodes: Vec<NodeId>,
},
QuotaStatus {
tenant: TenantId,
project: ProjectId,
actor: UserId,
policy_label: Option<String>,
limits: ResourceLimits,
window_seconds: BTreeMap<LimitKind, u64>,
usage: BTreeMap<LimitKind, u64>,
window_started_epoch_seconds: BTreeMap<LimitKind, u64>,
},
TaskEvents {
events: Vec<TaskCompletionEvent>,
},
TaskSnapshots {
snapshots: Vec<TaskAttemptSnapshot>,
},
RecentLogs {
entries: Vec<RecentLogEntry>,
next_sequence: Option<u64>,
history_truncated: bool,
},
TaskRestart {
process: ProcessId,
task: TaskInstanceId,
restarted_task_instance: Option<TaskInstanceId>,
restarted_attempt_id: Option<String>,
actor: UserId,
accepted: bool,
clean_boundary_available: bool,
active_task: bool,
completed_event_observed: bool,
requires_whole_process_restart: bool,
message: String,
audit_event: DebugAuditEvent,
},
TaskFailureResolved {
process: ProcessId,
task: TaskInstanceId,
attempt_id: String,
resolution: TaskFailureResolution,
},
DebugAttach {
process: ProcessId,
actor: UserId,
authorization: Authorization,
audit_event: DebugAuditEvent,
},
DebugEpoch {
process: ProcessId,
actor: UserId,
epoch: u64,
command: String,
affected_tasks: Vec<TaskCancellationTarget>,
all_stop_requested: bool,
audit_event: DebugAuditEvent,
},
DebugEpochStatus {
process: ProcessId,
actor: UserId,
epoch: u64,
command: String,
expected_tasks: Vec<TaskCancellationTarget>,
acknowledgements: Vec<DebugParticipantAcknowledgement>,
fully_frozen: bool,
partially_frozen: bool,
fully_resumed: bool,
failed: bool,
failure_messages: Vec<String>,
audit_event: DebugAuditEvent,
},
Artifacts {
artifacts: Vec<ArtifactSummary>,
next_cursor: Option<String>,
},
Artifact {
artifact: ArtifactSummary,
},
ArtifactDownloadLink {
link: DownloadLink,
},
ArtifactDownloadLinkRevoked {},
ArtifactDownloadStream {
#[serde(rename = "link")]
_link: DownloadLink,
content_bytes_available: bool,
content_offset: Option<u64>,
content_eof: bool,
content_base64: Option<String>,
},
WebBrowserLoginStarted {
transaction_id: String,
authorization_url: String,
expires_at_epoch_seconds: u64,
},
WebBrowserLoginCancelled {},
WebBrowserSession {
session: WireBrowserSession,
},
Error {
code: ApiErrorCode,
category: ApiErrorCategory,
message: String,
retryable: bool,
request_id: String,
},
}
#[cfg(test)]
mod tests {
use std::collections::BTreeSet;
use serde::Deserialize;
use serde_json::Value;
use super::*;
#[derive(Deserialize)]
struct OperationFixture {
operation: String,
boundary: String,
request: Value,
response: Value,
}
#[test]
fn website_operation_contract_fixtures_cover_every_typed_request() {
let fixtures: Vec<OperationFixture> =
serde_json::from_str(include_str!("../tests/fixtures/web_operations.json")).unwrap();
let expected = BTreeSet::from([
"abort_process",
"auth_status",
"begin_web_browser_login",
"cancel_web_browser_login",
"cancel_process",
"create_artifact_download_link",
"create_debug_epoch",
"create_node_enrollment_grant",
"create_project",
"debug_attach",
"exchange_web_login_handoff",
"get_artifact",
"inspect_debug_epoch",
"list_agent_public_keys",
"list_artifacts",
"list_node_summaries",
"list_process_summaries",
"list_projects",
"list_recent_logs",
"list_task_events",
"list_task_snapshots",
"open_artifact_download_stream",
"quota_status",
"register_agent_public_key",
"resolve_task_failure",
"restart_task",
"resume_debug_epoch",
"revoke_agent_public_key",
"revoke_artifact_download_link",
"revoke_cli_session",
"revoke_node_credential",
"rotate_agent_public_key",
"select_project",
])
.into_iter()
.map(str::to_owned)
.collect();
let mut observed = BTreeSet::new();
for fixture in fixtures {
assert_eq!(fixture.request["type"], fixture.operation);
let round_trip = match fixture.boundary.as_str() {
"control" => {
let typed: AuthenticatedRequest =
serde_json::from_value(fixture.request.clone()).unwrap();
serde_json::to_value(typed).unwrap()
}
"login" => {
let typed: LoginRequest =
serde_json::from_value(fixture.request.clone()).unwrap();
serde_json::to_value(typed).unwrap()
}
boundary => panic!("unknown fixture boundary {boundary}"),
};
assert_eq!(round_trip, fixture.request);
let response: WireResponse = serde_json::from_value(fixture.response).unwrap();
assert!(
!matches!(response, WireResponse::Error { .. }),
"{} must carry its operation-specific success response fixture",
fixture.operation
);
assert!(observed.insert(fixture.operation));
}
assert_eq!(observed, expected);
}
}

View file

@ -1,250 +0,0 @@
use std::collections::{BTreeMap, VecDeque};
use std::future::Future;
use std::pin::Pin;
use std::sync::atomic::{AtomicU64, Ordering};
use std::sync::{Arc, Mutex};
use std::time::Duration;
use clusterflux_control::{endpoint_identity, ControlSession};
use thiserror::Error;
pub type TransportFuture =
Pin<Box<dyn Future<Output = Result<TransportResponse, ClientTransportError>> + Send + 'static>>;
#[derive(Clone, Debug, PartialEq, Eq)]
pub struct TransportRequest {
pub api_path: String,
pub body: Vec<u8>,
}
#[derive(Clone, Debug, PartialEq, Eq)]
pub struct TransportResponse {
pub body: Vec<u8>,
}
#[derive(Clone, Debug, Error, PartialEq, Eq)]
pub enum ClientTransportError {
#[error("client transport failed: {0}")]
Failed(String),
#[error("client transport task failed: {0}")]
Task(String),
}
pub trait ClientTransport: Send + Sync + 'static {
fn send(&self, request: TransportRequest) -> TransportFuture;
}
type ControlSessionPool = Arc<Vec<Mutex<Option<ControlSession>>>>;
pub struct ControlTransport {
endpoint: String,
connect_timeout: Duration,
io_timeout: Duration,
sessions: Arc<Mutex<BTreeMap<String, ControlSessionPool>>>,
next_session: Arc<AtomicU64>,
}
// A small pool allows independent HTMX reads to progress concurrently while
// keeping connection and blocking-worker use strictly bounded.
const SESSIONS_PER_API_PATH: usize = 4;
impl ControlTransport {
pub fn new(endpoint: impl Into<String>) -> Result<Self, ClientTransportError> {
Self::with_timeouts(endpoint, Duration::from_secs(10), Duration::from_secs(30))
}
pub fn with_timeouts(
endpoint: impl Into<String>,
connect_timeout: Duration,
io_timeout: Duration,
) -> Result<Self, ClientTransportError> {
let endpoint = endpoint.into();
endpoint_identity(&endpoint)
.map_err(|error| ClientTransportError::Failed(error.to_string()))?;
Ok(Self {
endpoint,
connect_timeout,
io_timeout,
sessions: Arc::new(Mutex::new(BTreeMap::new())),
next_session: Arc::new(AtomicU64::new(0)),
})
}
}
impl ClientTransport for ControlTransport {
fn send(&self, request: TransportRequest) -> TransportFuture {
let endpoint = self.endpoint.clone();
let connect_timeout = self.connect_timeout;
let io_timeout = self.io_timeout;
let sessions = Arc::clone(&self.sessions);
let next_session = Arc::clone(&self.next_session);
Box::pin(async move {
let pool = {
let mut sessions = sessions.lock().map_err(|_| {
ClientTransportError::Failed(
"client transport pool lock was poisoned".to_owned(),
)
})?;
Arc::clone(sessions.entry(request.api_path.clone()).or_insert_with(|| {
Arc::new(
(0..SESSIONS_PER_API_PATH)
.map(|_| Mutex::new(None))
.collect(),
)
}))
};
let slot_index =
next_session.fetch_add(1, Ordering::Relaxed) as usize % SESSIONS_PER_API_PATH;
tokio::task::spawn_blocking(move || {
let value = serde_json::from_slice(&request.body)
.map_err(|error| ClientTransportError::Failed(error.to_string()))?;
let mut selected = None;
for offset in 0..SESSIONS_PER_API_PATH {
let index = (slot_index + offset) % SESSIONS_PER_API_PATH;
match pool[index].try_lock() {
Ok(session) if session.is_some() => {
selected = Some(session);
break;
}
Ok(_) | Err(std::sync::TryLockError::WouldBlock) => {}
Err(std::sync::TryLockError::Poisoned(_)) => {
return Err(ClientTransportError::Failed(
"client transport session lock was poisoned".to_owned(),
));
}
}
}
if selected.is_none() {
for offset in 0..SESSIONS_PER_API_PATH {
let index = (slot_index + offset) % SESSIONS_PER_API_PATH;
match pool[index].try_lock() {
Ok(session) => {
selected = Some(session);
break;
}
Err(std::sync::TryLockError::WouldBlock) => {}
Err(std::sync::TryLockError::Poisoned(_)) => {
return Err(ClientTransportError::Failed(
"client transport session lock was poisoned".to_owned(),
));
}
}
}
}
let mut session = match selected {
Some(session) => session,
None => pool[slot_index].lock().map_err(|_| {
ClientTransportError::Failed(
"client transport session lock was poisoned".to_owned(),
)
})?,
};
if session.is_none() {
*session = Some(
ControlSession::connect_to_api_path_with_timeouts(
&endpoint,
&request.api_path,
connect_timeout,
io_timeout,
)
.map_err(|error| ClientTransportError::Failed(error.to_string()))?,
);
}
let response = session
.as_mut()
.expect("session was initialized for the requested API path")
.request(&value);
match response {
Ok(response) => serde_json::to_vec(&response)
.map(|body| TransportResponse { body })
.map_err(|error| ClientTransportError::Failed(error.to_string())),
Err(error) => {
*session = None;
Err(ClientTransportError::Failed(error.to_string()))
}
}
})
.await
.map_err(|error| ClientTransportError::Task(error.to_string()))?
})
}
}
#[derive(Clone, Default)]
pub struct MockTransport {
state: Arc<Mutex<MockTransportState>>,
}
#[derive(Default)]
struct MockTransportState {
responses: VecDeque<Result<Vec<u8>, ClientTransportError>>,
requests: Vec<TransportRequest>,
}
impl MockTransport {
pub fn from_json_responses(responses: impl IntoIterator<Item = impl Into<String>>) -> Self {
let responses = responses
.into_iter()
.map(|response| Ok(response.into().into_bytes()))
.collect();
Self {
state: Arc::new(Mutex::new(MockTransportState {
responses,
requests: Vec::new(),
})),
}
}
pub fn push_json_response(&self, response: impl Into<String>) {
self.state
.lock()
.expect("mock transport lock is not poisoned")
.responses
.push_back(Ok(response.into().into_bytes()));
}
pub fn push_error(&self, message: impl Into<String>) {
self.state
.lock()
.expect("mock transport lock is not poisoned")
.responses
.push_back(Err(ClientTransportError::Failed(message.into())));
}
pub fn request_bodies(&self) -> Vec<String> {
self.state
.lock()
.expect("mock transport lock is not poisoned")
.requests
.iter()
.map(|request| String::from_utf8_lossy(&request.body).into_owned())
.collect()
}
pub fn requests(&self) -> Vec<TransportRequest> {
self.state
.lock()
.expect("mock transport lock is not poisoned")
.requests
.clone()
}
}
impl ClientTransport for MockTransport {
fn send(&self, request: TransportRequest) -> TransportFuture {
let state = Arc::clone(&self.state);
Box::pin(async move {
let mut state = state.lock().map_err(|_| {
ClientTransportError::Failed("mock transport lock was poisoned".to_owned())
})?;
state.requests.push(request);
state
.responses
.pop_front()
.ok_or_else(|| {
ClientTransportError::Failed("mock transport has no queued response".to_owned())
})?
.map(|body| TransportResponse { body })
})
}
}

View file

@ -1,468 +0,0 @@
use std::collections::BTreeMap;
use clusterflux_core::{
AgentId, ArtifactId, Authorization, Digest, LimitKind, NodeCapabilities, NodeId, Placement,
ProcessId, ProjectId, ResourceLimits, TaskBoundaryValue, TaskDefinitionId, TaskFailurePolicy,
TaskInstanceId, TenantId, UserId, VfsPath,
};
use serde::{Deserialize, Serialize};
use std::fmt;
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub struct AccountStatus {
pub tenant: TenantId,
pub project: ProjectId,
pub actor: UserId,
pub authenticated: bool,
pub account_status: String,
pub suspended: bool,
pub disabled: bool,
pub deleted: bool,
pub manual_review: bool,
pub sanitized_reason: Option<String>,
pub next_actions: Vec<String>,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub struct Project {
pub id: ProjectId,
pub tenant: TenantId,
pub name: String,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub struct AgentPublicKey {
pub tenant: TenantId,
pub project: ProjectId,
pub user: UserId,
pub agent: AgentId,
pub public_key: String,
pub public_key_fingerprint: Digest,
pub version: u64,
pub revoked: bool,
pub scopes: Vec<String>,
pub human_account_creation_privilege: bool,
pub browser_interaction_required_each_run: bool,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub struct NodeEnrollmentGrant {
pub tenant: TenantId,
pub project: ProjectId,
pub grant: String,
pub scope: String,
pub expires_at_epoch_seconds: u64,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub struct NodeSummary {
pub id: NodeId,
pub display_name: String,
pub online: bool,
pub stale: bool,
pub last_seen_epoch_seconds: Option<u64>,
pub capabilities: NodeCapabilities,
pub direct_connectivity: bool,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub struct NodePage {
pub nodes: Vec<NodeSummary>,
pub next_cursor: Option<String>,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub struct NodeRevocation {
pub node: NodeId,
pub tenant: TenantId,
pub project: ProjectId,
pub actor: UserId,
pub descriptor_removed: bool,
pub queued_assignments_removed: usize,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum ProcessLifecycleState {
Active,
RecentTerminal,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum ProcessActivityState {
Running,
WaitingForNode,
WaitingForTask,
AwaitingAction,
DebugEpochPartial,
Cancelling,
Completed,
Failed,
Cancelled,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum ProcessFinalResult {
Completed,
Failed,
Cancelled,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub struct DebugEpochSummary {
pub epoch: u64,
pub command: String,
pub fully_frozen: bool,
pub partially_frozen: bool,
pub fully_resumed: bool,
pub failed: bool,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub struct ProcessSummary {
pub process: ProcessId,
pub lifecycle: ProcessLifecycleState,
pub activity: ProcessActivityState,
pub main_wait_state: Option<String>,
pub started_at_epoch_seconds: u64,
pub ended_at_epoch_seconds: Option<u64>,
pub final_result: Option<ProcessFinalResult>,
pub connected_nodes: Vec<NodeId>,
pub current_debug_epoch: Option<DebugEpochSummary>,
pub order_cursor: String,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub struct ProcessPage {
pub processes: Vec<ProcessSummary>,
pub next_cursor: Option<String>,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum TaskTerminalState {
Completed,
Failed,
Cancelled,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum TaskExecutor {
CoordinatorMain,
Node,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub struct TaskCompletionEvent {
pub tenant: TenantId,
pub project: ProjectId,
pub process: ProcessId,
pub node: NodeId,
pub executor: TaskExecutor,
pub task_definition: TaskDefinitionId,
pub task: TaskInstanceId,
pub attempt_id: Option<String>,
pub placement: Option<Placement>,
pub terminal_state: TaskTerminalState,
pub status_code: Option<i32>,
pub stdout_bytes: u64,
pub stderr_bytes: u64,
pub stdout_tail: String,
pub stderr_tail: String,
pub stdout_truncated: bool,
pub stderr_truncated: bool,
pub artifact_path: Option<VfsPath>,
pub artifact_digest: Option<Digest>,
pub artifact_size_bytes: Option<u64>,
pub result: Option<TaskBoundaryValue>,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum TaskAttemptState {
Queued,
Running,
FailedAwaitingAction,
Completed,
Failed,
Cancelled,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub struct TaskAttemptSnapshot {
pub process: ProcessId,
pub task: TaskInstanceId,
pub attempt_id: String,
pub attempt_number: u32,
pub task_definition: TaskDefinitionId,
pub display_name: String,
pub state: TaskAttemptState,
pub current: bool,
pub node: Option<NodeId>,
pub environment_id: Option<String>,
pub environment_digest: Option<Digest>,
pub argument_summary: Vec<String>,
pub handle_summary: Vec<String>,
pub command_state: Option<String>,
pub vfs_checkpoint: String,
pub probe_symbol: Option<String>,
pub source_path: Option<String>,
pub source_line: Option<u32>,
pub restart_compatible: bool,
pub failure_policy: TaskFailurePolicy,
pub artifact_path: Option<VfsPath>,
pub artifact_digest: Option<Digest>,
pub artifact_size_bytes: Option<u64>,
pub status_code: Option<i32>,
pub error: Option<String>,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum TaskLogStream {
Stdout,
Stderr,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub struct RecentLogEntry {
pub sequence: u64,
pub process: ProcessId,
pub task: TaskInstanceId,
pub stream: TaskLogStream,
pub text: String,
pub server_timestamp_epoch_seconds: u64,
pub truncated: bool,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub struct RecentLogPage {
pub entries: Vec<RecentLogEntry>,
pub next_sequence: Option<u64>,
pub history_truncated: bool,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum ArtifactAvailability {
Available,
NodeOffline,
Unavailable,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum ArtifactRetentionState {
NodeRetained,
ExplicitStorage,
Lost,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub struct ArtifactSummary {
pub id: ArtifactId,
pub display_path: String,
pub display_name: String,
pub process: ProcessId,
pub producer_task: TaskInstanceId,
pub safe_node: Option<NodeId>,
pub digest: Digest,
pub size_bytes: u64,
pub availability: ArtifactAvailability,
pub downloadable_now: bool,
pub retention_state: ArtifactRetentionState,
pub explicit_storage: bool,
pub order_cursor: String,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub struct ArtifactPage {
pub artifacts: Vec<ArtifactSummary>,
pub next_cursor: Option<String>,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub struct QuotaStatus {
pub tenant: TenantId,
pub project: ProjectId,
pub actor: UserId,
pub policy_label: Option<String>,
pub limits: ResourceLimits,
pub window_seconds: BTreeMap<LimitKind, u64>,
pub usage: BTreeMap<LimitKind, u64>,
pub window_started_epoch_seconds: BTreeMap<LimitKind, u64>,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub struct TaskCancellationTarget {
pub process: ProcessId,
pub task: TaskInstanceId,
pub node: NodeId,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub struct ProcessCancellation {
pub process: ProcessId,
pub affected_tasks: Vec<TaskCancellationTarget>,
pub affected_nodes: Vec<NodeId>,
pub aborted: bool,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub struct DebugAuditEvent {
pub tenant: TenantId,
pub project: ProjectId,
pub process: ProcessId,
pub task: Option<TaskInstanceId>,
pub actor: UserId,
pub operation: String,
pub allowed: bool,
pub reason: String,
pub charged_debug_read_bytes: u64,
pub used_debug_read_bytes: u64,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub struct DebugAttach {
pub process: ProcessId,
pub actor: UserId,
pub authorization: Authorization,
pub audit_event: DebugAuditEvent,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub struct DebugEpochControl {
pub process: ProcessId,
pub actor: UserId,
pub epoch: u64,
pub command: String,
pub affected_tasks: Vec<TaskCancellationTarget>,
pub all_stop_requested: bool,
pub audit_event: DebugAuditEvent,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum DebugAcknowledgementState {
Frozen,
Running,
Failed,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub struct DebugParticipantAcknowledgement {
pub node: NodeId,
pub task_definition: TaskDefinitionId,
pub task: TaskInstanceId,
pub epoch: u64,
pub state: DebugAcknowledgementState,
pub stack_frames: Vec<String>,
pub local_values: Vec<(String, String)>,
pub task_args: Vec<(String, String)>,
pub handles: Vec<(String, String)>,
pub command_status: Option<String>,
pub recent_output: Vec<String>,
pub message: Option<String>,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub struct DebugEpochStatus {
pub process: ProcessId,
pub actor: UserId,
pub epoch: u64,
pub command: String,
pub expected_tasks: Vec<TaskCancellationTarget>,
pub acknowledgements: Vec<DebugParticipantAcknowledgement>,
pub fully_frozen: bool,
pub partially_frozen: bool,
pub fully_resumed: bool,
pub failed: bool,
pub failure_messages: Vec<String>,
pub audit_event: DebugAuditEvent,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub struct TaskRestart {
pub process: ProcessId,
pub task: TaskInstanceId,
pub restarted_task_instance: Option<TaskInstanceId>,
pub restarted_attempt_id: Option<String>,
pub actor: UserId,
pub accepted: bool,
pub clean_boundary_available: bool,
pub active_task: bool,
pub completed_event_observed: bool,
pub requires_whole_process_restart: bool,
pub message: String,
pub audit_event: DebugAuditEvent,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub struct TaskFailureResolutionResult {
pub process: ProcessId,
pub task: TaskInstanceId,
pub attempt_id: String,
pub resolution: TaskFailureResolution,
}
#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum TaskFailureResolution {
AcceptFailure,
Cancel,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub struct TaskReplacementBundle {
pub bundle_digest: Digest,
pub wasm_module_base64: String,
pub source_snapshot: Option<Digest>,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub struct BrowserLoginStart {
pub transaction_id: String,
pub authorization_url: String,
pub expires_at_epoch_seconds: u64,
}
#[derive(Clone, PartialEq, Eq)]
pub struct SessionCredential(pub(crate) String);
impl SessionCredential {
pub fn from_secret(secret: impl Into<String>) -> Self {
Self(secret.into())
}
pub fn expose_secret(&self) -> &str {
&self.0
}
}
impl fmt::Debug for SessionCredential {
fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result {
formatter.write_str("SessionCredential([REDACTED])")
}
}
#[derive(Clone, Debug, PartialEq, Eq)]
pub struct BrowserSession {
pub tenant: TenantId,
pub project: ProjectId,
pub user: UserId,
pub credential: SessionCredential,
pub expires_at_epoch_seconds: u64,
}
#[derive(Clone, Debug, PartialEq, Eq)]
pub enum ArtifactDownloadPoll {
Pending,
Chunk {
offset: u64,
bytes: Vec<u8>,
eof: bool,
},
}

View file

@ -1,254 +0,0 @@
use std::net::TcpListener;
use std::time::Duration;
use clusterflux_client::{
ApiErrorCategory, ApiErrorCode, ArtifactDownloadPoll, ArtifactId, ClientError,
ClusterfluxClient, ControlTransport, MockTransport, ProjectId, SessionCredential, TenantId,
UserId, CLIENT_API_VERSION,
};
use clusterflux_control::{CONTROL_API_PATH, LOGIN_API_PATH};
use clusterflux_coordinator::service::CoordinatorService;
use serde_json::{json, Value};
#[tokio::test]
async fn mock_transport_exercises_typed_envelope_and_session_plumbing() {
let transport = MockTransport::from_json_responses([json!({
"type": "projects",
"projects": [{
"id": "project-one",
"tenant": "tenant-one",
"name": "Project one"
}],
"actor": "user-one"
})
.to_string()]);
let client = ClusterfluxClient::with_transport(transport.clone())
.with_session_credential(&SessionCredential::from_secret("test-session-secret"));
let projects = client.list_projects().await.unwrap();
assert_eq!(projects.len(), 1);
assert_eq!(projects[0].id, ProjectId::from("project-one"));
let requests = transport.requests();
assert_eq!(requests.len(), 1);
assert_eq!(requests[0].api_path, CONTROL_API_PATH);
let envelope: Value = serde_json::from_slice(&requests[0].body).unwrap();
assert_eq!(envelope["protocol_version"], CLIENT_API_VERSION);
assert_eq!(envelope["request_id"], "client-1");
assert_eq!(envelope["operation"], "authenticated");
assert_eq!(envelope["payload"]["type"], "authenticated");
assert_eq!(envelope["payload"]["request"]["type"], "list_projects");
assert_eq!(envelope["payload"]["session_secret"], "test-session-secret");
}
#[tokio::test]
async fn structured_errors_retain_machine_fields_and_originating_request_id() {
let transport = MockTransport::from_json_responses([json!({
"type": "error",
"code": "account_suspended",
"category": "authorization",
"message": "account access is suspended",
"retryable": false,
"request_id": "client-1"
})
.to_string()]);
let client = ClusterfluxClient::with_transport(transport)
.with_session_credential(&SessionCredential::from_secret("test-session-secret"));
let ClientError::Api(error) = client.account_status().await.unwrap_err() else {
panic!("expected typed API error");
};
assert_eq!(error.code, ApiErrorCode::AccountSuspended);
assert_eq!(error.category, ApiErrorCategory::Authorization);
assert_eq!(error.request_id, "client-1");
assert!(!error.retryable);
}
#[tokio::test]
async fn browser_login_cancellation_uses_the_login_boundary_and_exact_transaction() {
let transport = MockTransport::from_json_responses([
json!({ "type": "web_browser_login_cancelled" }).to_string(),
]);
let client = ClusterfluxClient::with_transport(transport.clone());
client
.cancel_browser_login("login-transaction")
.await
.unwrap();
let requests = transport.requests();
assert_eq!(requests.len(), 1);
assert_eq!(requests[0].api_path, LOGIN_API_PATH);
let envelope: Value = serde_json::from_slice(&requests[0].body).unwrap();
assert_eq!(envelope["operation"], "cancel_web_browser_login");
assert_eq!(
envelope["payload"],
json!({
"type": "cancel_web_browser_login",
"transaction_id": "login-transaction"
})
);
}
#[tokio::test]
async fn a_mismatched_error_request_id_is_rejected_as_a_protocol_error() {
let transport = MockTransport::from_json_responses([json!({
"type": "error",
"code": "validation_error",
"category": "validation",
"message": "bad request",
"retryable": false,
"request_id": "another-request"
})
.to_string()]);
let client = ClusterfluxClient::with_transport(transport)
.with_session_credential(&SessionCredential::from_secret("test-session-secret"));
let ClientError::Protocol(message) = client.list_projects().await.unwrap_err() else {
panic!("expected protocol error");
};
assert!(message.contains("does not match client-1"));
}
#[test]
fn session_credentials_are_redacted_from_debug_output() {
let credential = SessionCredential::from_secret("must-not-appear");
assert_eq!(format!("{credential:?}"), "SessionCredential([REDACTED])");
}
#[tokio::test]
async fn artifact_download_is_a_typed_bounded_chunk_stream() {
let link = json!({
"artifact": "artifact-one",
"artifact_digest": "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa",
"artifact_size_bytes": 5,
"source": { "RetainedNode": "node-one" },
"url_path": "/artifacts/tenant-one/project-one/process-one/artifact-one",
"scoped_token_digest": "sha256:bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb",
"expires_at_epoch_seconds": 1000,
"tenant": "tenant-one",
"project": "project-one",
"process": "process-one",
"actor": { "User": "user-one" },
"max_bytes": 1024,
"policy_context_digest": "sha256:cccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccc"
});
let transport = MockTransport::from_json_responses([
json!({ "type": "artifact_download_link", "link": link.clone() }).to_string(),
json!({
"type": "artifact_download_stream",
"link": link.clone(),
"streamed_bytes": 0,
"charged_download_bytes": 0,
"content_bytes_available": false,
"content_eof": false
})
.to_string(),
json!({
"type": "artifact_download_stream",
"link": link,
"streamed_bytes": 5,
"charged_download_bytes": 5,
"content_bytes_available": true,
"content_offset": 0,
"content_eof": true,
"content_base64": "aGVsbG8="
})
.to_string(),
]);
let client = ClusterfluxClient::with_transport(transport)
.with_session_credential(&SessionCredential::from_secret("test-session-secret"));
let mut download = client
.begin_artifact_download(ArtifactId::from("artifact-one"), 1024, 60, 64)
.await
.unwrap();
assert_eq!(
download.next_chunk().await.unwrap(),
ArtifactDownloadPoll::Pending
);
assert_eq!(
download.next_chunk().await.unwrap(),
ArtifactDownloadPoll::Chunk {
offset: 0,
bytes: b"hello".to_vec(),
eof: true,
}
);
}
#[tokio::test]
async fn typed_client_runs_against_the_real_strict_control_endpoint() {
let listener = TcpListener::bind("127.0.0.1:0").unwrap();
let address = listener.local_addr().unwrap();
let server = std::thread::spawn(move || {
let mut service = CoordinatorService::new(7);
service
.issue_cli_session(
TenantId::from("tenant-one"),
ProjectId::from("project-one"),
UserId::from("user-one"),
"real-endpoint-session",
None,
)
.unwrap();
let (stream, _) = listener.accept().unwrap();
service.handle_stream(stream).unwrap();
});
let client = ClusterfluxClient::connect(format!("clusterflux+tcp://{address}"))
.unwrap()
.with_session_credential(&SessionCredential::from_secret("real-endpoint-session"));
let projects = client.list_projects().await.unwrap();
assert_eq!(projects.len(), 1);
assert_eq!(projects[0].id, ProjectId::from("project-one"));
let status = client.account_status().await.unwrap();
assert!(status.authenticated);
assert_eq!(status.actor, UserId::from("user-one"));
drop(client);
server.join().unwrap();
}
#[tokio::test]
async fn concurrent_requests_expand_the_bounded_pool_without_serializing() {
let listener = TcpListener::bind("127.0.0.1:0").unwrap();
let address = listener.local_addr().unwrap();
let server = std::thread::spawn(move || {
let handlers = (0..2)
.map(|_| {
let (stream, _) = listener.accept().unwrap();
std::thread::spawn(move || {
let mut service = CoordinatorService::new(7);
service
.issue_cli_session(
TenantId::from("tenant-one"),
ProjectId::from("project-one"),
UserId::from("user-one"),
"concurrent-session",
None,
)
.unwrap();
service.handle_stream(stream).unwrap();
})
})
.collect::<Vec<_>>();
for handler in handlers {
handler.join().unwrap();
}
});
let transport = ControlTransport::with_timeouts(
format!("clusterflux+tcp://{address}"),
Duration::from_secs(2),
Duration::from_secs(5),
)
.unwrap();
let client = ClusterfluxClient::with_transport(transport)
.with_session_credential(&SessionCredential::from_secret("concurrent-session"));
let (first, second) = tokio::join!(client.account_status(), client.account_status());
assert!(first.unwrap().authenticated);
assert!(second.unwrap().authenticated);
drop(client);
server.join().unwrap();
}

View file

@ -1,200 +0,0 @@
[
{
"operation": "begin_web_browser_login",
"boundary": "login",
"request": { "type": "begin_web_browser_login" },
"response": { "type": "web_browser_login_started", "transaction_id": "login-transaction", "authorization_url": "https://auth.clusterflux.lesstuff.com/authorize?state=opaque", "expires_at_epoch_seconds": 1000 }
},
{
"operation": "cancel_web_browser_login",
"boundary": "login",
"request": { "type": "cancel_web_browser_login", "transaction_id": "login-transaction" },
"response": { "type": "web_browser_login_cancelled" }
},
{
"operation": "exchange_web_login_handoff",
"boundary": "login",
"request": { "type": "exchange_web_login_handoff", "transaction_id": "login-transaction", "handoff_code": "one-time-handoff" },
"response": { "type": "web_browser_session", "session": { "tenant": "tenant-one", "project": "project-one", "user": "user-one", "session_secret": "server-side-session", "expires_at_epoch_seconds": 2000 } }
},
{
"operation": "auth_status",
"boundary": "control",
"request": { "type": "auth_status" },
"response": { "type": "auth_status", "tenant": "tenant-one", "project": "project-one", "actor": "user-one", "authenticated": true, "account_status": "active", "suspended": false, "disabled": false, "deleted": false, "manual_review": false, "sanitized_reason": null, "next_actions": [] }
},
{
"operation": "revoke_cli_session",
"boundary": "control",
"request": { "type": "revoke_cli_session" },
"response": { "type": "cli_session_revoked" }
},
{
"operation": "create_project",
"boundary": "control",
"request": { "type": "create_project", "project": "project-new", "name": "New project" },
"response": { "type": "project_created", "project": { "id": "project-new", "tenant": "tenant-one", "name": "New project" } }
},
{
"operation": "select_project",
"boundary": "control",
"request": { "type": "select_project", "project": "project-selected" },
"response": { "type": "project_selected", "project": { "id": "project-selected", "tenant": "tenant-one", "name": "Selected project" } }
},
{
"operation": "list_projects",
"boundary": "control",
"request": { "type": "list_projects" },
"response": { "type": "projects", "projects": [{ "id": "project-one", "tenant": "tenant-one", "name": "Project one" }] }
},
{
"operation": "register_agent_public_key",
"boundary": "control",
"request": { "type": "register_agent_public_key", "agent": "agent-browser", "public_key": "ed25519:test-public-key" },
"response": { "type": "agent_public_key", "record": { "tenant": "tenant-one", "project": "project-one", "user": "user-one", "agent": "agent-browser", "public_key": "ed25519:test-public-key", "public_key_fingerprint": "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", "version": 1, "revoked": false, "scopes": ["project"], "human_account_creation_privilege": false, "browser_interaction_required_each_run": false } }
},
{
"operation": "list_agent_public_keys",
"boundary": "control",
"request": { "type": "list_agent_public_keys" },
"response": { "type": "agent_public_keys", "records": [] }
},
{
"operation": "rotate_agent_public_key",
"boundary": "control",
"request": { "type": "rotate_agent_public_key", "agent": "agent-browser", "public_key": "ed25519:rotated-public-key" },
"response": { "type": "agent_public_key", "record": { "tenant": "tenant-one", "project": "project-one", "user": "user-one", "agent": "agent-browser", "public_key": "ed25519:rotated-public-key", "public_key_fingerprint": "sha256:bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb", "version": 2, "revoked": false, "scopes": ["project"], "human_account_creation_privilege": false, "browser_interaction_required_each_run": false } }
},
{
"operation": "revoke_agent_public_key",
"boundary": "control",
"request": { "type": "revoke_agent_public_key", "agent": "agent-browser" },
"response": { "type": "agent_public_key", "record": { "tenant": "tenant-one", "project": "project-one", "user": "user-one", "agent": "agent-browser", "public_key": "ed25519:rotated-public-key", "public_key_fingerprint": "sha256:bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb", "version": 3, "revoked": true, "scopes": ["project"], "human_account_creation_privilege": false, "browser_interaction_required_each_run": false } }
},
{
"operation": "create_node_enrollment_grant",
"boundary": "control",
"request": { "type": "create_node_enrollment_grant", "ttl_seconds": 300 },
"response": { "type": "node_enrollment_grant_created", "tenant": "tenant-one", "project": "project-one", "grant": "one-time-grant", "scope": "tenant-one/project-one", "expires_at_epoch_seconds": 1000 }
},
{
"operation": "list_node_summaries",
"boundary": "control",
"request": { "type": "list_node_summaries", "cursor": null, "limit": 200 },
"response": { "type": "node_summaries", "nodes": [], "next_cursor": null }
},
{
"operation": "revoke_node_credential",
"boundary": "control",
"request": { "type": "revoke_node_credential", "node": "node-one" },
"response": { "type": "node_credential_revoked", "node": "node-one", "tenant": "tenant-one", "project": "project-one", "actor": "user-one", "descriptor_removed": true, "queued_assignments_removed": 0 }
},
{
"operation": "list_process_summaries",
"boundary": "control",
"request": { "type": "list_process_summaries", "cursor": "process:1", "limit": 20 },
"response": { "type": "process_summaries", "processes": [], "next_cursor": null }
},
{
"operation": "cancel_process",
"boundary": "control",
"request": { "type": "cancel_process", "process": "process-one" },
"response": { "type": "process_cancellation_requested", "process": "process-one", "cancelled_tasks": [], "affected_nodes": [] }
},
{
"operation": "abort_process",
"boundary": "control",
"request": { "type": "abort_process", "process": "process-one", "launch_attempt": null },
"response": { "type": "process_aborted", "process": "process-one", "aborted_tasks": [], "affected_nodes": [] }
},
{
"operation": "quota_status",
"boundary": "control",
"request": { "type": "quota_status" },
"response": { "type": "quota_status", "tenant": "tenant-one", "project": "project-one", "actor": "user-one", "policy_label": "community tier", "limits": { "limits": { "ApiCall": 10000 } }, "window_seconds": { "ApiCall": 60 }, "usage": { "ApiCall": 12 }, "window_started_epoch_seconds": { "ApiCall": 900 } }
},
{
"operation": "list_task_events",
"boundary": "control",
"request": { "type": "list_task_events", "process": "process-one" },
"response": { "type": "task_events", "events": [] }
},
{
"operation": "list_task_snapshots",
"boundary": "control",
"request": { "type": "list_task_snapshots", "process": "process-one" },
"response": { "type": "task_snapshots", "snapshots": [] }
},
{
"operation": "list_recent_logs",
"boundary": "control",
"request": { "type": "list_recent_logs", "process": "process-one", "task": "task-one", "after_sequence": 7, "limit": 100 },
"response": { "type": "recent_logs", "entries": [{ "sequence": 1, "process": "process-one", "task": "task-one", "stream": "stdout", "text": "building", "server_timestamp_epoch_seconds": 1000, "truncated": false }], "next_sequence": 1, "history_truncated": false }
},
{
"operation": "restart_task",
"boundary": "control",
"request": { "type": "restart_task", "process": "process-one", "task": "task-one", "replacement_bundle": null },
"response": { "type": "task_restart", "process": "process-one", "task": "task-one", "restarted_task_instance": "task-one-retry", "restarted_attempt_id": "attempt-2", "actor": "user-one", "accepted": true, "clean_boundary_available": true, "active_task": false, "completed_event_observed": true, "requires_whole_process_restart": false, "message": "task restart accepted", "audit_event": { "tenant": "tenant-one", "project": "project-one", "process": "process-one", "task": "task-one", "actor": "user-one", "operation": "restart_task", "allowed": true, "reason": "authorized", "charged_debug_read_bytes": 0, "used_debug_read_bytes": 0 } }
},
{
"operation": "resolve_task_failure",
"boundary": "control",
"request": { "type": "resolve_task_failure", "process": "process-one", "task": "task-one", "resolution": "accept_failure" },
"response": { "type": "task_failure_resolved", "process": "process-one", "task": "task-one", "attempt_id": "attempt-1", "resolution": "accept_failure" }
},
{
"operation": "debug_attach",
"boundary": "control",
"request": { "type": "debug_attach", "process": "process-one" },
"response": { "type": "debug_attach", "process": "process-one", "actor": "user-one", "authorization": { "allowed": true, "reason": "authorized" }, "audit_event": { "tenant": "tenant-one", "project": "project-one", "process": "process-one", "task": null, "actor": "user-one", "operation": "debug_attach", "allowed": true, "reason": "authorized", "charged_debug_read_bytes": 0, "used_debug_read_bytes": 0 } }
},
{
"operation": "create_debug_epoch",
"boundary": "control",
"request": { "type": "create_debug_epoch", "process": "process-one", "stopped_task": "task-one", "reason": "breakpoint" },
"response": { "type": "debug_epoch", "process": "process-one", "actor": "user-one", "epoch": 3, "command": "freeze", "affected_tasks": [], "all_stop_requested": true, "audit_event": { "tenant": "tenant-one", "project": "project-one", "process": "process-one", "task": "task-one", "actor": "user-one", "operation": "create_debug_epoch", "allowed": true, "reason": "authorized", "charged_debug_read_bytes": 0, "used_debug_read_bytes": 0 } }
},
{
"operation": "resume_debug_epoch",
"boundary": "control",
"request": { "type": "resume_debug_epoch", "process": "process-one", "epoch": 3 },
"response": { "type": "debug_epoch", "process": "process-one", "actor": "user-one", "epoch": 3, "command": "resume", "affected_tasks": [], "all_stop_requested": false, "audit_event": { "tenant": "tenant-one", "project": "project-one", "process": "process-one", "task": null, "actor": "user-one", "operation": "resume_debug_epoch", "allowed": true, "reason": "authorized", "charged_debug_read_bytes": 0, "used_debug_read_bytes": 0 } }
},
{
"operation": "inspect_debug_epoch",
"boundary": "control",
"request": { "type": "inspect_debug_epoch", "process": "process-one", "epoch": 3 },
"response": { "type": "debug_epoch_status", "process": "process-one", "actor": "user-one", "epoch": 3, "command": "freeze", "expected_tasks": [], "acknowledgements": [], "fully_frozen": false, "partially_frozen": true, "fully_resumed": false, "failed": false, "failure_messages": [], "audit_event": { "tenant": "tenant-one", "project": "project-one", "process": "process-one", "task": null, "actor": "user-one", "operation": "inspect_debug_epoch", "allowed": true, "reason": "authorized", "charged_debug_read_bytes": 0, "used_debug_read_bytes": 0 } }
},
{
"operation": "list_artifacts",
"boundary": "control",
"request": { "type": "list_artifacts", "process": "process-one", "cursor": "artifact:1", "limit": 50 },
"response": { "type": "artifacts", "artifacts": [], "next_cursor": null }
},
{
"operation": "get_artifact",
"boundary": "control",
"request": { "type": "get_artifact", "artifact": "artifact-one" },
"response": { "type": "artifact", "artifact": { "id": "artifact-one", "display_path": "/out/artifact-one", "display_name": "artifact-one", "process": "process-one", "producer_task": "task-one", "safe_node": "node-one", "digest": "sha256:cccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccc", "size_bytes": 5, "availability": "available", "downloadable_now": true, "retention_state": "node_retained", "explicit_storage": false, "order_cursor": "artifact:1" } }
},
{
"operation": "create_artifact_download_link",
"boundary": "control",
"request": { "type": "create_artifact_download_link", "artifact": "artifact-one", "max_bytes": 1048576, "ttl_seconds": 120 },
"response": { "type": "artifact_download_link", "link": { "artifact": "artifact-one", "artifact_digest": "sha256:cccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccc", "artifact_size_bytes": 5, "source": { "RetainedNode": "node-one" }, "url_path": "/artifacts/tenant-one/project-one/process-one/artifact-one", "scoped_token_digest": "sha256:dddddddddddddddddddddddddddddddddddddddddddddddddddddddddddddddd", "expires_at_epoch_seconds": 1100, "tenant": "tenant-one", "project": "project-one", "process": "process-one", "actor": { "User": "user-one" }, "max_bytes": 1048576, "policy_context_digest": "sha256:eeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeee" } }
},
{
"operation": "open_artifact_download_stream",
"boundary": "control",
"request": { "type": "open_artifact_download_stream", "artifact": "artifact-one", "max_bytes": 1048576, "token_digest": "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", "chunk_bytes": 65536 },
"response": { "type": "artifact_download_stream", "link": { "artifact": "artifact-one", "artifact_digest": "sha256:cccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccc", "artifact_size_bytes": 5, "source": { "RetainedNode": "node-one" }, "url_path": "/artifacts/tenant-one/project-one/process-one/artifact-one", "scoped_token_digest": "sha256:dddddddddddddddddddddddddddddddddddddddddddddddddddddddddddddddd", "expires_at_epoch_seconds": 1100, "tenant": "tenant-one", "project": "project-one", "process": "process-one", "actor": { "User": "user-one" }, "max_bytes": 1048576, "policy_context_digest": "sha256:eeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeee" }, "streamed_bytes": 5, "charged_download_bytes": 5, "content_bytes_available": true, "content_offset": 0, "content_eof": true, "content_base64": "aGVsbG8=" }
},
{
"operation": "revoke_artifact_download_link",
"boundary": "control",
"request": { "type": "revoke_artifact_download_link", "artifact": "artifact-one", "token_digest": "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" },
"response": { "type": "artifact_download_link_revoked" }
}
]

View file

@ -61,21 +61,7 @@ impl ControlSession {
endpoint: &str,
api_path: &str,
) -> Result<Self, ControlTransportError> {
Self::connect_to_api_path_with_timeouts(
endpoint,
api_path,
Duration::from_secs(10),
Duration::from_secs(30),
)
}
pub fn connect_to_api_path_with_timeouts(
endpoint: &str,
api_path: &str,
connect_timeout: Duration,
io_timeout: Duration,
) -> Result<Self, ControlTransportError> {
let session = Self::connect_with_timeouts(endpoint, connect_timeout, io_timeout)?;
let session = Self::connect(endpoint)?;
#[cfg(not(target_arch = "wasm32"))]
{
let mut session = session;

View file

@ -585,22 +585,6 @@ impl Coordinator {
.count()
}
pub fn tenant_count(&self) -> usize {
self.durable.tenants.len()
}
pub fn user_count(&self) -> usize {
self.durable.users.len()
}
pub fn project_count(&self) -> usize {
self.durable.projects.len()
}
pub fn node_identity_count(&self) -> usize {
self.durable.node_identities.len()
}
pub fn node_identity(
&self,
tenant: &TenantId,

View file

@ -5,40 +5,17 @@ use clusterflux_core::{ProjectId, TenantId, UserId};
use serde_json::json;
fn main() -> Result<(), Box<dyn std::error::Error>> {
let raw_args = std::env::args().skip(1).collect::<Vec<_>>();
match raw_args.as_slice() {
[flag] if matches!(flag.as_str(), "--version" | "-V") => {
println!("clusterflux-coordinator {}", env!("CARGO_PKG_VERSION"));
return Ok(());
}
[flag] if matches!(flag.as_str(), "--help" | "-h") => {
println!(
"Clusterflux coordinator.\n\n\
Usage: clusterflux-coordinator [OPTIONS]\n\n\
Options:\n \
--listen <ADDRESS> [default: 127.0.0.1:0]\n \
--allow-local-trusted-loopback\n \
-h, --help\n \
-V, --version"
);
return Ok(());
}
_ => {}
}
let mut listen = "127.0.0.1:0".to_owned();
let mut allow_local_trusted = std::env::var("CLUSTERFLUX_ALLOW_LOCAL_TRUSTED_LOOPBACK")
.ok()
.as_deref()
== Some("1");
let mut args = raw_args.into_iter();
let mut args = std::env::args().skip(1);
while let Some(arg) = args.next() {
if arg == "--listen" {
listen = args.next().ok_or("--listen requires an address")?;
} else if arg == "--allow-local-trusted-loopback" {
allow_local_trusted = true;
} else {
return Err(format!("unknown argument: {arg}").into());
}
}

View file

@ -7,10 +7,9 @@ use std::collections::{BTreeMap, BTreeSet, VecDeque};
use std::time::{SystemTime, UNIX_EPOCH};
use clusterflux_core::{
Actor, AgentId, ApiError, ApiErrorCategory, ApiErrorCode, ArtifactRegistry,
CapabilityReportError, CredentialKind, Digest, DownloadError, LimitError, NativeQuicTransport,
NodeDescriptor, NodeId, PanelError, PanelState, Placement, ProcessId, ProjectId, RateLimit,
TenantId, TransportError, UserId,
Actor, AgentId, ArtifactRegistry, CapabilityReportError, CredentialKind, Digest, LimitError,
NativeQuicTransport, NodeDescriptor, NodeId, PanelState, Placement, ProcessId, ProjectId,
RateLimit, TenantId, TransportError, UserId,
};
use thiserror::Error;
@ -35,7 +34,6 @@ mod quota;
mod relay;
mod routing;
mod signed_nodes;
mod summaries;
mod tcp;
mod wire_protocol;
use authorization::authorize_authenticated_user_operation;
@ -45,14 +43,12 @@ use keys::{
ProcessControlKey, TaskAssignmentKey, TaskControlKey, TaskRestartKey,
};
pub use protocol::{
ArtifactAvailability, ArtifactRetentionState, ArtifactSummary, ArtifactTransferAssignment,
AuthenticatedCoordinatorRequest, CoordinatorRequest, CoordinatorResponse,
DebugAcknowledgementState, DebugAuditEvent, DebugEpochSummary, DebugParticipantAcknowledgement,
NodeSummary, ProcessActivityState, ProcessFinalResult, ProcessLifecycleState, ProcessSummary,
RecentLogEntry, SourcePreparationDisposition, SourcePreparationStatus, TaskAssignment,
TaskAttemptSnapshot, TaskAttemptState, TaskCancellationTarget, TaskCompletionEvent,
TaskExecutor, TaskFailureResolution, TaskLogStream, TaskReplacementBundle, TaskTerminalState,
VirtualProcessStatus, WorkflowActor,
ArtifactTransferAssignment, AuthenticatedCoordinatorRequest, CoordinatorRequest,
CoordinatorResponse, DebugAcknowledgementState, DebugAuditEvent,
DebugParticipantAcknowledgement, SourcePreparationDisposition, SourcePreparationStatus,
TaskAssignment, TaskAttemptSnapshot, TaskAttemptState, TaskCancellationTarget,
TaskCompletionEvent, TaskExecutor, TaskFailureResolution, TaskReplacementBundle,
TaskTerminalState, VirtualProcessStatus, WorkflowActor,
};
pub use quota::CoordinatorQuotaConfiguration;
pub use relay::{
@ -73,15 +69,9 @@ const MAX_RESTART_CHECKPOINTS_PER_PROCESS: usize = 128;
const MAX_TASK_EVENTS_TOTAL: usize = 8_192;
const MAX_DEBUG_AUDIT_EVENTS_TOTAL: usize = 8_192;
const MAX_RESTART_CHECKPOINTS_TOTAL: usize = 4_096;
const MAX_TASK_ATTEMPT_HISTORIES: usize = 1_000_000;
const MAX_TASK_ATTEMPT_HISTORIES: usize = 4_096;
const MAX_IN_FLIGHT_TASKS_PER_PROCESS: usize = 256;
const MAX_NODE_REPORTED_OBJECTS_PER_KIND: usize = 1_024;
const MAX_RECENT_LOG_ENTRIES_PER_PROCESS: usize = 256;
const MAX_RECENT_LOG_ENTRIES_PER_PROJECT: usize = 1_024;
const MAX_RECENT_LOG_BYTES_PER_PROJECT: usize = 512 * 1024;
const MAX_RECENT_LOG_CHUNK_BYTES: usize = 16 * 1024;
const MAX_RECENT_PROCESS_SUMMARIES_PER_PROJECT: usize = 32;
const MAX_RECENT_PROCESS_SUMMARIES_TOTAL: usize = 8_192;
const DEFAULT_NODE_STALE_AFTER_SECONDS: u64 = 30;
fn bounded_ttl(requested: u64, maximum: u64) -> u64 {
requested.clamp(1, maximum)
@ -121,24 +111,6 @@ pub struct CoordinatorAdmission {
pub max_artifact_download_ttl_seconds: u64,
}
#[derive(Clone, Debug, Default, PartialEq, Eq)]
pub struct CoordinatorOperationalMetrics {
pub tenants: usize,
pub users: usize,
pub projects: usize,
pub enrolled_nodes: usize,
pub reported_nodes: usize,
pub live_nodes: usize,
pub active_processes: usize,
pub active_coordinator_mains: usize,
pub max_active_coordinator_mains: usize,
pub active_tasks: usize,
pub queued_tasks: usize,
pub artifacts: usize,
pub retained_download_links: usize,
pub relay: ArtifactRelayUsage,
}
impl Default for CoordinatorAdmission {
fn default() -> Self {
Self {
@ -179,98 +151,6 @@ pub enum CoordinatorServiceError {
Durable(String),
}
impl CoordinatorServiceError {
pub fn api_error(&self, request_id: impl Into<String>) -> ApiError {
let request_id = request_id.into();
let message = self.to_string();
let (code, category, retryable) = match self {
Self::Io(_) => (
ApiErrorCode::TemporaryCapacity,
ApiErrorCategory::Availability,
true,
),
Self::Json(_)
| Self::CapabilityReport(_)
| Self::InvalidArtifactPath(_)
| Self::InvalidTaskLogTail(_) => (
ApiErrorCode::ValidationError,
ApiErrorCategory::Validation,
false,
),
Self::Protocol(_) | Self::Coordinator(CoordinatorError::Unauthorized(_)) => {
return ApiError::from_message(request_id, message);
}
Self::Coordinator(CoordinatorError::UnknownNode) => {
(ApiErrorCode::NotFound, ApiErrorCategory::State, false)
}
Self::Coordinator(CoordinatorError::Enrollment(_)) => (
ApiErrorCode::Unauthenticated,
ApiErrorCategory::Authentication,
false,
),
Self::Coordinator(CoordinatorError::StaleProcessEpoch { .. }) => {
(ApiErrorCode::Conflict, ApiErrorCategory::State, true)
}
Self::Download(DownloadError::NotFound) => {
(ApiErrorCode::NotFound, ApiErrorCategory::State, false)
}
Self::Download(DownloadError::Unavailable)
| Self::Download(DownloadError::DirectConnectivityUnavailable(_)) => (
ApiErrorCode::ArtifactUnavailable,
ApiErrorCategory::Availability,
true,
),
Self::Download(DownloadError::LimitExceeded { .. }) => (
ApiErrorCode::ArtifactLimitExceeded,
ApiErrorCategory::Resource,
false,
),
Self::Download(DownloadError::Unauthorized(_))
| Self::Download(DownloadError::InvalidToken)
| Self::Download(DownloadError::Expired)
| Self::Download(DownloadError::Revoked) => (
ApiErrorCode::Forbidden,
ApiErrorCategory::Authorization,
false,
),
Self::Download(DownloadError::Usage(_)) | Self::Resource(_) => (
ApiErrorCode::QuotaExceeded,
ApiErrorCategory::Resource,
true,
),
Self::Scheduler(_) => (
ApiErrorCode::NoCapableNode,
ApiErrorCategory::Availability,
true,
),
Self::Transport(_) => (
ApiErrorCode::NodeOffline,
ApiErrorCategory::Availability,
true,
),
Self::Panel(PanelError::RateLimited) => (
ApiErrorCode::QuotaExceeded,
ApiErrorCategory::Resource,
true,
),
Self::Panel(PanelError::UnknownWidget(_)) => {
(ApiErrorCode::NotFound, ApiErrorCategory::State, false)
}
Self::Panel(_) => (
ApiErrorCode::Forbidden,
ApiErrorCategory::Authorization,
false,
),
Self::Durable(_) => (
ApiErrorCode::InternalError,
ApiErrorCategory::Internal,
true,
),
};
ApiError::new(code, category, message, retryable, request_id)
}
}
pub struct CoordinatorService {
coordinator: Coordinator,
store: RuntimeDurableStore,
@ -281,37 +161,6 @@ pub struct CoordinatorService {
enrollment_grants: BTreeMap<EnrollmentGrantKey, clusterflux_core::EnrollmentGrant>,
task_events: VecDeque<TaskCompletionEvent>,
process_scope_history: VecDeque<ProcessControlKey>,
process_summaries: BTreeMap<ProcessControlKey, summaries::StoredProcessSummary>,
process_summary_order: VecDeque<ProcessControlKey>,
next_process_summary_order: u64,
task_terminal_states: BTreeMap<TaskRestartKey, TaskTerminalState>,
recent_logs: BTreeMap<(TenantId, ProjectId), VecDeque<RecentLogEntry>>,
recent_log_dropped_through: BTreeMap<ProcessControlKey, u64>,
recent_log_accounted_bytes: BTreeMap<
(
TenantId,
ProjectId,
ProcessId,
clusterflux_core::TaskInstanceId,
String,
),
u64,
>,
recent_log_truncated_streams: BTreeSet<(
TenantId,
ProjectId,
ProcessId,
clusterflux_core::TaskInstanceId,
String,
)>,
recent_log_quota_truncated_streams: BTreeSet<(
TenantId,
ProjectId,
ProcessId,
clusterflux_core::TaskInstanceId,
String,
)>,
next_recent_log_sequence: u64,
debug_audit_events: VecDeque<DebugAuditEvent>,
debug_epochs: BTreeMap<ProcessControlKey, u64>,
debug_epoch_runtime: BTreeMap<ProcessControlKey, debug::DebugEpochRuntime>,
@ -366,29 +215,6 @@ impl CoordinatorService {
self.artifact_relay.usage()
}
pub fn operational_metrics(&self) -> CoordinatorOperationalMetrics {
CoordinatorOperationalMetrics {
tenants: self.coordinator.tenant_count(),
users: self.coordinator.user_count(),
projects: self.coordinator.project_count(),
enrolled_nodes: self.coordinator.node_identity_count(),
reported_nodes: self.node_descriptors.len(),
live_nodes: self
.node_descriptors
.keys()
.filter(|scope| self.node_is_live(scope))
.count(),
active_processes: self.coordinator.active_process_count(),
active_coordinator_mains: self.main_runtime.active_main_count(),
max_active_coordinator_mains: self.main_runtime.max_active_mains(),
active_tasks: self.active_tasks.len(),
queued_tasks: self.pending_task_launches.len(),
artifacts: self.artifact_registry.artifact_count(),
retained_download_links: self.artifact_registry.retained_download_link_count(),
relay: self.artifact_relay.usage(),
}
}
fn commit_artifact_relay(
&mut self,
candidate: relay::ArtifactRelayLedger,
@ -578,16 +404,6 @@ impl CoordinatorService {
enrollment_grants: BTreeMap::new(),
task_events: VecDeque::new(),
process_scope_history: VecDeque::new(),
process_summaries: BTreeMap::new(),
process_summary_order: VecDeque::new(),
next_process_summary_order: 1,
task_terminal_states: BTreeMap::new(),
recent_logs: BTreeMap::new(),
recent_log_dropped_through: BTreeMap::new(),
recent_log_accounted_bytes: BTreeMap::new(),
recent_log_truncated_streams: BTreeSet::new(),
recent_log_quota_truncated_streams: BTreeSet::new(),
next_recent_log_sequence: 1,
debug_audit_events: VecDeque::new(),
debug_epochs: BTreeMap::new(),
debug_epoch_runtime: BTreeMap::new(),

View file

@ -17,7 +17,6 @@ pub(super) enum PublicUserOperation {
RevokeAgentPublicKey,
CreateNodeEnrollmentGrant,
ListNodeDescriptors,
ListNodeSummaries,
RevokeNodeCredential,
StartProcess,
ScheduleTask,
@ -25,7 +24,6 @@ pub(super) enum PublicUserOperation {
CancelProcess,
AbortProcess,
ListProcesses,
ListProcessSummaries,
QuotaStatus,
RestartTask,
ResolveTaskFailure,
@ -37,10 +35,7 @@ pub(super) enum PublicUserOperation {
InspectDebugEpoch,
ListTaskEvents,
ListTaskSnapshots,
ListRecentLogs,
JoinTask,
ListArtifacts,
GetArtifact,
CreateArtifactDownloadLink,
OpenArtifactDownloadStream,
RevokeArtifactDownloadLink,
@ -61,7 +56,6 @@ impl PublicUserOperation {
Self::RevokeAgentPublicKey => "revoke_agent_public_key",
Self::CreateNodeEnrollmentGrant => "create_node_enrollment_grant",
Self::ListNodeDescriptors => "list_node_descriptors",
Self::ListNodeSummaries => "list_node_summaries",
Self::RevokeNodeCredential => "revoke_node_credential",
Self::StartProcess => "start_process",
Self::ScheduleTask => "schedule_task",
@ -69,7 +63,6 @@ impl PublicUserOperation {
Self::CancelProcess => "cancel_process",
Self::AbortProcess => "abort_process",
Self::ListProcesses => "list_processes",
Self::ListProcessSummaries => "list_process_summaries",
Self::QuotaStatus => "quota_status",
Self::RestartTask => "restart_task",
Self::ResolveTaskFailure => "resolve_task_failure",
@ -81,10 +74,7 @@ impl PublicUserOperation {
Self::InspectDebugEpoch => "inspect_debug_epoch",
Self::ListTaskEvents => "list_task_events",
Self::ListTaskSnapshots => "list_task_snapshots",
Self::ListRecentLogs => "list_recent_logs",
Self::JoinTask => "join_task",
Self::ListArtifacts => "list_artifacts",
Self::GetArtifact => "get_artifact",
Self::CreateArtifactDownloadLink => "create_artifact_download_link",
Self::OpenArtifactDownloadStream => "open_artifact_download_stream",
Self::RevokeArtifactDownloadLink => "revoke_artifact_download_link",
@ -115,7 +105,6 @@ impl From<&AuthenticatedCoordinatorRequest> for PublicUserOperation {
Self::CreateNodeEnrollmentGrant
}
AuthenticatedCoordinatorRequest::ListNodeDescriptors => Self::ListNodeDescriptors,
AuthenticatedCoordinatorRequest::ListNodeSummaries { .. } => Self::ListNodeSummaries,
AuthenticatedCoordinatorRequest::RevokeNodeCredential { .. } => {
Self::RevokeNodeCredential
}
@ -125,9 +114,6 @@ impl From<&AuthenticatedCoordinatorRequest> for PublicUserOperation {
AuthenticatedCoordinatorRequest::CancelProcess { .. } => Self::CancelProcess,
AuthenticatedCoordinatorRequest::AbortProcess { .. } => Self::AbortProcess,
AuthenticatedCoordinatorRequest::ListProcesses => Self::ListProcesses,
AuthenticatedCoordinatorRequest::ListProcessSummaries { .. } => {
Self::ListProcessSummaries
}
AuthenticatedCoordinatorRequest::QuotaStatus => Self::QuotaStatus,
AuthenticatedCoordinatorRequest::RestartTask { .. } => Self::RestartTask,
AuthenticatedCoordinatorRequest::ResolveTaskFailure { .. } => Self::ResolveTaskFailure,
@ -143,10 +129,7 @@ impl From<&AuthenticatedCoordinatorRequest> for PublicUserOperation {
AuthenticatedCoordinatorRequest::InspectDebugEpoch { .. } => Self::InspectDebugEpoch,
AuthenticatedCoordinatorRequest::ListTaskEvents { .. } => Self::ListTaskEvents,
AuthenticatedCoordinatorRequest::ListTaskSnapshots { .. } => Self::ListTaskSnapshots,
AuthenticatedCoordinatorRequest::ListRecentLogs { .. } => Self::ListRecentLogs,
AuthenticatedCoordinatorRequest::JoinTask { .. } => Self::JoinTask,
AuthenticatedCoordinatorRequest::ListArtifacts { .. } => Self::ListArtifacts,
AuthenticatedCoordinatorRequest::GetArtifact { .. } => Self::GetArtifact,
AuthenticatedCoordinatorRequest::CreateArtifactDownloadLink { .. } => {
Self::CreateArtifactDownloadLink
}

View file

@ -9,10 +9,7 @@ use super::keys::{process_control_key, task_control_key, task_restart_key};
use super::protocol::TaskAttemptState;
use super::{
artifact_id_from_path, CoordinatorResponse, CoordinatorService, CoordinatorServiceError,
RecentLogEntry, TaskCompletionEvent, TaskLogStream, TaskTerminalState,
MAX_RECENT_LOG_BYTES_PER_PROJECT, MAX_RECENT_LOG_CHUNK_BYTES,
MAX_RECENT_LOG_ENTRIES_PER_PROCESS, MAX_RECENT_LOG_ENTRIES_PER_PROJECT,
MAX_TASK_LOG_TAIL_BYTES,
TaskCompletionEvent, TaskTerminalState, MAX_TASK_LOG_TAIL_BYTES,
};
impl CoordinatorService {
@ -39,44 +36,23 @@ impl CoordinatorService {
self.authorize_node_for_process_or_termination(&node, &tenant, &project, &process)?;
validate_task_log_tail("stdout_tail", &stdout_tail)?;
validate_task_log_tail("stderr_tail", &stderr_tail)?;
let reported_bytes = checked_reported_log_bytes(stdout_bytes, stderr_bytes)?;
let now_epoch_seconds = self.current_epoch_seconds()?;
let stdout_retained = self.accept_final_log_stream(
&tenant,
&project,
&process,
&task,
TaskLogStream::Stdout,
stdout_bytes,
&stdout_tail,
stdout_truncated,
now_epoch_seconds,
)?;
let stderr_retained = self.accept_final_log_stream(
&tenant,
&project,
&process,
&task,
TaskLogStream::Stderr,
stderr_bytes,
&stderr_tail,
stderr_truncated,
now_epoch_seconds,
)?;
self.quota
.can_charge_log_bytes(&tenant, &project, reported_bytes, now_epoch_seconds)?;
self.quota
.charge_log_bytes(&tenant, &project, reported_bytes, now_epoch_seconds)?;
Ok(CoordinatorResponse::TaskLogRecorded {
process,
task,
stdout_bytes,
stderr_bytes,
stdout_tail: if !stdout_retained {
"[log output truncated at project log quota]".to_owned()
} else if stdout_truncated {
stdout_tail: if stdout_truncated {
format!("{stdout_tail}\n... truncated")
} else {
stdout_tail
},
stderr_tail: if !stderr_retained {
"[log output truncated at project log quota]".to_owned()
} else if stderr_truncated {
stderr_tail: if stderr_truncated {
format!("{stderr_tail}\n... truncated")
} else {
stderr_tail
@ -85,158 +61,6 @@ impl CoordinatorService {
})
}
#[allow(clippy::too_many_arguments)]
pub(super) fn handle_report_task_log_chunk(
&mut self,
tenant: String,
project: String,
process: String,
node: String,
task: String,
stream: TaskLogStream,
offset: u64,
source_bytes: u64,
text: String,
truncated: bool,
) -> Result<CoordinatorResponse, CoordinatorServiceError> {
if text.len() > MAX_RECENT_LOG_CHUNK_BYTES {
return Err(CoordinatorServiceError::InvalidTaskLogTail(format!(
"live log chunk is {} bytes; max is {MAX_RECENT_LOG_CHUNK_BYTES}",
text.len()
)));
}
if source_bytes == 0 && !text.is_empty() && !truncated {
return Err(CoordinatorServiceError::Protocol(
"live log chunk source_bytes must describe non-empty text".to_owned(),
));
}
if source_bytes > (MAX_RECENT_LOG_CHUNK_BYTES as u64).saturating_mul(4) {
return Err(CoordinatorServiceError::Protocol(
"live log chunk source_bytes exceeds the bounded chunk allowance".to_owned(),
));
}
let tenant = TenantId::new(tenant);
let project = ProjectId::new(project);
let process = ProcessId::new(process);
let node = NodeId::new(node);
let task = TaskInstanceId::new(task);
self.authorize_node_for_process_or_termination(&node, &tenant, &project, &process)?;
let key = recent_log_offset_key(&tenant, &project, &process, &task, &stream);
let expected = self
.recent_log_accounted_bytes
.get(&key)
.copied()
.unwrap_or(0);
let end = offset.checked_add(source_bytes).ok_or_else(|| {
CoordinatorServiceError::Protocol(
"live log chunk offset exceeds the supported range".to_owned(),
)
})?;
let state_marker = source_bytes == 0 && truncated;
if end < expected || (end == expected && !state_marker) {
return Ok(CoordinatorResponse::TaskLogChunkRecorded {
process,
task,
sequence: None,
next_offset: expected,
});
}
let now_epoch_seconds = self.current_epoch_seconds()?;
if self.recent_log_quota_truncated_streams.contains(&key) {
if end > expected {
self.recent_log_accounted_bytes.insert(key, end);
}
return Ok(CoordinatorResponse::TaskLogChunkRecorded {
process,
task,
sequence: None,
next_offset: end.max(expected),
});
}
let newly_accounted = end.saturating_sub(expected);
if self
.quota
.charge_log_bytes(&tenant, &project, newly_accounted, now_epoch_seconds)
.is_err()
{
if end > expected {
self.recent_log_accounted_bytes.insert(key.clone(), end);
}
let sequence = self.mark_log_quota_truncated(
&tenant,
&project,
&process,
&task,
&stream,
now_epoch_seconds,
);
return Ok(CoordinatorResponse::TaskLogChunkRecorded {
process,
task,
sequence,
next_offset: end.max(expected),
});
}
if offset > expected {
self.record_recent_log(
tenant.clone(),
project.clone(),
process.clone(),
task.clone(),
stream.clone(),
format!("[log output lost: {} bytes]", offset - expected),
true,
now_epoch_seconds,
);
} else if offset < expected {
self.record_recent_log(
tenant.clone(),
project.clone(),
process.clone(),
task.clone(),
stream.clone(),
format!(
"[log output overlap omitted: {} new source bytes]",
end - expected
),
true,
now_epoch_seconds,
);
}
let marker_is_new = if truncated {
self.recent_log_truncated_streams.insert(key.clone())
} else {
false
};
let text = if state_marker && text.is_empty() {
"[log output truncated at source]".to_owned()
} else {
text
};
let sequence = (!text.is_empty() && offset >= expected && (!state_marker || marker_is_new))
.then(|| {
self.record_recent_log(
tenant.clone(),
project.clone(),
process.clone(),
task.clone(),
stream,
text,
truncated,
now_epoch_seconds,
)
});
if end > expected {
self.recent_log_accounted_bytes.insert(key, end);
}
Ok(CoordinatorResponse::TaskLogChunkRecorded {
process,
task,
sequence,
next_offset: end,
})
}
pub(super) fn handle_report_vfs_metadata(
&mut self,
tenant: String,
@ -355,37 +179,20 @@ impl CoordinatorService {
artifact_size_bytes,
result,
};
let reported_bytes = checked_reported_log_bytes(event.stdout_bytes, event.stderr_bytes)?;
let now_epoch_seconds = self.current_epoch_seconds()?;
let stdout_retained = self.accept_final_log_stream(
self.quota.can_charge_log_bytes(
&event.tenant,
&event.project,
&event.process,
&event.task,
TaskLogStream::Stdout,
event.stdout_bytes,
&event.stdout_tail,
event.stdout_truncated,
reported_bytes,
now_epoch_seconds,
)?;
let stderr_retained = self.accept_final_log_stream(
self.quota.charge_log_bytes(
&event.tenant,
&event.project,
&event.process,
&event.task,
TaskLogStream::Stderr,
event.stderr_bytes,
&event.stderr_tail,
event.stderr_truncated,
reported_bytes,
now_epoch_seconds,
)?;
if !stdout_retained {
event.stdout_tail = "[log output truncated at project log quota]".to_owned();
event.stdout_truncated = true;
}
if !stderr_retained {
event.stderr_tail = "[log output truncated at project log quota]".to_owned();
event.stderr_truncated = true;
}
let task_key = task_control_key(
&event.tenant,
&event.project,
@ -414,12 +221,6 @@ impl CoordinatorService {
self.task_aborts.remove(&task_key);
self.debug_commands.remove(&task_key);
self.active_tasks.remove(&task_key);
self.clear_recent_log_offsets_for_task(
&event.tenant,
&event.project,
&event.process,
&event.task,
);
self.task_assignments.retain(|_, assignments| {
assignments.retain(|assignment| {
assignment.tenant != event.tenant
@ -519,7 +320,7 @@ impl CoordinatorService {
Ok(CoordinatorResponse::TaskSnapshots { snapshots })
}
pub(super) fn authorize_task_event_process_scope(
fn authorize_task_event_process_scope(
&self,
tenant: &TenantId,
project: &ProjectId,
@ -559,47 +360,6 @@ impl CoordinatorService {
Ok(())
}
pub(super) fn handle_list_recent_logs(
&mut self,
tenant: String,
project: String,
actor_user: String,
process: String,
task: Option<String>,
after_sequence: Option<u64>,
limit: u32,
) -> Result<CoordinatorResponse, CoordinatorServiceError> {
let tenant = TenantId::new(tenant);
let project = ProjectId::new(project);
let _actor = UserId::new(actor_user);
let process = ProcessId::new(process);
let task = task.map(TaskInstanceId::new);
self.authorize_task_event_process_scope(&tenant, &project, &process)?;
let after_sequence = after_sequence.unwrap_or(0);
let retained = self.recent_logs.get(&(tenant.clone(), project.clone()));
let history_truncated = self
.recent_log_dropped_through
.get(&process_control_key(&tenant, &project, &process))
.is_some_and(|dropped_through| *dropped_through > after_sequence);
let entries = retained
.into_iter()
.flatten()
.filter(|entry| {
entry.process == process
&& entry.sequence > after_sequence
&& task.as_ref().is_none_or(|task| &entry.task == task)
})
.take(limit as usize)
.cloned()
.collect::<Vec<_>>();
let next_sequence = entries.last().map(|entry| entry.sequence);
Ok(CoordinatorResponse::RecentLogs {
entries,
next_sequence,
history_truncated,
})
}
pub(super) fn handle_join_task(
&mut self,
tenant: String,
@ -718,22 +478,6 @@ impl CoordinatorService {
pub(super) fn record_task_completion_event(&mut self, mut event: TaskCompletionEvent) {
event.stdout_tail = bounded_log_tail(event.stdout_tail, &mut event.stdout_truncated);
event.stderr_tail = bounded_log_tail(event.stderr_tail, &mut event.stderr_truncated);
match event.executor {
super::TaskExecutor::CoordinatorMain => self.record_main_terminal_state(
&event.tenant,
&event.project,
&event.process,
event.task_definition.clone(),
event.task.clone(),
event.terminal_state.clone(),
),
super::TaskExecutor::Node => {
self.task_terminal_states.insert(
task_restart_key(&event.tenant, &event.project, &event.process, &event.task),
event.terminal_state.clone(),
);
}
}
let process_scope = (
event.tenant.clone(),
event.project.clone(),
@ -771,321 +515,6 @@ impl CoordinatorService {
self.task_events.push_back(event);
}
#[allow(clippy::too_many_arguments)]
fn record_recent_log(
&mut self,
tenant: TenantId,
project: ProjectId,
process: ProcessId,
task: TaskInstanceId,
stream: TaskLogStream,
mut text: String,
mut truncated: bool,
server_timestamp_epoch_seconds: u64,
) -> u64 {
if text.len() > MAX_RECENT_LOG_CHUNK_BYTES {
let mut boundary = MAX_RECENT_LOG_CHUNK_BYTES;
while !text.is_char_boundary(boundary) {
boundary -= 1;
}
text.truncate(boundary);
truncated = true;
}
let sequence = self.next_recent_log_sequence;
self.next_recent_log_sequence = self.next_recent_log_sequence.saturating_add(1);
let logs = self
.recent_logs
.entry((tenant.clone(), project.clone()))
.or_default();
let mut dropped = Vec::new();
while logs.iter().filter(|entry| entry.process == process).count()
>= MAX_RECENT_LOG_ENTRIES_PER_PROCESS
{
let Some(index) = logs.iter().position(|entry| entry.process == process) else {
break;
};
if let Some(entry) = logs.remove(index) {
dropped.push(entry);
}
}
while logs.len() >= MAX_RECENT_LOG_ENTRIES_PER_PROJECT
|| logs
.iter()
.map(|entry| entry.text.len())
.sum::<usize>()
.saturating_add(text.len())
> MAX_RECENT_LOG_BYTES_PER_PROJECT
{
match logs.pop_front() {
Some(entry) => dropped.push(entry),
None => break,
}
}
logs.push_back(RecentLogEntry {
sequence,
process,
task,
stream,
text,
server_timestamp_epoch_seconds,
truncated,
});
for entry in dropped {
let key = process_control_key(&tenant, &project, &entry.process);
self.recent_log_dropped_through
.entry(key)
.and_modify(|dropped_through| {
*dropped_through = (*dropped_through).max(entry.sequence);
})
.or_insert(entry.sequence);
}
sequence
}
#[allow(clippy::too_many_arguments)]
fn accept_final_log_stream(
&mut self,
tenant: &TenantId,
project: &ProjectId,
process: &ProcessId,
task: &TaskInstanceId,
stream: TaskLogStream,
total_source_bytes: u64,
final_tail: &str,
source_truncated: bool,
now_epoch_seconds: u64,
) -> Result<bool, CoordinatorServiceError> {
let key = recent_log_offset_key(tenant, project, process, task, &stream);
let accounted = self
.recent_log_accounted_bytes
.get(&key)
.copied()
.unwrap_or(0);
let remaining = total_source_bytes.checked_sub(accounted).ok_or_else(|| {
let stream_name = match stream {
TaskLogStream::Stdout => "stdout",
TaskLogStream::Stderr => "stderr",
};
CoordinatorServiceError::Protocol(format!(
"final {stream_name} byte count {total_source_bytes} is below the {accounted} live bytes already accounted"
))
})?;
if self.recent_log_quota_truncated_streams.contains(&key) {
self.recent_log_accounted_bytes
.insert(key, total_source_bytes);
return Ok(false);
}
if self
.quota
.charge_log_bytes(tenant, project, remaining, now_epoch_seconds)
.is_err()
{
self.recent_log_accounted_bytes
.insert(key, total_source_bytes);
self.mark_log_quota_truncated(
tenant,
project,
process,
task,
&stream,
now_epoch_seconds,
);
return Ok(false);
}
self.reconcile_final_log_stream(
tenant,
project,
process,
task,
stream,
total_source_bytes,
final_tail,
source_truncated,
now_epoch_seconds,
);
Ok(true)
}
#[allow(clippy::too_many_arguments)]
fn mark_log_quota_truncated(
&mut self,
tenant: &TenantId,
project: &ProjectId,
process: &ProcessId,
task: &TaskInstanceId,
stream: &TaskLogStream,
now_epoch_seconds: u64,
) -> Option<u64> {
let key = recent_log_offset_key(tenant, project, process, task, stream);
if !self.recent_log_quota_truncated_streams.insert(key.clone()) {
return None;
}
self.recent_log_truncated_streams.insert(key);
Some(self.record_recent_log(
tenant.clone(),
project.clone(),
process.clone(),
task.clone(),
stream.clone(),
"[log output truncated at project log quota]".to_owned(),
true,
now_epoch_seconds,
))
}
#[allow(clippy::too_many_arguments)]
fn reconcile_final_log_stream(
&mut self,
tenant: &TenantId,
project: &ProjectId,
process: &ProcessId,
task: &TaskInstanceId,
stream: TaskLogStream,
total_source_bytes: u64,
final_tail: &str,
source_truncated: bool,
now_epoch_seconds: u64,
) {
let key = recent_log_offset_key(tenant, project, process, task, &stream);
let accounted = self
.recent_log_accounted_bytes
.get(&key)
.copied()
.unwrap_or(0);
let mut visible_truncation = false;
if total_source_bytes > accounted {
let missing = total_source_bytes - accounted;
if final_tail.is_empty() {
self.record_recent_log(
tenant.clone(),
project.clone(),
process.clone(),
task.clone(),
stream.clone(),
format!("[log output unavailable: {missing} source bytes]"),
true,
now_epoch_seconds,
);
visible_truncation = true;
} else if (final_tail.len() as u64) <= total_source_bytes {
let tail_source_start = total_source_bytes - final_tail.len() as u64;
if accounted < tail_source_start {
self.record_recent_log(
tenant.clone(),
project.clone(),
process.clone(),
task.clone(),
stream.clone(),
format!(
"[log output lost before final tail: {} source bytes]",
tail_source_start - accounted
),
true,
now_epoch_seconds,
);
visible_truncation = true;
}
let source_start = accounted.max(tail_source_start) - tail_source_start;
let mut byte_start = usize::try_from(source_start)
.unwrap_or(final_tail.len())
.min(final_tail.len());
while byte_start < final_tail.len() && !final_tail.is_char_boundary(byte_start) {
byte_start += 1;
}
let suffix = &final_tail[byte_start..];
if !suffix.is_empty() {
self.record_recent_log(
tenant.clone(),
project.clone(),
process.clone(),
task.clone(),
stream.clone(),
suffix.to_owned(),
source_truncated || visible_truncation,
now_epoch_seconds,
);
visible_truncation |= source_truncated;
}
} else if accounted == 0 {
self.record_recent_log(
tenant.clone(),
project.clone(),
process.clone(),
task.clone(),
stream.clone(),
final_tail.to_owned(),
source_truncated,
now_epoch_seconds,
);
visible_truncation |= source_truncated;
} else {
self.record_recent_log(
tenant.clone(),
project.clone(),
process.clone(),
task.clone(),
stream.clone(),
format!(
"[{missing} additional source bytes could not be merged without duplicating redacted output]"
),
true,
now_epoch_seconds,
);
visible_truncation = true;
}
self.recent_log_accounted_bytes
.insert(key.clone(), total_source_bytes);
}
let marker_is_new = (source_truncated || visible_truncation)
&& self.recent_log_truncated_streams.insert(key);
if marker_is_new && !visible_truncation {
self.record_recent_log(
tenant.clone(),
project.clone(),
process.clone(),
task.clone(),
stream,
"[log output truncated at source]".to_owned(),
true,
now_epoch_seconds,
);
}
}
pub(super) fn clear_recent_log_offsets_for_task(
&mut self,
tenant: &TenantId,
project: &ProjectId,
process: &ProcessId,
task: &TaskInstanceId,
) {
self.recent_log_accounted_bytes.retain(
|(entry_tenant, entry_project, entry_process, entry_task, _), _| {
entry_tenant != tenant
|| entry_project != project
|| entry_process != process
|| entry_task != task
},
);
self.recent_log_truncated_streams.retain(
|(entry_tenant, entry_project, entry_process, entry_task, _)| {
entry_tenant != tenant
|| entry_project != project
|| entry_process != process
|| entry_task != task
},
);
self.recent_log_quota_truncated_streams.retain(
|(entry_tenant, entry_project, entry_process, entry_task, _)| {
entry_tenant != tenant
|| entry_project != project
|| entry_process != process
|| entry_task != task
},
);
}
fn finish_task_attempt(&mut self, event: &mut TaskCompletionEvent) -> bool {
let key = task_restart_key(&event.tenant, &event.project, &event.process, &event.task);
let Some(attempt) = self
@ -1169,11 +598,13 @@ impl CoordinatorService {
return Ok(false);
}
let main_completed = self
.process_summaries
.get(&process_key)
.and_then(|summary| summary.main_terminal_state.as_ref())
.is_some_and(|state| matches!(state, TaskTerminalState::Completed));
let main_completed = self.task_events.iter().rev().any(|event| {
&event.tenant == tenant
&& &event.project == project
&& &event.process == process
&& matches!(event.executor, super::TaskExecutor::CoordinatorMain)
&& matches!(event.terminal_state, TaskTerminalState::Completed)
});
let cancellation_completed = self.process_cancellations.contains(&process_key);
if !main_completed && !cancellation_completed {
return Ok(false);
@ -1188,36 +619,6 @@ impl CoordinatorService {
{
return Ok(false);
}
let final_result = if cancellation_completed {
super::ProcessFinalResult::Cancelled
} else if self.task_terminal_states.iter().any(
|((task_tenant, task_project, task_process, _), terminal_state)| {
task_tenant == tenant
&& task_project == project
&& task_process == process
&& terminal_state == &TaskTerminalState::Failed
},
) {
super::ProcessFinalResult::Failed
} else if self.task_terminal_states.iter().any(
|((task_tenant, task_project, task_process, _), terminal_state)| {
task_tenant == tenant
&& task_project == project
&& task_process == process
&& terminal_state == &TaskTerminalState::Cancelled
},
) {
super::ProcessFinalResult::Cancelled
} else {
super::ProcessFinalResult::Completed
};
self.record_process_terminal(
tenant,
project,
process,
final_result,
self.current_epoch_seconds()?,
);
self.coordinator.abort_process(tenant, project, process)?;
self.clear_debug_state_for_process(tenant, project, process);
self.clear_operator_panel_state(tenant, project, process);
@ -1324,24 +725,15 @@ impl CoordinatorService {
}
}
fn recent_log_offset_key(
tenant: &TenantId,
project: &ProjectId,
process: &ProcessId,
task: &TaskInstanceId,
stream: &TaskLogStream,
) -> (TenantId, ProjectId, ProcessId, TaskInstanceId, String) {
(
tenant.clone(),
project.clone(),
process.clone(),
task.clone(),
match stream {
TaskLogStream::Stdout => "stdout",
TaskLogStream::Stderr => "stderr",
}
.to_owned(),
fn checked_reported_log_bytes(
stdout_bytes: u64,
stderr_bytes: u64,
) -> Result<u64, CoordinatorServiceError> {
stdout_bytes.checked_add(stderr_bytes).ok_or_else(|| {
CoordinatorServiceError::Protocol(
"reported task log byte counts exceed the supported range".to_owned(),
)
})
}
fn validate_task_log_tail(kind: &str, value: &str) -> Result<(), CoordinatorServiceError> {
@ -1358,11 +750,11 @@ fn bounded_log_tail(mut value: String, truncated: &mut bool) -> String {
if value.len() <= MAX_TASK_LOG_TAIL_BYTES {
return value;
}
let mut boundary = value.len() - MAX_TASK_LOG_TAIL_BYTES;
while boundary < value.len() && !value.is_char_boundary(boundary) {
boundary += 1;
let mut boundary = MAX_TASK_LOG_TAIL_BYTES;
while !value.is_char_boundary(boundary) {
boundary -= 1;
}
value.drain(..boundary);
value.truncate(boundary);
*truncated = true;
value
}

View file

@ -109,14 +109,6 @@ impl Default for CoordinatorMainRuntime {
}
impl CoordinatorMainRuntime {
pub(super) fn active_main_count(&self) -> usize {
self.controls.len()
}
pub(super) fn max_active_mains(&self) -> usize {
self.max_active_mains
}
pub(super) fn configure(
&mut self,
configuration: super::CoordinatorMainRuntimeConfiguration,
@ -737,13 +729,6 @@ impl CoordinatorService {
&process,
"coordinator main launch failed admission or validation",
);
self.record_process_terminal(
&tenant,
&project,
&process,
super::ProcessFinalResult::Failed,
self.liveness_now_epoch_seconds(),
);
let _ = self.coordinator.abort_process(&tenant, &project, &process);
}
result
@ -1007,13 +992,6 @@ impl CoordinatorService {
}
}
self.process_aborts.insert(process_key.clone());
self.record_process_terminal(
&scope.tenant,
&scope.project,
&scope.process,
super::ProcessFinalResult::Failed,
self.liveness_now_epoch_seconds(),
);
let _ = self
.coordinator
.abort_process(&scope.tenant, &scope.project, &scope.process);

View file

@ -671,9 +671,7 @@ impl CoordinatorService {
));
};
self.task_attempts.remove(&removable);
self.task_terminal_states.remove(&removable);
}
self.task_terminal_states.remove(&key);
let attempts = self.task_attempts.entry(key).or_default();
for attempt in attempts.iter_mut() {
attempt.current = false;

View file

@ -382,10 +382,6 @@ impl CoordinatorService {
|| attempt_project != &project
|| attempt_process != &process
});
self.task_terminal_states
.retain(|(task_tenant, task_project, task_process, _), _| {
task_tenant != &tenant || task_project != &project || task_process != &process
});
self.restart_launches
.retain(|(attempt_tenant, attempt_project, attempt_process, _)| {
attempt_tenant != &tenant
@ -396,12 +392,11 @@ impl CoordinatorService {
event.tenant != tenant || event.project != project || event.process != process
});
let active = self.coordinator.start_process_for_launch_attempt(
tenant.clone(),
project.clone(),
tenant,
project,
process.clone(),
launch_attempt.map(clusterflux_core::LaunchAttemptId::new),
);
self.record_process_started(&tenant, &project, &process, now_epoch_seconds);
Ok(CoordinatorResponse::ProcessStarted {
process,
launch_attempt: active
@ -518,13 +513,6 @@ impl CoordinatorService {
}
let process_key = process_control_key(&tenant, &project, &process);
if cancelled_tasks.is_empty() && !self.main_runtime.controls.contains_key(&process_key) {
self.record_process_terminal(
&tenant,
&project,
&process,
super::ProcessFinalResult::Cancelled,
self.current_epoch_seconds()?,
);
self.coordinator
.abort_process(&tenant, &project, &process)?;
self.clear_operator_panel_state(&tenant, &project, &process);
@ -610,13 +598,6 @@ impl CoordinatorService {
}
}
self.record_process_terminal(
&tenant,
&project,
&process,
super::ProcessFinalResult::Cancelled,
self.current_epoch_seconds()?,
);
if let Some(launch_attempt) = launch_attempt.as_ref() {
self.coordinator.abort_process_for_launch_attempt(
&tenant,
@ -675,18 +656,10 @@ impl CoordinatorService {
.map(|active| {
let process_key = process_control_key(&active.tenant, &active.project, &active.id);
let main = self.main_runtime.controls.get(&process_key);
let stored = self.process_summaries.get(&process_key);
let stored_main_state = stored
.and_then(|summary| summary.main_terminal_state.as_ref())
.map(|state| match state {
super::TaskTerminalState::Completed => "completed",
super::TaskTerminalState::Failed => "failed",
super::TaskTerminalState::Cancelled => "cancelled",
});
let state = if self.process_cancellations.contains(&process_key) {
"cancelling"
} else {
"running"
main.map_or("running", |main| main.state.as_str())
};
let main_wait_state = main.and_then(|main| {
if main.state != "running" {
@ -711,15 +684,9 @@ impl CoordinatorService {
VirtualProcessStatus {
process: active.id,
state: state.to_owned(),
main_task_definition: main.map(|main| main.task_definition.clone()).or_else(
|| stored.and_then(|summary| summary.main_task_definition.clone()),
),
main_task_instance: main
.map(|main| main.task_instance.clone())
.or_else(|| stored.and_then(|summary| summary.main_task_instance.clone())),
main_state: main
.map(|main| main.state.clone())
.or_else(|| stored_main_state.map(str::to_owned)),
main_task_definition: main.map(|main| main.task_definition.clone()),
main_task_instance: main.map(|main| main.task_instance.clone()),
main_state: main.map(|main| main.state.clone()),
main_wait_state,
main_debug_epoch: main.and_then(|main| main.debug.requested_epoch()),
connected_nodes: active.connected_nodes.into_iter().collect(),

View file

@ -150,15 +150,6 @@ pub enum CoordinatorRequest {
project: String,
actor_user: String,
},
ListNodeSummaries {
tenant: String,
project: String,
actor_user: String,
#[serde(default)]
cursor: Option<String>,
#[serde(default = "default_page_limit")]
limit: u32,
},
RevokeNodeCredential {
tenant: String,
project: String,
@ -312,15 +303,6 @@ pub enum CoordinatorRequest {
project: String,
actor_user: String,
},
ListProcessSummaries {
tenant: String,
project: String,
actor_user: String,
#[serde(default)]
cursor: Option<String>,
#[serde(default = "default_page_limit")]
limit: u32,
},
QuotaStatus {
tenant: String,
project: String,
@ -447,19 +429,6 @@ pub enum CoordinatorRequest {
stderr_truncated: bool,
backpressured: bool,
},
ReportTaskLogChunk {
tenant: String,
project: String,
process: String,
node: String,
task: String,
stream: TaskLogStream,
offset: u64,
source_bytes: u64,
text: String,
#[serde(default)]
truncated: bool,
},
ReportVfsMetadata {
tenant: String,
project: String,
@ -509,18 +478,6 @@ pub enum CoordinatorRequest {
actor_user: String,
process: String,
},
ListRecentLogs {
tenant: String,
project: String,
actor_user: String,
process: String,
#[serde(default)]
task: Option<String>,
#[serde(default)]
after_sequence: Option<u64>,
#[serde(default = "default_log_page_limit")]
limit: u32,
},
JoinTask {
tenant: String,
project: String,
@ -553,23 +510,6 @@ pub enum CoordinatorRequest {
#[serde(default = "default_download_ttl_seconds")]
ttl_seconds: u64,
},
ListArtifacts {
tenant: String,
project: String,
actor_user: String,
#[serde(default)]
process: Option<String>,
#[serde(default)]
cursor: Option<String>,
#[serde(default = "default_page_limit")]
limit: u32,
},
GetArtifact {
tenant: String,
project: String,
actor_user: String,
artifact: String,
},
OpenArtifactDownloadStream {
tenant: String,
project: String,
@ -726,18 +666,6 @@ fn validate_coordinator_request(request: &CoordinatorRequest, path: &str) -> Res
validate_tenant_project(tenant, project, path)?;
validate_user(actor_user, &format!("{path}.actor_user"))
}
CoordinatorRequest::ListNodeSummaries {
tenant,
project,
actor_user,
cursor,
limit,
} => {
validate_tenant_project(tenant, project, path)?;
validate_user(actor_user, &format!("{path}.actor_user"))?;
validate_optional_cursor(cursor.as_deref(), &format!("{path}.cursor"))?;
validate_page_limit(*limit, &format!("{path}.limit"), 200)
}
CoordinatorRequest::ExchangeNodeEnrollmentGrant {
tenant,
project,
@ -973,14 +901,6 @@ fn validate_coordinator_request(request: &CoordinatorRequest, path: &str) -> Res
task,
..
}
| CoordinatorRequest::ReportTaskLogChunk {
tenant,
project,
process,
node,
task,
..
}
| CoordinatorRequest::ReportVfsMetadata {
tenant,
project,
@ -1059,23 +979,6 @@ fn validate_coordinator_request(request: &CoordinatorRequest, path: &str) -> Res
validate_user(actor_user, &format!("{path}.actor_user"))?;
validate_process(process, &format!("{path}.process"))
}
CoordinatorRequest::ListRecentLogs {
tenant,
project,
actor_user,
process,
task,
limit,
..
} => {
validate_tenant_project(tenant, project, path)?;
validate_user(actor_user, &format!("{path}.actor_user"))?;
validate_process(process, &format!("{path}.process"))?;
if let Some(task) = task {
validate_task_instance(task, &format!("{path}.task"))?;
}
validate_page_limit(*limit, &format!("{path}.limit"), 200)
}
CoordinatorRequest::AbortProcess {
tenant,
project,
@ -1104,18 +1007,6 @@ fn validate_coordinator_request(request: &CoordinatorRequest, path: &str) -> Res
validate_tenant_project(tenant, project, path)?;
validate_user(actor_user, &format!("{path}.actor_user"))
}
CoordinatorRequest::ListProcessSummaries {
tenant,
project,
actor_user,
cursor,
limit,
} => {
validate_tenant_project(tenant, project, path)?;
validate_user(actor_user, &format!("{path}.actor_user"))?;
validate_optional_cursor(cursor.as_deref(), &format!("{path}.cursor"))?;
validate_page_limit(*limit, &format!("{path}.limit"), 100)
}
CoordinatorRequest::RestartTask {
tenant,
project,
@ -1189,20 +1080,6 @@ fn validate_coordinator_request(request: &CoordinatorRequest, path: &str) -> Res
validate_process(process, &format!("{path}.process"))?;
validate_external_token(widget_id, &format!("{path}.widget_id"), 256)
}
CoordinatorRequest::ListArtifacts {
tenant,
project,
actor_user,
process,
cursor,
limit,
} => {
validate_tenant_project(tenant, project, path)?;
validate_user(actor_user, &format!("{path}.actor_user"))?;
validate_optional_process(process.as_deref(), &format!("{path}.process"))?;
validate_optional_cursor(cursor.as_deref(), &format!("{path}.cursor"))?;
validate_page_limit(*limit, &format!("{path}.limit"), 200)
}
CoordinatorRequest::CreateArtifactDownloadLink {
tenant,
project,
@ -1223,12 +1100,6 @@ fn validate_coordinator_request(request: &CoordinatorRequest, path: &str) -> Res
actor_user,
artifact,
..
}
| CoordinatorRequest::GetArtifact {
tenant,
project,
actor_user,
artifact,
} => {
validate_tenant_project(tenant, project, path)?;
validate_user(actor_user, &format!("{path}.actor_user"))?;
@ -1262,14 +1133,6 @@ fn validate_authenticated_request(
| AuthenticatedCoordinatorRequest::ListNodeDescriptors
| AuthenticatedCoordinatorRequest::ListProcesses
| AuthenticatedCoordinatorRequest::QuotaStatus => Ok(()),
AuthenticatedCoordinatorRequest::ListNodeSummaries { cursor, limit } => {
validate_optional_cursor(cursor.as_deref(), &format!("{path}.cursor"))?;
validate_page_limit(*limit, &format!("{path}.limit"), 200)
}
AuthenticatedCoordinatorRequest::ListProcessSummaries { cursor, limit } => {
validate_optional_cursor(cursor.as_deref(), &format!("{path}.cursor"))?;
validate_page_limit(*limit, &format!("{path}.limit"), 100)
}
AuthenticatedCoordinatorRequest::CreateProject { project, .. }
| AuthenticatedCoordinatorRequest::SelectProject { project } => {
validate_project(project, &format!("{path}.project"))
@ -1325,18 +1188,6 @@ fn validate_authenticated_request(
| AuthenticatedCoordinatorRequest::ListTaskSnapshots { process } => {
validate_process(process, &format!("{path}.process"))
}
AuthenticatedCoordinatorRequest::ListRecentLogs {
process,
task,
limit,
..
} => {
validate_process(process, &format!("{path}.process"))?;
if let Some(task) = task {
validate_task_instance(task, &format!("{path}.task"))?;
}
validate_page_limit(*limit, &format!("{path}.limit"), 200)
}
AuthenticatedCoordinatorRequest::RestartTask { process, task, .. }
| AuthenticatedCoordinatorRequest::ResolveTaskFailure { process, task, .. }
| AuthenticatedCoordinatorRequest::JoinTask { process, task } => {
@ -1354,19 +1205,9 @@ fn validate_authenticated_request(
AuthenticatedCoordinatorRequest::ListTaskEvents { process } => {
validate_optional_process(process.as_deref(), &format!("{path}.process"))
}
AuthenticatedCoordinatorRequest::ListArtifacts {
process,
cursor,
limit,
} => {
validate_optional_process(process.as_deref(), &format!("{path}.process"))?;
validate_optional_cursor(cursor.as_deref(), &format!("{path}.cursor"))?;
validate_page_limit(*limit, &format!("{path}.limit"), 200)
}
AuthenticatedCoordinatorRequest::CreateArtifactDownloadLink { artifact, .. }
| AuthenticatedCoordinatorRequest::OpenArtifactDownloadStream { artifact, .. }
| AuthenticatedCoordinatorRequest::RevokeArtifactDownloadLink { artifact, .. }
| AuthenticatedCoordinatorRequest::GetArtifact { artifact } => {
| AuthenticatedCoordinatorRequest::RevokeArtifactDownloadLink { artifact, .. } => {
validate_artifact(artifact, &format!("{path}.artifact"))
}
AuthenticatedCoordinatorRequest::ExportArtifactToNode {
@ -1521,19 +1362,6 @@ fn validate_optional_process(value: Option<&str>, path: &str) -> Result<(), Stri
value.map_or(Ok(()), |value| validate_process(value, path))
}
fn validate_optional_cursor(value: Option<&str>, path: &str) -> Result<(), String> {
value.map_or(Ok(()), |value| validate_external_token(value, path, 256))
}
fn validate_page_limit(value: u32, path: &str, maximum: u32) -> Result<(), String> {
if value == 0 || value > maximum {
return Err(format!(
"malformed pagination limit {path}: expected 1 through {maximum}, received {value}"
));
}
Ok(())
}
fn validate_optional_launch_attempt(value: Option<&str>, path: &str) -> Result<(), String> {
value.map_or(Ok(()), |value| validate_launch_attempt(value, path))
}
@ -1858,12 +1686,6 @@ pub enum AuthenticatedCoordinatorRequest {
ttl_seconds: u64,
},
ListNodeDescriptors,
ListNodeSummaries {
#[serde(default)]
cursor: Option<String>,
#[serde(default = "default_page_limit")]
limit: u32,
},
RevokeNodeCredential {
node: String,
},
@ -1900,12 +1722,6 @@ pub enum AuthenticatedCoordinatorRequest {
launch_attempt: Option<String>,
},
ListProcesses,
ListProcessSummaries {
#[serde(default)]
cursor: Option<String>,
#[serde(default = "default_page_limit")]
limit: u32,
},
QuotaStatus,
RestartTask {
process: String,
@ -1950,15 +1766,6 @@ pub enum AuthenticatedCoordinatorRequest {
ListTaskSnapshots {
process: String,
},
ListRecentLogs {
process: String,
#[serde(default)]
task: Option<String>,
#[serde(default)]
after_sequence: Option<u64>,
#[serde(default = "default_log_page_limit")]
limit: u32,
},
JoinTask {
process: String,
task: String,
@ -1969,17 +1776,6 @@ pub enum AuthenticatedCoordinatorRequest {
#[serde(default = "default_download_ttl_seconds")]
ttl_seconds: u64,
},
ListArtifacts {
#[serde(default)]
process: Option<String>,
#[serde(default)]
cursor: Option<String>,
#[serde(default = "default_page_limit")]
limit: u32,
},
GetArtifact {
artifact: String,
},
OpenArtifactDownloadStream {
artifact: String,
max_bytes: u64,
@ -2002,14 +1798,6 @@ fn default_download_ttl_seconds() -> u64 {
900
}
fn default_page_limit() -> u32 {
50
}
fn default_log_page_limit() -> u32 {
100
}
fn default_node_enrollment_ttl_seconds() -> u64 {
900
}

View file

@ -183,121 +183,6 @@ pub struct VirtualProcessStatus {
pub coordinator_epoch: u64,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub struct NodeSummary {
pub id: NodeId,
pub display_name: String,
pub online: bool,
pub stale: bool,
pub last_seen_epoch_seconds: Option<u64>,
pub capabilities: NodeCapabilities,
pub direct_connectivity: bool,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum ProcessLifecycleState {
Active,
RecentTerminal,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum ProcessActivityState {
Running,
WaitingForNode,
WaitingForTask,
AwaitingAction,
DebugEpochPartial,
Cancelling,
Completed,
Failed,
Cancelled,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum ProcessFinalResult {
Completed,
Failed,
Cancelled,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub struct DebugEpochSummary {
pub epoch: u64,
pub command: String,
pub fully_frozen: bool,
pub partially_frozen: bool,
pub fully_resumed: bool,
pub failed: bool,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub struct ProcessSummary {
pub process: ProcessId,
pub lifecycle: ProcessLifecycleState,
pub activity: ProcessActivityState,
pub main_wait_state: Option<String>,
pub started_at_epoch_seconds: u64,
pub ended_at_epoch_seconds: Option<u64>,
pub final_result: Option<ProcessFinalResult>,
pub connected_nodes: Vec<NodeId>,
pub current_debug_epoch: Option<DebugEpochSummary>,
pub order_cursor: String,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum ArtifactAvailability {
Available,
NodeOffline,
Unavailable,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum ArtifactRetentionState {
NodeRetained,
ExplicitStorage,
Lost,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub struct ArtifactSummary {
pub id: ArtifactId,
pub display_path: String,
pub display_name: String,
pub process: ProcessId,
pub producer_task: TaskInstanceId,
pub safe_node: Option<NodeId>,
pub digest: Digest,
pub size_bytes: u64,
pub availability: ArtifactAvailability,
pub downloadable_now: bool,
pub retention_state: ArtifactRetentionState,
pub explicit_storage: bool,
pub order_cursor: String,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum TaskLogStream {
Stdout,
Stderr,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub struct RecentLogEntry {
pub sequence: u64,
pub process: ProcessId,
pub task: TaskInstanceId,
pub stream: TaskLogStream,
pub text: String,
pub server_timestamp_epoch_seconds: u64,
pub truncated: bool,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub enum SourcePreparationDisposition {
Pending { reason: String },
@ -397,11 +282,6 @@ pub enum CoordinatorResponse {
descriptors: Vec<NodeDescriptor>,
actor: UserId,
},
NodeSummaries {
nodes: Vec<NodeSummary>,
next_cursor: Option<String>,
actor: UserId,
},
NodeCredentialRevoked {
node: NodeId,
tenant: TenantId,
@ -493,11 +373,6 @@ pub enum CoordinatorResponse {
processes: Vec<VirtualProcessStatus>,
actor: UserId,
},
ProcessSummaries {
processes: Vec<ProcessSummary>,
next_cursor: Option<String>,
actor: UserId,
},
QuotaStatus {
tenant: TenantId,
project: ProjectId,
@ -608,17 +483,6 @@ pub enum CoordinatorResponse {
stderr_tail: String,
backpressured: bool,
},
TaskLogChunkRecorded {
process: ProcessId,
task: TaskInstanceId,
sequence: Option<u64>,
next_offset: u64,
},
RecentLogs {
entries: Vec<RecentLogEntry>,
next_sequence: Option<u64>,
history_truncated: bool,
},
VfsMetadataRecorded {
process: ProcessId,
task: TaskInstanceId,
@ -655,13 +519,6 @@ pub enum CoordinatorResponse {
ArtifactDownloadLink {
link: DownloadLink,
},
Artifacts {
artifacts: Vec<ArtifactSummary>,
next_cursor: Option<String>,
},
Artifact {
artifact: ArtifactSummary,
},
ArtifactDownloadLinkRevoked {
link: DownloadLink,
},
@ -686,24 +543,6 @@ pub enum CoordinatorResponse {
artifact_size_bytes: u64,
},
Error {
#[serde(flatten)]
error: clusterflux_core::ApiError,
message: String,
},
}
impl CoordinatorResponse {
pub fn error(request_id: impl Into<String>, message: impl Into<String>) -> Self {
Self::Error {
error: clusterflux_core::ApiError::from_message(request_id, message),
}
}
pub fn service_error(
request_id: impl Into<String>,
error: &crate::service::CoordinatorServiceError,
) -> Self {
Self::Error {
error: error.api_error(request_id),
}
}
}

View file

@ -260,6 +260,22 @@ impl CoordinatorQuota {
self.charge(tenant, project, LimitKind::ApiCall, 1, now_epoch_seconds)
}
pub(super) fn can_charge_log_bytes(
&self,
tenant: &TenantId,
project: &ProjectId,
bytes: u64,
now_epoch_seconds: u64,
) -> Result<(), LimitError> {
self.can_charge(
tenant,
project,
LimitKind::LogBytes,
bytes,
now_epoch_seconds,
)
}
pub(super) fn charge_log_bytes(
&mut self,
tenant: &TenantId,

View file

@ -65,16 +65,6 @@ pub struct ArtifactRelayUsage {
pub egress_bytes: u64,
pub abandoned_or_failed_bytes: u64,
pub reserved_bytes: u64,
pub lifetime_ingress_bytes: u64,
pub lifetime_egress_bytes: u64,
pub lifetime_abandoned_or_failed_bytes: u64,
pub completed_transfers: u64,
pub failed_transfers: u64,
pub cancelled_transfers: u64,
pub expired_transfers: u64,
pub tracked_scopes: usize,
pub max_active_global: usize,
pub max_tracked_scopes: usize,
}
#[derive(Clone, Debug, Default, PartialEq, Eq, Serialize, Deserialize)]
@ -87,20 +77,6 @@ pub struct ArtifactRelayDurableState {
pub ingress_used: u64,
pub egress_used: u64,
pub abandoned_or_failed_used: u64,
#[serde(default)]
pub lifetime_ingress_bytes: u64,
#[serde(default)]
pub lifetime_egress_bytes: u64,
#[serde(default)]
pub lifetime_abandoned_or_failed_bytes: u64,
#[serde(default)]
pub completed_transfers: u64,
#[serde(default)]
pub failed_transfers: u64,
#[serde(default)]
pub cancelled_transfers: u64,
#[serde(default)]
pub expired_transfers: u64,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
@ -169,13 +145,6 @@ pub(super) struct ArtifactRelayLedger {
ingress_used: u64,
egress_used: u64,
abandoned_or_failed_used: u64,
lifetime_ingress_bytes: u64,
lifetime_egress_bytes: u64,
lifetime_abandoned_or_failed_bytes: u64,
completed_transfers: u64,
failed_transfers: u64,
cancelled_transfers: u64,
expired_transfers: u64,
}
impl Default for ArtifactRelayLedger {
@ -196,13 +165,6 @@ impl ArtifactRelayLedger {
ingress_used: 0,
egress_used: 0,
abandoned_or_failed_used: 0,
lifetime_ingress_bytes: 0,
lifetime_egress_bytes: 0,
lifetime_abandoned_or_failed_bytes: 0,
completed_transfers: 0,
failed_transfers: 0,
cancelled_transfers: 0,
expired_transfers: 0,
}
}
@ -256,13 +218,6 @@ impl ArtifactRelayLedger {
ingress_used: state.ingress_used,
egress_used: state.egress_used,
abandoned_or_failed_used: state.abandoned_or_failed_used,
lifetime_ingress_bytes: state.lifetime_ingress_bytes,
lifetime_egress_bytes: state.lifetime_egress_bytes,
lifetime_abandoned_or_failed_bytes: state.lifetime_abandoned_or_failed_bytes,
completed_transfers: state.completed_transfers,
failed_transfers: state.failed_transfers,
cancelled_transfers: state.cancelled_transfers,
expired_transfers: state.expired_transfers,
}
}
@ -305,13 +260,6 @@ impl ArtifactRelayLedger {
ingress_used: self.ingress_used,
egress_used: self.egress_used,
abandoned_or_failed_used: self.abandoned_or_failed_used,
lifetime_ingress_bytes: self.lifetime_ingress_bytes,
lifetime_egress_bytes: self.lifetime_egress_bytes,
lifetime_abandoned_or_failed_bytes: self.lifetime_abandoned_or_failed_bytes,
completed_transfers: self.completed_transfers,
failed_transfers: self.failed_transfers,
cancelled_transfers: self.cancelled_transfers,
expired_transfers: self.expired_transfers,
}
}
@ -543,11 +491,9 @@ impl ArtifactRelayLedger {
if ingress {
reservation.ingress_bytes = reservation.ingress_bytes.saturating_add(bytes);
self.ingress_used = self.ingress_used.saturating_add(bytes);
self.lifetime_ingress_bytes = self.lifetime_ingress_bytes.saturating_add(bytes);
} else {
reservation.egress_bytes = reservation.egress_bytes.saturating_add(bytes);
self.egress_used = self.egress_used.saturating_add(bytes);
self.lifetime_egress_bytes = self.lifetime_egress_bytes.saturating_add(bytes);
}
let project_key = (
reservation.scope.tenant.clone(),
@ -608,29 +554,10 @@ impl ArtifactRelayLedger {
pub(super) fn finish(&mut self, key: &str, reason: RelayFinishReason) {
if let Some(reservation) = self.reservations.remove(key) {
if reason != RelayFinishReason::Completed {
let abandoned_or_failed_bytes = reservation
.ingress_bytes
.saturating_add(reservation.egress_bytes);
self.abandoned_or_failed_used = self
.abandoned_or_failed_used
.saturating_add(abandoned_or_failed_bytes);
self.lifetime_abandoned_or_failed_bytes = self
.lifetime_abandoned_or_failed_bytes
.saturating_add(abandoned_or_failed_bytes);
}
match reason {
RelayFinishReason::Completed => {
self.completed_transfers = self.completed_transfers.saturating_add(1);
}
RelayFinishReason::Failed => {
self.failed_transfers = self.failed_transfers.saturating_add(1);
}
RelayFinishReason::Cancelled => {
self.cancelled_transfers = self.cancelled_transfers.saturating_add(1);
}
RelayFinishReason::Expired => {
self.expired_transfers = self.expired_transfers.saturating_add(1);
}
.saturating_add(reservation.ingress_bytes)
.saturating_add(reservation.egress_bytes);
}
}
}
@ -653,18 +580,6 @@ impl ArtifactRelayLedger {
ingress_bytes: self.ingress_used,
egress_bytes: self.egress_used,
abandoned_or_failed_bytes: self.abandoned_or_failed_used,
lifetime_ingress_bytes: self.lifetime_ingress_bytes,
lifetime_egress_bytes: self.lifetime_egress_bytes,
lifetime_abandoned_or_failed_bytes: self.lifetime_abandoned_or_failed_bytes,
completed_transfers: self.completed_transfers,
failed_transfers: self.failed_transfers,
cancelled_transfers: self.cancelled_transfers,
expired_transfers: self.expired_transfers,
tracked_scopes: self.project_used.len()
+ self.tenant_used.len()
+ self.account_used.len(),
max_active_global: self.configuration.max_active_global,
max_tracked_scopes: self.configuration.max_tracked_scopes,
reserved_bytes: self
.reservations
.values()
@ -673,68 +588,3 @@ impl ArtifactRelayLedger {
}
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn lifetime_metrics_survive_period_rollover_and_durable_round_trip() {
let mut ledger = ArtifactRelayLedger::new(ArtifactRelayConfiguration::unlimited());
ledger
.reserve(
"transfer".to_owned(),
TenantId::from("tenant"),
ProjectId::from("project"),
UserId::from("user"),
16,
16,
100,
1,
)
.unwrap();
ledger.charge_ingress("transfer", 24, 1).unwrap();
ledger.charge_egress("transfer", 24, 1).unwrap();
ledger.finish("transfer", RelayFinishReason::Completed);
let usage = ledger.usage();
assert_eq!(usage.lifetime_ingress_bytes, 24);
assert_eq!(usage.lifetime_egress_bytes, 24);
assert_eq!(usage.completed_transfers, 1);
ledger.prepare_period(u64::MAX);
let usage = ledger.usage();
assert_eq!(usage.ingress_bytes, 0);
assert_eq!(usage.egress_bytes, 0);
assert_eq!(usage.lifetime_ingress_bytes, 24);
assert_eq!(usage.lifetime_egress_bytes, 24);
let restored = ArtifactRelayLedger::from_durable(
ArtifactRelayConfiguration::unlimited(),
ledger.durable_state(),
);
let usage = restored.usage();
assert_eq!(usage.lifetime_ingress_bytes, 24);
assert_eq!(usage.lifetime_egress_bytes, 24);
assert_eq!(usage.completed_transfers, 1);
}
#[test]
fn older_durable_relay_state_defaults_new_metrics() {
let state: ArtifactRelayDurableState = serde_json::from_value(serde_json::json!({
"reservations": {},
"period": 1,
"project_used": [],
"tenant_used": [],
"account_used": [],
"ingress_used": 3,
"egress_used": 4,
"abandoned_or_failed_used": 2
}))
.unwrap();
assert_eq!(state.lifetime_ingress_bytes, 0);
assert_eq!(state.lifetime_egress_bytes, 0);
assert_eq!(state.completed_transfers, 0);
}
}

View file

@ -298,13 +298,6 @@ impl CoordinatorService {
project,
actor_user,
} => self.handle_list_node_descriptors(tenant, project, actor_user),
CoordinatorRequest::ListNodeSummaries {
tenant,
project,
actor_user,
cursor,
limit,
} => self.handle_list_node_summaries(tenant, project, actor_user, cursor, limit),
CoordinatorRequest::RevokeNodeCredential {
tenant,
project,
@ -428,13 +421,6 @@ impl CoordinatorService {
project,
actor_user,
} => self.handle_list_processes(tenant, project, actor_user),
CoordinatorRequest::ListProcessSummaries {
tenant,
project,
actor_user,
cursor,
limit,
} => self.handle_list_process_summaries(tenant, project, actor_user, cursor, limit),
CoordinatorRequest::QuotaStatus {
tenant,
project,
@ -480,8 +466,7 @@ impl CoordinatorService {
CoordinatorRequest::PollDebugCommand { .. }
| CoordinatorRequest::ReportDebugState { .. }
| CoordinatorRequest::ReportDebugProbeHit { .. } => self.reject_unsigned_node_request(),
CoordinatorRequest::ReportTaskLog { .. }
| CoordinatorRequest::ReportTaskLogChunk { .. } => self.reject_unsigned_node_request(),
CoordinatorRequest::ReportTaskLog { .. } => self.reject_unsigned_node_request(),
CoordinatorRequest::ReportVfsMetadata { .. } => self.reject_unsigned_node_request(),
CoordinatorRequest::TaskCompleted { .. } => self.reject_unsigned_node_request(),
CoordinatorRequest::ListTaskEvents {
@ -496,23 +481,6 @@ impl CoordinatorService {
actor_user,
process,
} => self.handle_list_task_snapshots(tenant, project, actor_user, process),
CoordinatorRequest::ListRecentLogs {
tenant,
project,
actor_user,
process,
task,
after_sequence,
limit,
} => self.handle_list_recent_logs(
tenant,
project,
actor_user,
process,
task,
after_sequence,
limit,
),
CoordinatorRequest::JoinTask {
tenant,
project,
@ -559,20 +527,6 @@ impl CoordinatorService {
max_bytes,
ttl_seconds,
),
CoordinatorRequest::ListArtifacts {
tenant,
project,
actor_user,
process,
cursor,
limit,
} => self.handle_list_artifacts(tenant, project, actor_user, process, cursor, limit),
CoordinatorRequest::GetArtifact {
tenant,
project,
actor_user,
artifact,
} => self.handle_get_artifact(tenant, project, actor_user, artifact),
CoordinatorRequest::OpenArtifactDownloadStream {
tenant,
project,
@ -742,14 +696,6 @@ impl CoordinatorService {
context.project.as_str().to_owned(),
actor.as_str().to_owned(),
),
AuthenticatedCoordinatorRequest::ListNodeSummaries { cursor, limit } => self
.handle_list_node_summaries(
context.tenant.as_str().to_owned(),
context.project.as_str().to_owned(),
actor.as_str().to_owned(),
cursor,
limit,
),
AuthenticatedCoordinatorRequest::RevokeNodeCredential { node } => self
.handle_revoke_node_credential(
context.tenant.as_str().to_owned(),
@ -801,14 +747,6 @@ impl CoordinatorService {
context.project.as_str().to_owned(),
actor.as_str().to_owned(),
),
AuthenticatedCoordinatorRequest::ListProcessSummaries { cursor, limit } => self
.handle_list_process_summaries(
context.tenant.as_str().to_owned(),
context.project.as_str().to_owned(),
actor.as_str().to_owned(),
cursor,
limit,
),
AuthenticatedCoordinatorRequest::QuotaStatus => self.handle_quota_status(
context.tenant.as_str().to_owned(),
context.project.as_str().to_owned(),
@ -864,20 +802,6 @@ impl CoordinatorService {
actor.as_str().to_owned(),
process,
),
AuthenticatedCoordinatorRequest::ListRecentLogs {
process,
task,
after_sequence,
limit,
} => self.handle_list_recent_logs(
context.tenant.as_str().to_owned(),
context.project.as_str().to_owned(),
actor.as_str().to_owned(),
process,
task,
after_sequence,
limit,
),
AuthenticatedCoordinatorRequest::JoinTask { process, task } => self.handle_join_task(
context.tenant.as_str().to_owned(),
context.project.as_str().to_owned(),
@ -897,24 +821,6 @@ impl CoordinatorService {
max_bytes,
ttl_seconds,
),
AuthenticatedCoordinatorRequest::ListArtifacts {
process,
cursor,
limit,
} => self.handle_list_artifacts(
context.tenant.as_str().to_owned(),
context.project.as_str().to_owned(),
actor.as_str().to_owned(),
process,
cursor,
limit,
),
AuthenticatedCoordinatorRequest::GetArtifact { artifact } => self.handle_get_artifact(
context.tenant.as_str().to_owned(),
context.project.as_str().to_owned(),
actor.as_str().to_owned(),
artifact,
),
AuthenticatedCoordinatorRequest::OpenArtifactDownloadStream {
artifact,
max_bytes,

View file

@ -253,29 +253,6 @@ impl CoordinatorService {
stderr_truncated,
backpressured,
),
CoordinatorRequest::ReportTaskLogChunk {
tenant,
project,
process,
node,
task,
stream,
offset,
source_bytes,
text,
truncated,
} => self.handle_report_task_log_chunk(
tenant,
project,
process,
node,
task,
stream,
offset,
source_bytes,
text,
truncated,
),
CoordinatorRequest::ReportVfsMetadata {
tenant,
project,
@ -369,7 +346,6 @@ fn signed_node_request_kind(
CoordinatorRequest::ReportDebugState { .. } => Ok("report_debug_state"),
CoordinatorRequest::ReportDebugProbeHit { .. } => Ok("report_debug_probe_hit"),
CoordinatorRequest::ReportTaskLog { .. } => Ok("report_task_log"),
CoordinatorRequest::ReportTaskLogChunk { .. } => Ok("report_task_log_chunk"),
CoordinatorRequest::ReportVfsMetadata { .. } => Ok("report_vfs_metadata"),
CoordinatorRequest::TaskCompleted { .. } => Ok("task_completed"),
_ => Err(CoordinatorError::Unauthorized(
@ -465,12 +441,6 @@ fn signed_node_request_scope(
node,
..
}
| CoordinatorRequest::ReportTaskLogChunk {
tenant,
project,
node,
..
}
| CoordinatorRequest::ReportVfsMetadata {
tenant,
project,

View file

@ -1,565 +0,0 @@
use std::collections::BTreeSet;
use std::time::Instant;
use clusterflux_core::{
ArtifactId, ArtifactMetadata, NodeId, ProcessId, ProjectId, TaskDefinitionId, TaskInstanceId,
TenantId, UserId,
};
use super::keys::{process_control_key, ProcessControlKey};
use super::{
ArtifactAvailability, ArtifactRetentionState, ArtifactSummary, CoordinatorResponse,
CoordinatorService, CoordinatorServiceError, DebugAcknowledgementState, DebugEpochSummary,
NodeSummary, ProcessActivityState, ProcessFinalResult, ProcessLifecycleState, ProcessSummary,
TaskAttemptState,
};
#[derive(Clone, Debug, PartialEq, Eq)]
pub(super) struct StoredProcessSummary {
pub(super) started_at_epoch_seconds: u64,
pub(super) ended_at_epoch_seconds: Option<u64>,
pub(super) final_result: Option<ProcessFinalResult>,
pub(super) connected_nodes: Vec<NodeId>,
pub(super) main_task_definition: Option<TaskDefinitionId>,
pub(super) main_task_instance: Option<TaskInstanceId>,
pub(super) main_terminal_state: Option<super::TaskTerminalState>,
pub(super) order: u64,
}
impl CoordinatorService {
pub(super) fn handle_list_node_summaries(
&mut self,
tenant: String,
project: String,
actor_user: String,
cursor: Option<String>,
limit: u32,
) -> Result<CoordinatorResponse, CoordinatorServiceError> {
let tenant = TenantId::new(tenant);
let project = ProjectId::new(project);
let actor = UserId::new(actor_user);
let cursor = cursor.as_deref();
let mut nodes = self
.node_descriptors
.iter()
.filter(|(scope, descriptor)| {
scope.tenant == tenant
&& scope.project == project
&& cursor.is_none_or(|cursor| descriptor.id.as_str() > cursor)
})
.map(|(scope, descriptor)| {
let online = self.node_is_live(scope);
NodeSummary {
id: descriptor.id.clone(),
display_name: descriptor.id.as_str().to_owned(),
online,
stale: !online,
last_seen_epoch_seconds: self.node_last_seen_epoch_seconds.get(scope).copied(),
capabilities: descriptor.capabilities.clone(),
direct_connectivity: descriptor.direct_connectivity,
}
})
.collect::<Vec<_>>();
nodes.sort_by(|left, right| left.id.cmp(&right.id));
let has_more = nodes.len() > limit as usize;
nodes.truncate(limit as usize);
let next_cursor = has_more
.then(|| nodes.last().map(|node| node.id.as_str().to_owned()))
.flatten();
Ok(CoordinatorResponse::NodeSummaries {
nodes,
next_cursor,
actor,
})
}
pub(super) fn record_process_started(
&mut self,
tenant: &TenantId,
project: &ProjectId,
process: &ProcessId,
now_epoch_seconds: u64,
) {
let key = process_control_key(tenant, project, process);
if let Some(logs) = self.recent_logs.get_mut(&(tenant.clone(), project.clone())) {
logs.retain(|entry| &entry.process != process);
if logs.is_empty() {
self.recent_logs.remove(&(tenant.clone(), project.clone()));
}
}
self.recent_log_dropped_through.remove(&key);
self.recent_log_accounted_bytes.retain(
|(entry_tenant, entry_project, entry_process, _, _), _| {
entry_tenant != tenant || entry_project != project || entry_process != process
},
);
self.recent_log_truncated_streams.retain(
|(entry_tenant, entry_project, entry_process, _, _)| {
entry_tenant != tenant || entry_project != project || entry_process != process
},
);
self.recent_log_quota_truncated_streams.retain(
|(entry_tenant, entry_project, entry_process, _, _)| {
entry_tenant != tenant || entry_project != project || entry_process != process
},
);
self.process_summary_order
.retain(|retained| retained != &key);
self.evict_process_summaries_for_project(tenant, project);
self.evict_process_summaries_total();
let order = self.next_process_summary_order;
self.next_process_summary_order = self.next_process_summary_order.saturating_add(1);
self.process_summaries.insert(
key.clone(),
StoredProcessSummary {
started_at_epoch_seconds: now_epoch_seconds,
ended_at_epoch_seconds: None,
final_result: None,
connected_nodes: Vec::new(),
main_task_definition: None,
main_task_instance: None,
main_terminal_state: None,
order,
},
);
self.process_summary_order.push_back(key);
}
pub(super) fn record_process_terminal(
&mut self,
tenant: &TenantId,
project: &ProjectId,
process: &ProcessId,
final_result: ProcessFinalResult,
now_epoch_seconds: u64,
) {
let key = process_control_key(tenant, project, process);
let connected_nodes = self
.coordinator
.active_process(tenant, project, process)
.map(|active| active.connected_nodes.iter().cloned().collect())
.unwrap_or_default();
let order = self.next_process_summary_order;
let entry = self
.process_summaries
.entry(key.clone())
.or_insert_with(|| {
self.next_process_summary_order = self.next_process_summary_order.saturating_add(1);
self.process_summary_order.push_back(key.clone());
StoredProcessSummary {
started_at_epoch_seconds: now_epoch_seconds,
ended_at_epoch_seconds: None,
final_result: None,
connected_nodes: Vec::new(),
main_task_definition: None,
main_task_instance: None,
main_terminal_state: None,
order,
}
});
entry.ended_at_epoch_seconds = Some(now_epoch_seconds);
entry.final_result = Some(final_result);
entry.connected_nodes = connected_nodes;
}
pub(super) fn record_main_terminal_state(
&mut self,
tenant: &TenantId,
project: &ProjectId,
process: &ProcessId,
task_definition: TaskDefinitionId,
task_instance: TaskInstanceId,
terminal_state: super::TaskTerminalState,
) {
let key = process_control_key(tenant, project, process);
if !self.process_summaries.contains_key(&key) {
self.record_process_started(
tenant,
project,
process,
self.liveness_now_epoch_seconds(),
);
}
if let Some(summary) = self.process_summaries.get_mut(&key) {
summary.main_task_definition = Some(task_definition);
summary.main_task_instance = Some(task_instance);
summary.main_terminal_state = Some(terminal_state);
}
}
pub(super) fn handle_list_process_summaries(
&mut self,
tenant: String,
project: String,
actor_user: String,
cursor: Option<String>,
limit: u32,
) -> Result<CoordinatorResponse, CoordinatorServiceError> {
let tenant = TenantId::new(tenant);
let project = ProjectId::new(project);
let actor = UserId::new(actor_user);
let cursor = parse_order_cursor(cursor.as_deref(), "process")?;
let mut stored = self
.process_summaries
.iter()
.filter(|((entry_tenant, entry_project, _), summary)| {
entry_tenant == &tenant
&& entry_project == &project
&& cursor.is_none_or(|cursor| summary.order < cursor)
})
.map(|(key, summary)| (key.clone(), summary.clone()))
.collect::<Vec<_>>();
stored.sort_by(|(_, left), (_, right)| right.order.cmp(&left.order));
let has_more = stored.len() > limit as usize;
stored.truncate(limit as usize);
let processes = stored
.into_iter()
.map(|(key, stored)| self.process_summary_from_stored(&key, stored))
.collect::<Vec<_>>();
let next_cursor = has_more
.then(|| processes.last().map(|process| process.order_cursor.clone()))
.flatten();
Ok(CoordinatorResponse::ProcessSummaries {
processes,
next_cursor,
actor,
})
}
fn process_summary_from_stored(
&self,
key: &ProcessControlKey,
stored: StoredProcessSummary,
) -> ProcessSummary {
let (tenant, project, process) = key;
let active = self.coordinator.active_process(tenant, project, process);
let process_key = process_control_key(tenant, project, process);
let main_wait_state = active.and_then(|_| {
if self.pending_task_launches.iter().any(|pending| {
&pending.tenant == tenant
&& &pending.project == project
&& &pending.process == process
}) {
Some("waiting_for_node".to_owned())
} else if self
.main_runtime
.is_waiting_for_task(tenant, project, process)
{
Some("waiting_for_task".to_owned())
} else {
None
}
});
let current_debug_epoch = self.debug_epoch_summary(&process_key);
let awaiting_action = self.task_attempts.iter().any(
|((attempt_tenant, attempt_project, attempt_process, _), attempts)| {
attempt_tenant == tenant
&& attempt_project == project
&& attempt_process == process
&& attempts.iter().any(|attempt| {
attempt.current && attempt.state == TaskAttemptState::FailedAwaitingAction
})
},
);
let activity = if let Some(result) = &stored.final_result {
match result {
ProcessFinalResult::Completed => ProcessActivityState::Completed,
ProcessFinalResult::Failed => ProcessActivityState::Failed,
ProcessFinalResult::Cancelled => ProcessActivityState::Cancelled,
}
} else if self.process_cancellations.contains(&process_key) {
ProcessActivityState::Cancelling
} else if current_debug_epoch
.as_ref()
.is_some_and(|epoch| epoch.partially_frozen)
{
ProcessActivityState::DebugEpochPartial
} else if awaiting_action {
ProcessActivityState::AwaitingAction
} else {
match main_wait_state.as_deref() {
Some("waiting_for_node") => ProcessActivityState::WaitingForNode,
Some("waiting_for_task") => ProcessActivityState::WaitingForTask,
_ => ProcessActivityState::Running,
}
};
let connected_nodes = active
.map(|active| active.connected_nodes.iter().cloned().collect())
.unwrap_or(stored.connected_nodes);
ProcessSummary {
process: process.clone(),
lifecycle: if active.is_some() {
ProcessLifecycleState::Active
} else {
ProcessLifecycleState::RecentTerminal
},
activity,
main_wait_state,
started_at_epoch_seconds: stored.started_at_epoch_seconds,
ended_at_epoch_seconds: stored.ended_at_epoch_seconds,
final_result: stored.final_result,
connected_nodes,
current_debug_epoch,
order_cursor: format!("process:{}", stored.order),
}
}
fn debug_epoch_summary(&self, key: &ProcessControlKey) -> Option<DebugEpochSummary> {
let runtime = self.debug_epoch_runtime.get(key)?;
let acknowledgements = runtime.acknowledgements.values().collect::<Vec<_>>();
let all_acknowledged = !runtime.expected.is_empty()
&& runtime
.expected
.iter()
.all(|participant| runtime.acknowledgements.contains_key(participant));
let fully_frozen = runtime.command == "freeze"
&& all_acknowledged
&& acknowledgements
.iter()
.all(|ack| ack.state == DebugAcknowledgementState::Frozen);
let freeze_deadline_elapsed =
runtime.command == "freeze" && Instant::now() >= runtime.deadline;
let frozen_count = acknowledgements
.iter()
.filter(|ack| ack.state == DebugAcknowledgementState::Frozen)
.count();
let partially_frozen = freeze_deadline_elapsed && frozen_count > 0 && !fully_frozen;
let fully_resumed = runtime.command == "resume"
&& all_acknowledged
&& acknowledgements
.iter()
.all(|ack| ack.state == DebugAcknowledgementState::Running);
let failed = acknowledgements
.iter()
.any(|ack| ack.state == DebugAcknowledgementState::Failed)
|| (freeze_deadline_elapsed
&& runtime
.expected
.iter()
.any(|participant| !runtime.acknowledgements.contains_key(participant)));
Some(DebugEpochSummary {
epoch: runtime.epoch,
command: runtime.command.clone(),
fully_frozen,
partially_frozen,
fully_resumed,
failed,
})
}
pub(super) fn handle_list_artifacts(
&mut self,
tenant: String,
project: String,
actor_user: String,
process: Option<String>,
cursor: Option<String>,
limit: u32,
) -> Result<CoordinatorResponse, CoordinatorServiceError> {
let tenant = TenantId::new(tenant);
let project = ProjectId::new(project);
let _actor = UserId::new(actor_user);
let process = process.map(ProcessId::new);
if let Some(process) = &process {
self.authorize_task_event_process_scope(&tenant, &project, process)?;
}
let cursor = parse_order_cursor(cursor.as_deref(), "artifact")?;
let mut metadata = self
.artifact_registry
.metadata_for_project(&tenant, &project)
.filter(|metadata| {
process
.as_ref()
.is_none_or(|process| &metadata.process == process)
&& cursor.is_none_or(|cursor| metadata.flushed_epoch < cursor)
})
.cloned()
.collect::<Vec<_>>();
metadata.sort_by(|left, right| right.flushed_epoch.cmp(&left.flushed_epoch));
let has_more = metadata.len() > limit as usize;
metadata.truncate(limit as usize);
let artifacts = metadata
.into_iter()
.map(|metadata| self.artifact_summary(metadata))
.collect::<Vec<_>>();
let next_cursor = has_more
.then(|| {
artifacts
.last()
.map(|artifact| artifact.order_cursor.clone())
})
.flatten();
Ok(CoordinatorResponse::Artifacts {
artifacts,
next_cursor,
})
}
pub(super) fn handle_get_artifact(
&mut self,
tenant: String,
project: String,
actor_user: String,
artifact: String,
) -> Result<CoordinatorResponse, CoordinatorServiceError> {
let tenant = TenantId::new(tenant);
let project = ProjectId::new(project);
let _actor = UserId::new(actor_user);
let artifact = ArtifactId::new(artifact);
let metadata = self
.artifact_registry
.metadata(&tenant, &project, &artifact)
.cloned()
.ok_or(clusterflux_core::DownloadError::NotFound)?;
Ok(CoordinatorResponse::Artifact {
artifact: self.artifact_summary(metadata),
})
}
fn artifact_summary(&self, metadata: ArtifactMetadata) -> ArtifactSummary {
let live_retaining_nodes = metadata
.retaining_nodes
.iter()
.filter(|node| {
self.node_is_live(&crate::NodeScopeKey::from_refs(
&metadata.tenant,
&metadata.project,
node,
))
})
.cloned()
.collect::<BTreeSet<_>>();
let safe_node = live_retaining_nodes
.iter()
.next()
.cloned()
.or_else(|| metadata.retaining_nodes.iter().next().cloned());
let explicit_storage = !metadata.explicit_locations.is_empty();
let downloadable_now = !live_retaining_nodes.is_empty() || explicit_storage;
let availability = if downloadable_now {
ArtifactAvailability::Available
} else if !metadata.retaining_nodes.is_empty() {
ArtifactAvailability::NodeOffline
} else {
ArtifactAvailability::Unavailable
};
let retention_state = if explicit_storage {
ArtifactRetentionState::ExplicitStorage
} else if !metadata.retaining_nodes.is_empty() {
ArtifactRetentionState::NodeRetained
} else {
ArtifactRetentionState::Lost
};
let display_suffix = metadata.id.as_str().replace(':', "/");
let display_path = format!("/vfs/artifacts/{display_suffix}");
let display_name = display_suffix
.rsplit('/')
.next()
.unwrap_or(metadata.id.as_str())
.to_owned();
ArtifactSummary {
id: metadata.id,
display_path,
display_name,
process: metadata.process,
producer_task: metadata.producer_task,
safe_node,
digest: metadata.digest,
size_bytes: metadata.size,
availability,
downloadable_now,
retention_state,
explicit_storage,
order_cursor: format!("artifact:{}", metadata.flushed_epoch),
}
}
fn evict_process_summaries_for_project(&mut self, tenant: &TenantId, project: &ProjectId) {
while self
.process_summaries
.keys()
.filter(|(entry_tenant, entry_project, _)| {
entry_tenant == tenant && entry_project == project
})
.count()
>= super::MAX_RECENT_PROCESS_SUMMARIES_PER_PROJECT
{
let candidate = self.process_summary_order.iter().find(|key| {
&key.0 == tenant
&& &key.1 == project
&& self
.process_summaries
.get(*key)
.is_some_and(|summary| summary.final_result.is_some())
});
let Some(candidate) = candidate.cloned() else {
break;
};
self.remove_process_summary_state(&candidate);
}
}
fn evict_process_summaries_total(&mut self) {
while self.process_summaries.len() >= super::MAX_RECENT_PROCESS_SUMMARIES_TOTAL {
let candidate = self.process_summary_order.iter().find(|key| {
self.process_summaries
.get(*key)
.is_some_and(|summary| summary.final_result.is_some())
});
let Some(candidate) = candidate.cloned() else {
break;
};
self.remove_process_summary_state(&candidate);
}
}
fn remove_process_summary_state(&mut self, key: &super::ProcessControlKey) {
self.process_summaries.remove(key);
self.recent_log_dropped_through.remove(key);
if let Some(logs) = self.recent_logs.get_mut(&(key.0.clone(), key.1.clone())) {
logs.retain(|entry| entry.process != key.2);
if logs.is_empty() {
self.recent_logs.remove(&(key.0.clone(), key.1.clone()));
}
}
self.recent_log_accounted_bytes
.retain(|(tenant, project, process, _, _), _| {
tenant != &key.0 || project != &key.1 || process != &key.2
});
self.recent_log_truncated_streams
.retain(|(tenant, project, process, _, _)| {
tenant != &key.0 || project != &key.1 || process != &key.2
});
self.recent_log_quota_truncated_streams
.retain(|(tenant, project, process, _, _)| {
tenant != &key.0 || project != &key.1 || process != &key.2
});
self.process_summary_order
.retain(|retained| retained != key);
}
}
fn parse_order_cursor(
cursor: Option<&str>,
expected_kind: &str,
) -> Result<Option<u64>, CoordinatorServiceError> {
cursor
.map(|cursor| {
let (kind, order) = cursor.split_once(':').ok_or_else(|| {
CoordinatorServiceError::Protocol(format!(
"invalid {expected_kind} pagination cursor"
))
})?;
if kind != expected_kind {
return Err(CoordinatorServiceError::Protocol(format!(
"invalid {expected_kind} pagination cursor"
)));
}
order.parse::<u64>().map_err(|_| {
CoordinatorServiceError::Protocol(format!(
"invalid {expected_kind} pagination cursor"
))
})
})
.transpose()
}

View file

@ -79,15 +79,17 @@ impl CoordinatorService {
continue;
}
let response = match decode_wire_request(&line) {
Ok((request_id, request)) => {
match authorize_client_request(&request, authority_mode)
Ok(request) => match authorize_client_request(&request, authority_mode)
.and_then(|()| self.handle_request(request))
{
Ok(response) => response,
Err(err) => CoordinatorResponse::service_error(request_id, &err),
}
}
Err(err) => CoordinatorResponse::service_error(wire_request_id_hint(&line), &err),
Err(err) => CoordinatorResponse::Error {
message: err.to_string(),
},
},
Err(err) => CoordinatorResponse::Error {
message: err.to_string(),
},
};
serde_json::to_writer(&mut writer, &response)?;
writer.write_all(b"\n")?;
@ -112,19 +114,25 @@ fn handle_shared_stream(
continue;
}
let response = match decode_wire_request(&line) {
Ok((request_id, request)) => match authorize_client_request(&request, authority_mode) {
Ok(request) => match authorize_client_request(&request, authority_mode) {
Ok(()) => match service.lock() {
Ok(mut service) => match service.handle_request(request) {
Ok(response) => response,
Err(err) => CoordinatorResponse::service_error(request_id, &err),
Err(err) => CoordinatorResponse::Error {
message: err.to_string(),
},
Err(_) => {
CoordinatorResponse::error(request_id, "coordinator service lock poisoned")
}
},
Err(err) => CoordinatorResponse::service_error(request_id, &err),
Err(_) => CoordinatorResponse::Error {
message: "coordinator service lock poisoned".to_owned(),
},
},
Err(err) => CoordinatorResponse::Error {
message: err.to_string(),
},
},
Err(err) => CoordinatorResponse::Error {
message: err.to_string(),
},
Err(err) => CoordinatorResponse::service_error(wire_request_id_hint(&line), &err),
};
serde_json::to_writer(&mut writer, &response)?;
writer.write_all(b"\n")?;
@ -138,27 +146,12 @@ pub fn bind_listener(addr: &str) -> Result<(TcpListener, SocketAddr), Coordinato
Ok((listener, addr))
}
fn decode_wire_request(
line: &str,
) -> Result<(String, CoordinatorRequest), CoordinatorServiceError> {
fn decode_wire_request(line: &str) -> Result<CoordinatorRequest, CoordinatorServiceError> {
serde_json::from_str::<super::CoordinatorWireRequest>(line)?
.into_parts()
.into_request()
.map_err(CoordinatorServiceError::Protocol)
}
fn wire_request_id_hint(line: &str) -> String {
serde_json::from_str::<serde_json::Value>(line)
.ok()
.and_then(|value| {
value
.get("request_id")
.and_then(|value| value.as_str())
.map(str::to_owned)
})
.filter(|request_id| clusterflux_core::RequestId::try_new(request_id.clone()).is_ok())
.unwrap_or_else(|| "unavailable".to_owned())
}
fn authorize_client_request(
request: &CoordinatorRequest,
authority_mode: ClientAuthorityMode,
@ -187,11 +180,10 @@ fn authorize_client_request(
}
| CoordinatorRequest::AdminStatus { .. }
| CoordinatorRequest::SuspendTenant { .. } => Ok(()),
_ => Err(crate::CoordinatorError::Unauthorized(
_ => Err(CoordinatorServiceError::Protocol(
"strict Core Client authority requires an authenticated CLI session, signed Agent, signed Node, enrollment grant exchange, or admin credential; request-body identity fields are not authority"
.to_owned(),
)
.into()),
)),
}
}
@ -261,19 +253,16 @@ mod transport_boundary_tests {
let mut line = String::new();
reader.read_line(&mut line).unwrap();
let CoordinatorResponse::Error { error } =
let CoordinatorResponse::Error { message } =
serde_json::from_str::<CoordinatorResponse>(&line).unwrap()
else {
panic!("malformed identifier request unexpectedly succeeded");
};
assert!(
error.message.contains("malformed external identifier")
&& error.message.contains("request.request.process"),
"unexpected malformed identifier response: {}",
error.message
message.contains("malformed external identifier")
&& message.contains("request.request.process"),
"unexpected malformed identifier response: {message}"
);
assert_eq!(error.request_id, format!("malformed-{index}"));
assert_eq!(error.code, clusterflux_core::ApiErrorCode::ValidationError);
let valid = coordinator_wire_request(
format!("healthy-{index}"),

File diff suppressed because it is too large Load diff

View file

@ -12,12 +12,8 @@ pub enum CoordinatorWireRequest {
impl CoordinatorWireRequest {
pub fn into_request(self) -> Result<CoordinatorRequest, String> {
self.into_parts().map(|(_, request)| request)
}
pub fn into_parts(self) -> Result<(String, CoordinatorRequest), String> {
match self {
Self::Envelope(envelope) => envelope.into_parts(),
Self::Envelope(envelope) => envelope.into_request(),
}
}
}
@ -36,10 +32,6 @@ pub struct CoordinatorRequestEnvelope {
impl CoordinatorRequestEnvelope {
pub fn into_request(self) -> Result<CoordinatorRequest, String> {
self.into_parts().map(|(_, request)| request)
}
pub fn into_parts(self) -> Result<(String, CoordinatorRequest), String> {
if self.envelope_type != COORDINATOR_WIRE_REQUEST_TYPE {
return Err(format!(
"unsupported coordinator wire request type {}; expected {}",
@ -62,6 +54,6 @@ impl CoordinatorRequestEnvelope {
self.operation, payload_operation
));
}
Ok((self.request_id, self.payload))
Ok(self.payload)
}
}

View file

@ -1,296 +0,0 @@
use serde::{Deserialize, Serialize};
use std::fmt;
#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum ApiErrorCode {
Unauthenticated,
SessionExpired,
AccountSuspended,
Forbidden,
ValidationError,
NotFound,
Conflict,
ActiveProcessExists,
NodeOffline,
NoCapableNode,
TaskNotRestartable,
ArtifactUnavailable,
ArtifactLimitExceeded,
QuotaExceeded,
TemporaryCapacity,
DebugEpochPartial,
InternalError,
}
#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum ApiErrorCategory {
Authentication,
Authorization,
Validation,
State,
Availability,
Resource,
Internal,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
#[serde(deny_unknown_fields)]
pub struct ApiError {
pub code: ApiErrorCode,
pub category: ApiErrorCategory,
pub message: String,
pub retryable: bool,
pub request_id: String,
}
impl fmt::Display for ApiError {
fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result {
write!(
formatter,
"{} (code {:?}, request {})",
self.message, self.code, self.request_id
)
}
}
impl std::error::Error for ApiError {}
impl ApiError {
pub fn new(
code: ApiErrorCode,
category: ApiErrorCategory,
message: impl Into<String>,
retryable: bool,
request_id: impl Into<String>,
) -> Self {
Self {
code,
category,
message: message.into(),
retryable,
request_id: request_id.into(),
}
}
pub fn from_message(request_id: impl Into<String>, message: impl Into<String>) -> Self {
let request_id = request_id.into();
let message = message.into();
let normalized = message.to_ascii_lowercase();
let (code, category, retryable) = if normalized.contains("session credential has expired")
|| normalized.contains("session expired")
{
(
ApiErrorCode::SessionExpired,
ApiErrorCategory::Authentication,
false,
)
} else if normalized.contains("tenant is suspended")
|| normalized.contains("account is suspended")
|| normalized.contains("suspended by hosted")
{
(
ApiErrorCode::AccountSuspended,
ApiErrorCategory::Authorization,
false,
)
} else if normalized.contains("session credential")
|| normalized.contains("no authenticated")
|| normalized.contains("not authenticated")
|| normalized.contains("credential is not")
{
(
ApiErrorCode::Unauthenticated,
ApiErrorCategory::Authentication,
false,
)
} else if normalized.contains("project already has active virtual process") {
(
ApiErrorCode::ActiveProcessExists,
ApiErrorCategory::State,
false,
)
} else if normalized.contains("no capable node") {
(
ApiErrorCode::NoCapableNode,
ApiErrorCategory::Availability,
true,
)
} else if normalized.contains("node offline")
|| normalized.contains("node is not live")
|| normalized.contains("source node is not connected")
|| normalized.contains("direct connectivity unavailable")
{
(
ApiErrorCode::NodeOffline,
ApiErrorCategory::Availability,
true,
)
} else if normalized.contains("restart")
&& (normalized.contains("not restartable")
|| normalized.contains("requires whole")
|| normalized.contains("clean boundary"))
{
(
ApiErrorCode::TaskNotRestartable,
ApiErrorCategory::State,
false,
)
} else if normalized.contains("artifact")
&& (normalized.contains("exceeds download limit")
|| normalized.contains("download session limit")
|| normalized.contains("artifact limit"))
{
(
ApiErrorCode::ArtifactLimitExceeded,
ApiErrorCategory::Resource,
false,
)
} else if normalized.contains("artifact")
&& (normalized.contains("does not exist")
|| normalized.contains("not found")
|| normalized.contains("unknown artifact"))
{
(ApiErrorCode::NotFound, ApiErrorCategory::State, false)
} else if normalized.contains("artifact")
&& (normalized.contains("unavailable") || normalized.contains("retention"))
{
(
ApiErrorCode::ArtifactUnavailable,
ApiErrorCategory::Availability,
true,
)
} else if normalized.contains("resource limit")
|| normalized.contains("quota")
|| normalized.contains("limit exceeded")
{
(
ApiErrorCode::QuotaExceeded,
ApiErrorCategory::Resource,
true,
)
} else if normalized.contains("capacity")
|| normalized.contains("temporarily full")
|| normalized.contains("replay window is full")
{
(
ApiErrorCode::TemporaryCapacity,
ApiErrorCategory::Availability,
true,
)
} else if normalized.contains("partial debug epoch")
|| normalized.contains("debug epoch is partially")
{
(
ApiErrorCode::DebugEpochPartial,
ApiErrorCategory::State,
true,
)
} else if normalized.contains("malformed")
|| normalized.contains("invalid ")
|| normalized.contains("protocol")
|| normalized.contains("unknown field")
|| normalized.contains("missing field")
|| normalized.contains("must ")
{
(
ApiErrorCode::ValidationError,
ApiErrorCategory::Validation,
false,
)
} else if normalized.contains("outside")
|| normalized.contains("unauthorized")
|| normalized.contains("denied")
|| normalized.contains("requires an authenticated")
|| normalized.contains("may only")
{
(
ApiErrorCode::Forbidden,
ApiErrorCategory::Authorization,
false,
)
} else if normalized.contains("not found")
|| normalized.contains("does not exist")
|| normalized.contains("unknown ")
{
(ApiErrorCode::NotFound, ApiErrorCategory::State, false)
} else if normalized.contains("already")
|| normalized.contains("conflict")
|| normalized.contains("requires an active")
{
(ApiErrorCode::Conflict, ApiErrorCategory::State, false)
} else {
(
ApiErrorCode::InternalError,
ApiErrorCategory::Internal,
false,
)
};
Self::new(code, category, message, retryable, request_id)
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn required_machine_codes_are_stably_serialized() {
let required = [
ApiErrorCode::Unauthenticated,
ApiErrorCode::SessionExpired,
ApiErrorCode::AccountSuspended,
ApiErrorCode::Forbidden,
ApiErrorCode::ValidationError,
ApiErrorCode::NotFound,
ApiErrorCode::Conflict,
ApiErrorCode::ActiveProcessExists,
ApiErrorCode::NodeOffline,
ApiErrorCode::NoCapableNode,
ApiErrorCode::TaskNotRestartable,
ApiErrorCode::ArtifactUnavailable,
ApiErrorCode::ArtifactLimitExceeded,
ApiErrorCode::QuotaExceeded,
ApiErrorCode::TemporaryCapacity,
ApiErrorCode::DebugEpochPartial,
];
let serialized = required
.into_iter()
.map(|code| serde_json::to_value(code).unwrap())
.collect::<Vec<_>>();
assert_eq!(
serialized,
vec![
"unauthenticated",
"session_expired",
"account_suspended",
"forbidden",
"validation_error",
"not_found",
"conflict",
"active_process_exists",
"node_offline",
"no_capable_node",
"task_not_restartable",
"artifact_unavailable",
"artifact_limit_exceeded",
"quota_exceeded",
"temporary_capacity",
"debug_epoch_partial",
]
);
}
#[test]
fn message_classification_keeps_request_identity() {
let error = ApiError::from_message(
"request-17",
"CLI session credential has expired; run login again",
);
assert_eq!(error.code, ApiErrorCode::SessionExpired);
assert_eq!(error.category, ApiErrorCategory::Authentication);
assert_eq!(error.request_id, "request-17");
assert!(!error.retryable);
}
}

View file

@ -195,34 +195,7 @@ impl ArtifactRegistry {
protected_processes: &BTreeSet<ProcessId>,
) -> Result<ArtifactMetadata, String> {
let key = ArtifactScopeKey::from_refs(&flush.tenant, &flush.project, &flush.id);
let replacing = self.artifacts.contains_key(&key);
let retained_for_project = self
.artifacts
.values()
.filter(|metadata| metadata.tenant == flush.tenant && metadata.project == flush.project)
.count();
let eviction = if replacing || retained_for_project < MAX_ARTIFACT_METADATA_PER_PROJECT {
None
} else {
let mut protected_processes = protected_processes.clone();
protected_processes.insert(flush.process.clone());
Some(
self.project_metadata_eviction_candidate(
&flush.tenant,
&flush.project,
pinned,
&protected_processes,
)
.ok_or_else(|| {
format!(
"artifact metadata capacity of {MAX_ARTIFACT_METADATA_PER_PROJECT} is \
exhausted by active or retained artifacts"
)
})?,
)
};
self.next_epoch = self.next_epoch.saturating_add(1);
self.next_epoch += 1;
let metadata = ArtifactMetadata {
id: flush.id.clone(),
tenant: flush.tenant,
@ -237,10 +210,15 @@ impl ArtifactRegistry {
explicit_locations: Vec::new(),
coordinator_has_large_bytes: false,
};
if let Some(eviction) = eviction {
self.artifacts.remove(&eviction);
}
self.artifacts.insert(key, metadata.clone());
let mut protected_processes = protected_processes.clone();
protected_processes.insert(metadata.process.clone());
self.enforce_project_metadata_limit(
&metadata.tenant,
&metadata.project,
pinned,
&protected_processes,
);
Ok(metadata)
}
@ -259,29 +237,8 @@ impl ArtifactRegistry {
.count()
> MAX_ARTIFACT_METADATA_PER_PROJECT
{
let candidate = self.project_metadata_eviction_candidate(
tenant,
project,
pinned,
protected_processes,
);
let Some(candidate) = candidate else {
break;
};
self.artifacts.remove(&candidate);
evicted += 1;
}
evicted
}
fn project_metadata_eviction_candidate(
&self,
tenant: &TenantId,
project: &ProjectId,
pinned: &BTreeSet<ArtifactScopeKey>,
protected_processes: &BTreeSet<ProcessId>,
) -> Option<ArtifactScopeKey> {
self.artifacts
let candidate = self
.artifacts
.values()
.filter(|metadata| {
&metadata.tenant == tenant
@ -303,7 +260,14 @@ impl ArtifactRegistry {
.min_by_key(|metadata| metadata.flushed_epoch)
.map(|metadata| {
ArtifactScopeKey::from_refs(&metadata.tenant, &metadata.project, &metadata.id)
})
});
let Some(candidate) = candidate else {
break;
};
self.artifacts.remove(&candidate);
evicted += 1;
}
evicted
}
pub fn sync_to_explicit_store(
@ -360,16 +324,6 @@ impl ArtifactRegistry {
.get(&ArtifactScopeKey::from_refs(tenant, project, artifact))
}
pub fn metadata_for_project<'a>(
&'a self,
tenant: &'a TenantId,
project: &'a ProjectId,
) -> impl Iterator<Item = &'a ArtifactMetadata> + 'a {
self.artifacts
.values()
.filter(move |metadata| &metadata.tenant == tenant && &metadata.project == project)
}
pub fn download_action(
&self,
context: &AuthContext,
@ -1521,87 +1475,6 @@ mod tests {
);
}
#[test]
fn artifact_metadata_capacity_rejects_atomically_when_every_entry_is_protected() {
let mut registry = ArtifactRegistry::default();
let tenant = TenantId::from("tenant");
let project = ProjectId::from("project");
let active_process = ProcessId::from("active-process");
for index in 0..MAX_ARTIFACT_METADATA_PER_PROJECT {
registry.flush_metadata(ArtifactFlush {
id: ArtifactId::new(format!("artifact-{index}")),
tenant: tenant.clone(),
project: project.clone(),
process: active_process.clone(),
producer_task: TaskInstanceId::new(format!("task-{index}")),
retaining_node: NodeId::from("node"),
digest: Digest::sha256(format!("content-{index}")),
size: 1,
});
}
let replacement = ArtifactFlush {
id: ArtifactId::from("artifact-0"),
tenant: tenant.clone(),
project: project.clone(),
process: active_process.clone(),
producer_task: TaskInstanceId::from("replacement-task"),
retaining_node: NodeId::from("node"),
digest: Digest::sha256("replacement"),
size: 2,
};
registry
.flush_metadata_with_protected_processes(
replacement,
&BTreeSet::new(),
&BTreeSet::from([active_process.clone()]),
)
.expect("replacement must remain possible at the metadata bound");
assert_eq!(
registry
.metadata(&tenant, &project, &ArtifactId::from("artifact-0"))
.unwrap()
.digest,
Digest::sha256("replacement")
);
let epoch_before_rejection = registry.next_epoch;
let result = registry.flush_metadata_with_protected_processes(
ArtifactFlush {
id: ArtifactId::from("artifact-rejected"),
tenant: tenant.clone(),
project: project.clone(),
process: active_process.clone(),
producer_task: TaskInstanceId::from("rejected-task"),
retaining_node: NodeId::from("node"),
digest: Digest::sha256("rejected"),
size: 3,
},
&BTreeSet::new(),
&BTreeSet::from([active_process]),
);
assert!(result
.unwrap_err()
.contains("exhausted by active or retained artifacts"));
assert_eq!(registry.next_epoch, epoch_before_rejection);
assert_eq!(
registry.metadata_for_project(&tenant, &project).count(),
MAX_ARTIFACT_METADATA_PER_PROJECT
);
assert!(registry
.metadata(&tenant, &project, &ArtifactId::from("artifact-rejected"))
.is_none());
assert_eq!(
registry
.metadata(&tenant, &project, &ArtifactId::from("artifact-0"))
.unwrap()
.digest,
Digest::sha256("replacement"),
"rejection must not modify existing metadata"
);
}
#[test]
fn download_stream_accounts_usage_before_and_during_streaming() {
let mut registry = registry_with_artifact();

View file

@ -1,4 +1,3 @@
mod api_error;
pub mod artifact;
pub mod auth;
pub mod bundle;
@ -19,7 +18,6 @@ pub mod transport;
pub mod vfs;
pub mod wire;
pub use api_error::{ApiError, ApiErrorCategory, ApiErrorCode};
pub use artifact::{
ArtifactDownloadStream, ArtifactFlush, ArtifactHandle, ArtifactMetadata, ArtifactRegistry,
ArtifactScopeKey, ArtifactUnavailable, DownloadAction, DownloadError, DownloadLink,

View file

@ -36,26 +36,6 @@ use demo_backend::{LINUX_THREAD, MAIN_THREAD, PACKAGE_THREAD, WINDOWS_THREAD};
use virtual_model::{process_id, RuntimeBackend};
fn main() -> Result<()> {
let raw_args = std::env::args().skip(1).collect::<Vec<_>>();
match raw_args.as_slice() {
[flag] if matches!(flag.as_str(), "--version" | "-V") => {
println!("clusterflux-debug-dap {}", env!("CARGO_PKG_VERSION"));
return Ok(());
}
[flag] if matches!(flag.as_str(), "--help" | "-h") => {
println!(
"Clusterflux Debug Adapter Protocol server.\n\n\
Usage: clusterflux-debug-dap\n\n\
The adapter communicates over standard input and output.\n\n\
Options:\n \
-h, --help\n \
-V, --version"
);
return Ok(());
}
[] => {}
[argument, ..] => anyhow::bail!("unknown argument: {argument}"),
}
adapter::run_adapter()
}

View file

@ -742,16 +742,14 @@ fn fetch_current_process_status(
client_user_request(
state,
json!({
"type": "list_process_summaries",
"type": "list_processes",
"tenant": state.tenant,
"project": state.project_id,
"actor_user": state.actor_user,
"cursor": null,
"limit": 100,
}),
),
)?;
let summary = statuses
let current = statuses
.get("processes")
.and_then(Value::as_array)
.and_then(|processes| {
@ -760,9 +758,6 @@ fn fetch_current_process_status(
})
})
.cloned();
let current = merge_active_process_status(state, summary, |request| {
coordinator_request(coordinator, request)
})?;
Ok((statuses, current))
}
@ -773,15 +768,13 @@ fn fetch_current_process_status_in(
let statuses = session.request(client_user_request(
state,
json!({
"type": "list_process_summaries",
"type": "list_processes",
"tenant": state.tenant,
"project": state.project_id,
"actor_user": state.actor_user,
"cursor": null,
"limit": 100,
}),
))?;
let summary = statuses
let current = statuses
.get("processes")
.and_then(Value::as_array)
.and_then(|processes| {
@ -790,51 +783,9 @@ fn fetch_current_process_status_in(
})
})
.cloned();
let current = merge_active_process_status(state, summary, |request| session.request(request))?;
Ok((statuses, current))
}
fn merge_active_process_status<F>(
state: &AdapterState,
summary: Option<Value>,
mut request: F,
) -> Result<Option<Value>>
where
F: FnMut(Value) -> Result<Value>,
{
let Some(mut summary) = summary else {
return Ok(None);
};
if summary.get("lifecycle").and_then(Value::as_str) != Some("active") {
return Ok(Some(summary));
}
let active_statuses = request(client_user_request(
state,
json!({
"type": "list_processes",
"tenant": state.tenant,
"project": state.project_id,
"actor_user": state.actor_user,
}),
))?;
let active = active_statuses
.get("processes")
.and_then(Value::as_array)
.and_then(|processes| {
processes.iter().find(|process| {
process.get("process").and_then(Value::as_str) == Some(state.process.as_str())
})
});
if let (Some(summary), Some(active)) =
(summary.as_object_mut(), active.and_then(Value::as_object))
{
for (key, value) in active {
summary.entry(key.clone()).or_insert_with(|| value.clone());
}
}
Ok(Some(summary))
}
pub(crate) fn relaunch_services_main_runtime(state: &AdapterState) -> Result<RuntimeLaunchRecord> {
let coordinator =
crate::view_state::normalize_coordinator_endpoint(&state.coordinator_endpoint);
@ -908,7 +859,27 @@ pub(crate) fn attach_services_runtime(state: &AdapterState) -> Result<RuntimeLau
.count()
})
.unwrap_or(0);
let (process_statuses, process_status) = fetch_current_process_status(&coordinator, state)?;
let process_statuses = coordinator_request(
&coordinator,
client_user_request(
state,
json!({
"type": "list_processes",
"tenant": state.tenant,
"project": state.project_id,
"actor_user": state.actor_user,
}),
),
)?;
let process_status = process_statuses
.get("processes")
.and_then(Value::as_array)
.and_then(|processes| {
processes.iter().find(|process| {
process.get("process").and_then(Value::as_str) == Some(state.process.as_str())
})
})
.cloned();
let node = process_status
.as_ref()
.and_then(|status| status.get("connected_nodes"))
@ -1108,8 +1079,9 @@ pub(crate) fn observe_services_runtime(
}
}
}
// Events remain useful for output and attempt details, but terminal
// authority comes from the durable process summary read below.
// Terminal task events remain inspectable after the active process slot is
// released. Read them before active-process debug state so a fast main exit
// cannot turn a successful continuation into an authorization error.
let current_session = session.as_mut().expect("observer session connected");
let events = match current_session.request(client_user_request(
state,
@ -1146,6 +1118,65 @@ pub(crate) fn observe_services_runtime(
std::thread::sleep(reconnect_delay);
continue;
}
if let Some(mut outcome) = terminal_runtime_outcome(&coordinator, state, &events) {
let snapshot_request = if inject_snapshot_failure && !snapshot_failure_injected {
snapshot_failure_injected = true;
Err(anyhow!("injected task snapshot transport failure"))
} else {
fetch_task_snapshots_in(current_session, state)
};
let task_snapshots = match snapshot_request {
Ok(snapshots) => snapshots,
Err(error) => {
if !emit(RuntimeContinuationOutcome::Diagnostic(format!(
"runtime terminal snapshot observation failed: {error:#}; reconnecting in {} ms",
reconnect_delay.as_millis()
))) {
return Ok(());
}
session = None;
std::thread::sleep(reconnect_delay);
reconnect_delay = (reconnect_delay * 2).min(Duration::from_secs(5));
continue;
}
};
let process_status_request =
if inject_process_status_failure && !process_status_failure_injected {
process_status_failure_injected = true;
Err(anyhow!("injected process-status transport failure"))
} else {
fetch_current_process_status_in(current_session, state)
};
let (process_statuses, process_status) = match process_status_request {
Ok(status) => status,
Err(error) => {
if !emit(RuntimeContinuationOutcome::Diagnostic(format!(
"runtime terminal process observation failed: {error:#}; reconnecting in {} ms",
reconnect_delay.as_millis()
))) {
return Ok(());
}
session = None;
std::thread::sleep(reconnect_delay);
reconnect_delay = (reconnect_delay * 2).min(Duration::from_secs(5));
continue;
}
};
if !has_current_runtime_task(&task_snapshots) {
if let RuntimeContinuationOutcome::Terminal(record) = &mut outcome {
record.status_code =
whole_process_status_code(record.status_code, &task_snapshots);
record.node_report = json!({
"terminal_event": record.node_report.get("terminal_event"),
"task_snapshots": task_snapshots,
"process_status": process_status,
"process_statuses": process_statuses,
});
}
emit(outcome);
return Ok(());
}
}
let snapshot_request = if inject_snapshot_failure && !snapshot_failure_injected {
snapshot_failure_injected = true;
Err(anyhow!("injected task snapshot transport failure"))
@ -1190,22 +1221,6 @@ pub(crate) fn observe_services_runtime(
}
};
reconnect_delay = Duration::from_millis(100);
if !has_current_runtime_task(&task_snapshots) {
if let Some(mut outcome) =
terminal_runtime_outcome(&coordinator, state, &events, process_status.as_ref())
{
if let RuntimeContinuationOutcome::Terminal(record) = &mut outcome {
record.node_report = json!({
"terminal_event": record.node_report.get("terminal_event"),
"task_snapshots": task_snapshots,
"process_status": process_status,
"process_statuses": process_statuses,
});
}
emit(outcome);
return Ok(());
}
}
if let Some((failed_task, _attempt_id)) = failed_awaiting_action_snapshot(&task_snapshots) {
let failed_task = failed_task.to_owned();
let failed_event = events
@ -1337,9 +1352,7 @@ pub(crate) fn observe_services_runtime(
continue;
}
};
if let Some(mut outcome) =
terminal_runtime_outcome(&coordinator, state, &events, process_status.as_ref())
{
if let Some(mut outcome) = terminal_runtime_outcome(&coordinator, state, &events) {
let task_snapshots = match fetch_task_snapshots_in(current_session, state) {
Ok(snapshots) => snapshots,
Err(snapshot_error) => {
@ -1357,6 +1370,8 @@ pub(crate) fn observe_services_runtime(
};
if !has_current_runtime_task(&task_snapshots) {
if let RuntimeContinuationOutcome::Terminal(record) = &mut outcome {
record.status_code =
whole_process_status_code(record.status_code, &task_snapshots);
record.node_report = json!({
"terminal_event": record.node_report.get("terminal_event"),
"task_snapshots": task_snapshots,
@ -1532,7 +1547,6 @@ fn has_current_runtime_task(task_snapshots: &Value) -> bool {
})
}
#[cfg(test)]
pub(crate) fn whole_process_status_code(
main_status_code: Option<i32>,
task_snapshots: &Value,
@ -1568,75 +1582,66 @@ pub(crate) fn whole_process_status_code(
fn terminal_runtime_outcome(
coordinator: &str,
_state: &AdapterState,
state: &AdapterState,
events: &Value,
process_status: Option<&Value>,
) -> Option<RuntimeContinuationOutcome> {
let final_result = process_status?
.get("final_result")
.and_then(Value::as_str)?;
let status_code = match final_result {
"completed" => 0,
"failed" | "cancelled" => 1,
_ => return None,
};
let event = events
.get("events")
.and_then(Value::as_array)
.and_then(|events| {
events.iter().rev().find(|event| {
event.get("executor").and_then(Value::as_str) == Some("coordinator_main")
})
});
.and_then(Value::as_array)?
.get(state.runtime_event_count..)?
.iter()
.rev()
.find(|event| event.get("executor").and_then(Value::as_str) == Some("coordinator_main"))?;
let status_code = event
.get("status_code")
.and_then(Value::as_i64)
.map(|status| status as i32)
.or_else(
|| match event.get("terminal_state").and_then(Value::as_str) {
Some("completed") => Some(0),
Some("failed" | "cancelled") => Some(1),
_ => None,
},
);
Some(RuntimeContinuationOutcome::Terminal(RuntimeLaunchRecord {
coordinator: coordinator.to_owned(),
node: event
.and_then(|event| event.get("node"))
.get("node")
.and_then(Value::as_str)
.or_else(|| {
process_status
.and_then(|status| status.get("connected_nodes"))
.and_then(Value::as_array)
.and_then(|nodes| nodes.first())
.and_then(Value::as_str)
})
.unwrap_or("coordinator-main")
.to_owned(),
node_report: json!({
"terminal_event": event,
"process_status": process_status,
}),
node_report: json!({ "terminal_event": event }),
task_events: events.clone(),
placed_task_launched: true,
status_code: Some(status_code),
status_code,
stdout_bytes: event
.and_then(|event| event.get("stdout_bytes"))
.get("stdout_bytes")
.and_then(Value::as_u64)
.unwrap_or(0),
stderr_bytes: event
.and_then(|event| event.get("stderr_bytes"))
.get("stderr_bytes")
.and_then(Value::as_u64)
.unwrap_or(0),
stdout_tail: event
.and_then(|event| event.get("stdout_tail"))
.get("stdout_tail")
.and_then(Value::as_str)
.unwrap_or_default()
.to_owned(),
stderr_tail: event
.and_then(|event| event.get("stderr_tail"))
.get("stderr_tail")
.and_then(Value::as_str)
.unwrap_or_default()
.to_owned(),
stdout_truncated: event
.and_then(|event| event.get("stdout_truncated"))
.get("stdout_truncated")
.and_then(Value::as_bool)
.unwrap_or(false),
stderr_truncated: event
.and_then(|event| event.get("stderr_truncated"))
.get("stderr_truncated")
.and_then(Value::as_bool)
.unwrap_or(false),
artifact_path: event
.and_then(|event| event.get("artifact_path"))
.get("artifact_path")
.and_then(Value::as_str)
.map(str::to_owned),
event_count: events
@ -1722,53 +1727,6 @@ mod transactional_launch_tests {
assert!(error.contains("no virtual process was created"));
}
#[test]
fn durable_process_summary_is_terminal_authority_after_event_rotation() {
let state = AdapterState {
runtime_event_count: 10_000,
..AdapterState::default()
};
let completed = terminal_runtime_outcome(
"127.0.0.1:1",
&state,
&json!({ "events": [] }),
Some(&json!({
"process": state.process.as_str(),
"lifecycle": "recent_terminal",
"final_result": "completed",
"connected_nodes": []
})),
)
.expect("the durable summary should terminate observation");
let RuntimeContinuationOutcome::Terminal(completed) = completed else {
panic!("expected terminal outcome");
};
assert_eq!(completed.status_code, Some(0));
let failed = terminal_runtime_outcome(
"127.0.0.1:1",
&state,
&json!({
"events": [{
"executor": "coordinator_main",
"terminal_state": "completed",
"status_code": 0
}]
}),
Some(&json!({
"process": state.process.as_str(),
"lifecycle": "recent_terminal",
"final_result": "failed",
"connected_nodes": []
})),
)
.expect("the aggregate summary should override a successful main event");
let RuntimeContinuationOutcome::Terminal(failed) = failed else {
panic!("expected terminal outcome");
};
assert_eq!(failed.status_code, Some(1));
}
#[test]
fn failed_debug_launch_reconnects_and_aborts_the_process() {
let listener = TcpListener::bind("127.0.0.1:0").unwrap();

View file

@ -2,7 +2,7 @@ use std::collections::{BTreeMap, BTreeSet, HashMap};
use std::io::Read;
use std::path::PathBuf;
use std::process::Stdio;
use std::sync::atomic::{AtomicBool, AtomicU64, Ordering};
use std::sync::atomic::{AtomicBool, Ordering};
use std::sync::{Arc, Mutex};
use std::thread;
use std::time::{Duration, Instant};
@ -46,40 +46,6 @@ use validation::{
resolve_task_export, task_descriptors, verify_environment_digest, verify_source_snapshot,
};
#[derive(Debug)]
struct AssignmentExecutionError {
message: String,
stdout_source_bytes: u64,
stderr_source_bytes: u64,
}
impl std::fmt::Display for AssignmentExecutionError {
fn fmt(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
formatter.write_str(&self.message)
}
}
impl std::error::Error for AssignmentExecutionError {}
fn execution_error_with_log_bytes(
message: impl Into<String>,
stdout_source_bytes: &AtomicU64,
stderr_source_bytes: &AtomicU64,
) -> Box<dyn std::error::Error> {
Box::new(AssignmentExecutionError {
message: message.into(),
stdout_source_bytes: stdout_source_bytes.load(Ordering::Relaxed),
stderr_source_bytes: stderr_source_bytes.load(Ordering::Relaxed),
})
}
pub(crate) fn assignment_error_log_bytes(error: &(dyn std::error::Error + 'static)) -> (u64, u64) {
error
.downcast_ref::<AssignmentExecutionError>()
.map(|error| (error.stdout_source_bytes, error.stderr_source_bytes))
.unwrap_or((0, 0))
}
pub(crate) fn run_verified_wasmtime_assignment(
args: &Args,
task: &RuntimeTask,
@ -129,8 +95,6 @@ pub(crate) fn run_verified_wasmtime_assignment(
return Err("Wasm entrypoint assignment omitted its descriptor export".into());
}
};
let command_stdout_source_bytes = Arc::new(AtomicU64::new(0));
let command_stderr_source_bytes = Arc::new(AtomicU64::new(0));
let (stdout, boundary_result) = match abi {
WasmExportAbi::EntrypointV1 | WasmExportAbi::TaskV1 => {
let invocation = WasmTaskInvocation::new(
@ -138,8 +102,7 @@ pub(crate) fn run_verified_wasmtime_assignment(
task_spec.task_instance.clone(),
task_spec.args.clone(),
);
let result = WasmtimeTaskRuntime::new()?
.run_task_export_verified_with_task_host(
let result = WasmtimeTaskRuntime::new()?.run_task_export_verified_with_task_host(
&module,
expected_bundle_digest,
export,
@ -149,17 +112,8 @@ pub(crate) fn run_verified_wasmtime_assignment(
task,
node_private_key,
&module,
Arc::clone(&command_stdout_source_bytes),
Arc::clone(&command_stderr_source_bytes),
)?),
)
.map_err(|error| {
execution_error_with_log_bytes(
error.to_string(),
&command_stdout_source_bytes,
&command_stderr_source_bytes,
)
})?;
)?;
if std::env::var_os("CLUSTERFLUX_DEBUG_CONTROL_TRACE").is_some() {
eprintln!(
"clusterflux debug control: Wasm assignment returned for task {}",
@ -168,35 +122,17 @@ pub(crate) fn run_verified_wasmtime_assignment(
}
match result.outcome {
WasmTaskOutcome::Completed => {
let boundary = result.result.ok_or_else(|| {
execution_error_with_log_bytes(
"completed Wasm task omitted result",
&command_stdout_source_bytes,
&command_stderr_source_bytes,
)
})?;
let boundary = result.result.ok_or("completed Wasm task omitted result")?;
(
format!(
"{}\n",
serde_json::to_string(&boundary).map_err(|error| {
execution_error_with_log_bytes(
error.to_string(),
&command_stdout_source_bytes,
&command_stderr_source_bytes,
)
})?
),
format!("{}\n", serde_json::to_string(&boundary)?),
Some(boundary),
)
}
WasmTaskOutcome::Failed => {
return Err(execution_error_with_log_bytes(
result
return Err(result
.error
.unwrap_or_else(|| "Wasm task failed without an error".to_owned()),
&command_stdout_source_bytes,
&command_stderr_source_bytes,
))
.unwrap_or_else(|| "Wasm task failed without an error".to_owned())
.into())
}
}
}
@ -204,18 +140,12 @@ pub(crate) fn run_verified_wasmtime_assignment(
let task_id = TaskInstanceId::new(task.task.clone());
let artifacts = TaskArtifactStore::new(task_id.clone(), NodeId::new(args.node.clone()));
let manifest = artifacts.flush();
let stdout_source_bytes = command_stdout_source_bytes
.load(Ordering::Relaxed)
.saturating_add(stdout.len() as u64);
let stderr_source_bytes = command_stderr_source_bytes.load(Ordering::Relaxed);
Ok((
CommandOutput {
virtual_thread: task_id,
status_code: Some(0),
stdout,
stderr: String::new(),
stdout_source_bytes,
stderr_source_bytes,
stdout_truncated: false,
stderr_truncated: false,
log_backpressured: false,
@ -246,8 +176,6 @@ struct CoordinatorWasmTaskHost {
next_handle_id: u64,
handles: Arc<Mutex<HashMap<u64, TaskSpec>>>,
command_status: Arc<Mutex<Option<String>>>,
command_stdout_source_bytes: Arc<AtomicU64>,
command_stderr_source_bytes: Arc<AtomicU64>,
cancellation_requested: Arc<AtomicBool>,
abort_requested: Arc<AtomicBool>,
debug_control: Arc<WasmDebugControl>,
@ -260,8 +188,6 @@ impl CoordinatorWasmTaskHost {
parent: &RuntimeTask,
node_private_key: &str,
module: &[u8],
command_stdout_source_bytes: Arc<AtomicU64>,
command_stderr_source_bytes: Arc<AtomicU64>,
) -> Result<Self, Box<dyn std::error::Error>> {
let task_spec = parent
.task_spec
@ -342,8 +268,6 @@ impl CoordinatorWasmTaskHost {
next_handle_id: 1,
handles,
command_status,
command_stdout_source_bytes,
command_stderr_source_bytes,
cancellation_requested,
abort_requested,
debug_control,
@ -369,16 +293,7 @@ impl CoordinatorWasmTaskHost {
runtime_task_from_assignment(assignment).map_err(|error| error.to_string())?;
let execution =
run_verified_wasmtime_assignment(&self.args, &runtime_task, &self.node_private_key);
let (
terminal_state,
status_code,
stdout,
stderr,
stdout_source_bytes,
stderr_source_bytes,
result,
retained,
) = match execution {
let (terminal_state, status_code, stdout, stderr, result, retained) = match execution {
Ok((output, _manifest, result)) => {
let retained = retained_result_artifact(
self.args.project_root.as_deref(),
@ -391,40 +306,20 @@ impl CoordinatorWasmTaskHost {
output.status_code,
output.stdout,
output.stderr,
output.stdout_source_bytes,
output.stderr_source_bytes,
result,
retained,
),
Err(error) => ("failed", Some(1), String::new(), error, None, None),
}
}
Err(error) => (
"failed",
Some(1),
String::new(),
error.clone(),
output.stdout_source_bytes,
output
.stderr_source_bytes
.saturating_add(error.len() as u64),
error.to_string(),
None,
None,
),
}
}
Err(error) => {
let (stdout_source_bytes, stderr_source_bytes) =
assignment_error_log_bytes(error.as_ref());
let error = error.to_string();
(
"failed",
Some(1),
String::new(),
error.clone(),
stdout_source_bytes,
stderr_source_bytes.saturating_add(error.len() as u64),
None,
None,
)
}
};
let artifact_path = retained
.as_ref()
@ -444,8 +339,8 @@ impl CoordinatorWasmTaskHost {
"task": runtime_task.task,
"terminal_state": terminal_state,
"status_code": status_code,
"stdout_bytes": stdout_source_bytes,
"stderr_bytes": stderr_source_bytes,
"stdout_bytes": stdout.len(),
"stderr_bytes": stderr.len(),
"stdout_tail": stdout,
"stderr_tail": stderr,
"stdout_truncated": false,
@ -781,7 +676,6 @@ impl WasmTaskHost for CoordinatorWasmTaskHost {
let mut runner = CoordinatorControlledProcessRunner::new(
self,
Duration::from_millis(request.timeout_ms),
configured_secrets.clone(),
);
let output = LinuxRootlessPodmanBackend
.execute_local_checkout_task(

View file

@ -1,83 +1,4 @@
use super::*;
use std::sync::mpsc::{self, Receiver, SyncSender};
struct LiveLogChunk {
stream: &'static str,
offset: u64,
source_bytes: u64,
bytes: Vec<u8>,
truncated: bool,
}
fn append_bounded_tail(tail: &mut Vec<u8>, bytes: &[u8], maximum: usize) {
if maximum == 0 {
tail.clear();
return;
}
if bytes.len() >= maximum {
tail.clear();
tail.extend_from_slice(&bytes[bytes.len() - maximum..]);
return;
}
let overflow = tail
.len()
.saturating_add(bytes.len())
.saturating_sub(maximum);
if overflow > 0 {
tail.drain(..overflow);
}
tail.extend_from_slice(bytes);
}
fn redact_safe_live_log_prefix(
pending: &[u8],
configured_secrets: &[String],
final_chunk: bool,
) -> Option<(usize, String)> {
if pending.is_empty() {
return None;
}
let secret_bytes = configured_secrets
.iter()
.filter(|secret| secret.len() >= 4)
.map(String::as_bytes)
.collect::<Vec<_>>();
let maximum_secret_bytes = secret_bytes
.iter()
.map(|secret| secret.len())
.max()
.unwrap_or(0);
if !final_chunk && pending.len() <= maximum_secret_bytes {
return None;
}
let mut consumed = if final_chunk {
pending.len()
} else {
pending.len() - maximum_secret_bytes
};
loop {
let previous = consumed;
for secret in &secret_bytes {
for start in 0..=pending.len().saturating_sub(secret.len()) {
let end = start + secret.len();
if start < consumed && end > consumed && &pending[start..end] == *secret {
consumed = end;
}
}
}
if consumed == previous {
break;
}
}
if consumed == 0 {
return None;
}
let text = redact_configured_values(
String::from_utf8_lossy(&pending[..consumed]).into_owned(),
configured_secrets,
);
Some((consumed, text))
}
pub(super) struct CoordinatorControlledProcessRunner {
pub(super) args: Args,
@ -86,20 +7,13 @@ pub(super) struct CoordinatorControlledProcessRunner {
pub(super) node_private_key: String,
pub(super) debug_control: Arc<WasmDebugControl>,
pub(super) command_status: Arc<Mutex<Option<String>>>,
pub(super) stdout_source_bytes: Arc<AtomicU64>,
pub(super) stderr_source_bytes: Arc<AtomicU64>,
pub(super) timeout: Duration,
pub(super) configured_secrets: Vec<String>,
}
impl CoordinatorControlledProcessRunner {
const MAX_CAPTURE_BYTES: usize = 256 * 1024 + 1;
pub(super) fn new(
host: &CoordinatorWasmTaskHost,
timeout: Duration,
configured_secrets: Vec<String>,
) -> Self {
pub(super) fn new(host: &CoordinatorWasmTaskHost, timeout: Duration) -> Self {
Self {
args: host.args.clone(),
process: host.process.clone(),
@ -107,10 +21,7 @@ impl CoordinatorControlledProcessRunner {
node_private_key: host.node_private_key.clone(),
debug_control: Arc::clone(&host.debug_control),
command_status: Arc::clone(&host.command_status),
stdout_source_bytes: Arc::clone(&host.command_stdout_source_bytes),
stderr_source_bytes: Arc::clone(&host.command_stderr_source_bytes),
timeout,
configured_secrets,
}
}
@ -290,18 +201,10 @@ impl CoordinatorControlledProcessRunner {
fn drain_bounded(
mut reader: impl Read + Send + 'static,
maximum: usize,
stream: &'static str,
sender: SyncSender<LiveLogChunk>,
configured_secrets: Vec<String>,
source_bytes_total: Arc<AtomicU64>,
) -> thread::JoinHandle<Result<Vec<u8>, String>> {
thread::spawn(move || {
let mut captured = Vec::new();
let mut buffer = [0_u8; 16 * 1024];
let stream_base = source_bytes_total.load(Ordering::Relaxed);
let mut source_bytes_read = 0_u64;
let mut pending_offset = stream_base;
let mut pending = Vec::new();
loop {
let count = reader
.read(&mut buffer)
@ -309,128 +212,12 @@ impl CoordinatorControlledProcessRunner {
if count == 0 {
break;
}
let _ = source_bytes_total.fetch_update(
Ordering::Relaxed,
Ordering::Relaxed,
|current| Some(current.saturating_add(count as u64)),
);
source_bytes_read = source_bytes_read.saturating_add(count as u64);
append_bounded_tail(&mut captured, &buffer[..count], maximum);
pending.extend_from_slice(&buffer[..count]);
if let Some((consumed, text)) =
redact_safe_live_log_prefix(&pending, &configured_secrets, false)
{
let _ = sender.try_send(LiveLogChunk {
stream,
offset: pending_offset,
source_bytes: consumed as u64,
bytes: text.into_bytes(),
truncated: false,
});
pending.drain(..consumed);
pending_offset = pending_offset.saturating_add(consumed as u64);
}
}
if let Some((consumed, text)) =
redact_safe_live_log_prefix(&pending, &configured_secrets, true)
{
let _ = sender.try_send(LiveLogChunk {
stream,
offset: pending_offset,
source_bytes: consumed as u64,
bytes: text.into_bytes(),
truncated: false,
});
}
if source_bytes_read > maximum as u64 {
let _ = sender.try_send(LiveLogChunk {
stream,
offset: stream_base.saturating_add(source_bytes_read),
source_bytes: 0,
bytes: b"[log output truncated at node capture limit]".to_vec(),
truncated: true,
});
let remaining = maximum.saturating_sub(captured.len());
captured.extend_from_slice(&buffer[..count.min(remaining)]);
}
Ok(captured)
})
}
fn spawn_live_log_reporter(&self, receiver: Receiver<LiveLogChunk>) -> thread::JoinHandle<()> {
let args = self.args.clone();
let process = self.process.clone();
let task = self.task.clone();
let node_private_key = self.node_private_key.clone();
let configured_secrets = self.configured_secrets.clone();
let command_status = Arc::clone(&self.command_status);
thread::spawn(move || {
let mut log_session = None;
let mut delivery_available = true;
while let Ok(chunk) = receiver.recv() {
if !delivery_available {
continue;
}
let mut text = String::from_utf8_lossy(&chunk.bytes).into_owned();
text = redact_configured_values(text, &configured_secrets);
let mut delivered = false;
for _ in 0..2 {
if log_session.is_none() {
log_session = CoordinatorSession::connect_with_timeouts(
&args.coordinator,
Duration::from_millis(500),
Duration::from_millis(500),
)
.ok();
}
let Some(session) = log_session.as_mut() else {
continue;
};
let request = signed_node_request_json(
&args,
&node_private_key,
"report_task_log_chunk",
serde_json::json!({
"type": "report_task_log_chunk",
"tenant": &args.tenant,
"project": &args.project,
"process": &process,
"node": &args.node,
"task": &task,
"stream": chunk.stream,
"offset": chunk.offset,
"source_bytes": chunk.source_bytes,
"text": &text,
"truncated": chunk.truncated,
}),
);
let result = match request {
Ok(request) => session
.request(request)
.map(|_| ())
.map_err(|error| error.to_string()),
Err(error) => Err(error.to_string()),
};
match result {
Ok(()) => {
delivered = true;
break;
}
Err(_) => {
log_session = None;
}
}
}
if !delivered {
delivery_available = false;
if let Ok(mut current) = command_status.lock() {
*current = Some(
"live log delivery was interrupted; final bounded output remains available"
.to_owned(),
);
}
}
}
})
}
}
impl ProcessRunner for CoordinatorControlledProcessRunner {
@ -458,17 +245,12 @@ impl ProcessRunner for CoordinatorControlledProcessRunner {
command.program,
command.args.join(" ")
));
let (live_log_sender, live_log_receiver) = mpsc::sync_channel(64);
let stdout = Self::drain_bounded(
child
.stdout
.take()
.ok_or_else(|| BackendError::Command("command stdout pipe missing".to_owned()))?,
Self::MAX_CAPTURE_BYTES,
"stdout",
live_log_sender.clone(),
self.configured_secrets.clone(),
Arc::clone(&self.stdout_source_bytes),
);
let stderr = Self::drain_bounded(
child
@ -476,19 +258,11 @@ impl ProcessRunner for CoordinatorControlledProcessRunner {
.take()
.ok_or_else(|| BackendError::Command("command stderr pipe missing".to_owned()))?,
Self::MAX_CAPTURE_BYTES,
"stderr",
live_log_sender,
self.configured_secrets.clone(),
Arc::clone(&self.stderr_source_bytes),
);
let live_log_reporter = self.spawn_live_log_reporter(live_log_receiver);
let mut session = match CoordinatorSession::connect(&self.args.coordinator) {
Ok(session) => session,
Err(error) => {
Self::terminate_execution(&mut child, podman_container.as_deref());
let _ = stdout.join();
let _ = stderr.join();
let _ = live_log_reporter.join();
return Err(BackendError::Command(format!(
"establish execution control channel: {error}"
)));
@ -503,9 +277,6 @@ impl ProcessRunner for CoordinatorControlledProcessRunner {
Ok(None) => {}
Err(error) => {
Self::terminate_execution(&mut child, podman_container.as_deref());
let _ = stdout.join();
let _ = stderr.join();
let _ = live_log_reporter.join();
return Err(BackendError::Command(error.to_string()));
}
}
@ -518,7 +289,6 @@ impl ProcessRunner for CoordinatorControlledProcessRunner {
Self::terminate_execution(&mut child, podman_container.as_deref());
let _ = stdout.join();
let _ = stderr.join();
let _ = live_log_reporter.join();
return Err(BackendError::Command(format!(
"native command exceeded wall-clock timeout of {} ms",
self.timeout.as_millis()
@ -533,7 +303,6 @@ impl ProcessRunner for CoordinatorControlledProcessRunner {
Self::terminate_execution(&mut child, podman_container.as_deref());
let _ = stdout.join();
let _ = stderr.join();
let _ = live_log_reporter.join();
return Err(BackendError::Cancelled(
"coordinator requested cancellation or abort".to_owned(),
));
@ -543,7 +312,6 @@ impl ProcessRunner for CoordinatorControlledProcessRunner {
Self::terminate_execution(&mut child, podman_container.as_deref());
let _ = stdout.join();
let _ = stderr.join();
let _ = live_log_reporter.join();
return Err(error);
}
}
@ -592,9 +360,6 @@ impl ProcessRunner for CoordinatorControlledProcessRunner {
.join()
.map_err(|_| BackendError::Command("stderr reader panicked".to_owned()))?
.map_err(BackendError::Command)?;
live_log_reporter
.join()
.map_err(|_| BackendError::Command("live log reporter panicked".to_owned()))?;
self.set_command_status(format!(
"native command exited with status {:?}",
status.code()
@ -606,52 +371,3 @@ impl ProcessRunner for CoordinatorControlledProcessRunner {
})
}
}
#[cfg(test)]
mod tests {
use super::{redact_safe_live_log_prefix, CoordinatorControlledProcessRunner};
use std::io::Cursor;
use std::sync::atomic::{AtomicU64, Ordering};
use std::sync::{mpsc, Arc};
#[test]
fn live_log_redaction_holds_boundaries_until_split_secrets_are_complete() {
let secrets = vec!["correct-horse".to_owned()];
let mut pending = b"prefix correct-".to_vec();
let (consumed, first) = redact_safe_live_log_prefix(&pending, &secrets, false).unwrap();
pending.drain(..consumed);
pending.extend_from_slice(b"horse suffix");
let (_, second) = redact_safe_live_log_prefix(&pending, &secrets, true).unwrap();
let combined = format!("{first}{second}");
assert_eq!(combined, "prefix [REDACTED] suffix");
assert!(!combined.contains("correct-"));
assert!(!combined.contains("horse"));
}
#[test]
fn bounded_capture_retains_the_real_tail_and_complete_source_byte_count() {
let source_bytes = Arc::new(AtomicU64::new(5));
let (sender, receiver) = mpsc::sync_channel(8);
let reader = CoordinatorControlledProcessRunner::drain_bounded(
Cursor::new(b"0123456789abcdefghijklmnopqrstuv".to_vec()),
8,
"stdout",
sender,
Vec::new(),
Arc::clone(&source_bytes),
);
let captured = reader.join().unwrap().unwrap();
let chunks = receiver.into_iter().collect::<Vec<_>>();
assert_eq!(captured, b"opqrstuv");
assert_eq!(source_bytes.load(Ordering::Relaxed), 37);
assert_eq!(
chunks.iter().map(|chunk| chunk.source_bytes).sum::<u64>(),
32
);
assert_eq!(chunks[0].offset, 5);
assert_eq!(chunks.last().unwrap().offset, 37);
assert!(chunks.iter().any(|chunk| chunk.truncated));
}
}

View file

@ -48,10 +48,7 @@ fn test_controlled_runner(
),
debug_control: Arc::new(WasmDebugControl::default()),
command_status: Arc::new(Mutex::new(None)),
stdout_source_bytes: Arc::new(AtomicU64::new(0)),
stderr_source_bytes: Arc::new(AtomicU64::new(0)),
timeout,
configured_secrets: Vec::new(),
}
}
@ -92,10 +89,7 @@ fn controlled_process_runner_kills_running_group_when_abort_is_polled() {
),
debug_control: Arc::new(WasmDebugControl::default()),
command_status: Arc::new(Mutex::new(None)),
stdout_source_bytes: Arc::new(AtomicU64::new(0)),
stderr_source_bytes: Arc::new(AtomicU64::new(0)),
timeout: Duration::from_secs(30),
configured_secrets: Vec::new(),
};
let started = Instant::now();
let error = runner

View file

@ -1,6 +1,6 @@
use clusterflux_core::{
CommandInvocation, Digest, NativeCommandPolicy, NodeId, TaskInstanceId, VfsObject, VfsOverlay,
VfsPath,
CommandInvocation, Digest, LogBuffer, NativeCommandPolicy, NodeId, TaskInstanceId, VfsObject,
VfsOverlay, VfsPath,
};
use serde::{Deserialize, Serialize};
@ -42,8 +42,6 @@ pub struct CommandOutput {
pub status_code: Option<i32>,
pub stdout: String,
pub stderr: String,
pub stdout_source_bytes: u64,
pub stderr_source_bytes: u64,
pub stdout_truncated: bool,
pub stderr_truncated: bool,
pub log_backpressured: bool,
@ -85,8 +83,6 @@ impl LocalCommandExecutor {
&output.stderr,
max_log_bytes,
);
let stdout_source_bytes = output.stdout.len() as u64;
let stderr_source_bytes = output.stderr.len() as u64;
let staged_artifact = if let Some(path) = command.stage_stdout_as {
Some(overlay.write(
path,
@ -102,8 +98,6 @@ impl LocalCommandExecutor {
status_code: output.status.code(),
stdout: logs.stdout,
stderr: logs.stderr,
stdout_source_bytes,
stderr_source_bytes,
stdout_truncated: logs.stdout_truncated,
stderr_truncated: logs.stderr_truncated,
log_backpressured: logs.backpressured,
@ -113,20 +107,24 @@ impl LocalCommandExecutor {
}
pub(super) fn capture_command_logs(
_task: &TaskInstanceId,
task: &TaskInstanceId,
stdout: &[u8],
stderr: &[u8],
max_log_bytes: usize,
) -> CapturedCommandLogs {
let stdout_truncated = stdout.len() > max_log_bytes;
let stderr_truncated = stderr.len() > max_log_bytes;
let stdout_start = stdout.len().saturating_sub(max_log_bytes);
let stderr_start = stderr.len().saturating_sub(max_log_bytes);
let mut logs = LogBuffer::new(max_log_bytes);
logs.push(task.clone(), stdout);
logs.push(task.clone(), stderr);
let records = logs.records();
let stdout_record = &records[0];
let stderr_record = &records[1];
debug_assert_eq!(&stdout_record.task, task);
debug_assert_eq!(&stderr_record.task, task);
CapturedCommandLogs {
stdout: String::from_utf8_lossy(&stdout[stdout_start..]).into_owned(),
stderr: String::from_utf8_lossy(&stderr[stderr_start..]).into_owned(),
stdout_truncated,
stderr_truncated,
backpressured: stdout_truncated || stderr_truncated,
stdout: String::from_utf8_lossy(&stdout_record.bytes).into_owned(),
stderr: String::from_utf8_lossy(&stderr_record.bytes).into_owned(),
stdout_truncated: stdout_record.truncated,
stderr_truncated: stderr_record.truncated,
backpressured: logs.backpressured(),
}
}

View file

@ -3,7 +3,6 @@ use clusterflux_control::endpoint_identity;
use clusterflux_control::ControlSession;
use clusterflux_core::coordinator_wire_request;
use serde_json::Value;
use std::time::Duration;
pub(crate) struct CoordinatorSession {
inner: ControlSession,
@ -16,16 +15,6 @@ impl CoordinatorSession {
})
}
pub(crate) fn connect_with_timeouts(
addr: &str,
connect_timeout: Duration,
io_timeout: Duration,
) -> Result<Self, Box<dyn std::error::Error>> {
Ok(Self {
inner: ControlSession::connect_with_timeouts(addr, connect_timeout, io_timeout)?,
})
}
pub(crate) fn request(&mut self, value: Value) -> Result<Value, Box<dyn std::error::Error>> {
let request_id = format!("node-{}", self.inner.requests() + 1);
let wire_request = coordinator_wire_request(request_id, value);

View file

@ -11,7 +11,7 @@ use clusterflux_core::{
};
use serde_json::{json, Value};
use crate::assignment_runner::{assignment_error_log_bytes, run_verified_wasmtime_assignment};
use crate::assignment_runner::run_verified_wasmtime_assignment;
#[cfg(test)]
use crate::coordinator_session::control_endpoint_identity;
use crate::coordinator_session::CoordinatorSession;
@ -464,8 +464,6 @@ fn run_runtime_task(
capability_report,
debug_command,
node_private_key,
0,
0,
);
}
@ -500,13 +498,9 @@ fn run_runtime_task(
debug_command,
node_private_key,
&error,
output.stdout_source_bytes,
output.stderr_source_bytes,
),
},
Err(error) => {
let (stdout_source_bytes, stderr_source_bytes) =
assignment_error_log_bytes(error.as_ref());
let error = error.to_string();
if error.contains("task execution cancelled:") {
record_cancelled_task(
@ -518,8 +512,6 @@ fn run_runtime_task(
capability_report,
debug_command,
node_private_key,
stdout_source_bytes,
stderr_source_bytes,
)
} else {
record_failed_task(
@ -532,8 +524,6 @@ fn run_runtime_task(
debug_command,
node_private_key,
&error,
stdout_source_bytes,
stderr_source_bytes,
)
}
}

View file

@ -741,7 +741,7 @@ mod tests {
}
#[test]
fn linux_backend_retains_final_log_tail_without_truncating_staged_artifact_bytes() {
fn linux_backend_caps_logs_without_truncating_staged_artifact_bytes() {
let invocation = CommandInvocation {
program: "cargo".to_owned(),
args: vec!["build".to_owned()],
@ -774,7 +774,7 @@ mod tests {
.unwrap();
assert_eq!(output.virtual_thread, TaskInstanceId::from("compile-linux"));
assert_eq!(output.stdout, "cdef");
assert_eq!(output.stdout, "abcd");
assert!(output.stdout_truncated);
assert!(output.log_backpressured);
assert_eq!(output.staged_artifact.as_ref().unwrap().size, 6);
@ -992,7 +992,7 @@ mod tests {
#[cfg(unix)]
#[test]
fn local_command_executor_retains_each_stream_tail_and_reports_backpressure() {
fn local_command_executor_caps_logs_and_reports_backpressure_by_virtual_thread() {
let executor = LocalCommandExecutor {
node: clusterflux_core::NodeId::from("node"),
hosted_control_plane: false,
@ -1021,10 +1021,10 @@ mod tests {
.unwrap();
assert_eq!(output.virtual_thread, TaskInstanceId::from("compile-linux"));
assert_eq!(output.stdout, "cdef");
assert_eq!(output.stderr, "err");
assert_eq!(output.stdout, "abcd");
assert_eq!(output.stderr, "");
assert!(output.stdout_truncated);
assert!(!output.stderr_truncated);
assert!(output.stderr_truncated);
assert!(output.log_backpressured);
}

View file

@ -8,34 +8,5 @@ mod task_artifacts;
mod task_reports;
fn main() -> Result<(), Box<dyn std::error::Error>> {
let raw_args = std::env::args().skip(1).collect::<Vec<_>>();
match raw_args.as_slice() {
[flag] if matches!(flag.as_str(), "--version" | "-V") => {
println!("clusterflux-node {}", env!("CARGO_PKG_VERSION"));
return Ok(());
}
[flag] if matches!(flag.as_str(), "--help" | "-h") => {
println!(
"Clusterflux node worker.\n\n\
Usage: clusterflux-node --coordinator <URL> [OPTIONS]\n\n\
Options:\n \
--coordinator <URL>\n \
--tenant <TENANT> [default: tenant]\n \
--project-id <PROJECT> [default: project]\n \
--node <NODE> [default: node]\n \
--project-root <PATH>\n \
--enrollment-grant <GRANT>\n \
--public-key <KEY>\n \
--worker\n \
--emit-ready\n \
--control-poll-ms <MILLISECONDS>\n \
--assignment-poll-ms <MILLISECONDS> [default: 500]\n \
-h, --help\n \
-V, --version"
);
return Ok(());
}
_ => {}
}
daemon::run()
}

View file

@ -47,8 +47,8 @@ pub(crate) fn record_completed_task(
"process": &task.process,
"node": &args.node,
"task": &task.task,
"stdout_bytes": output.stdout_source_bytes,
"stderr_bytes": output.stderr_source_bytes,
"stdout_bytes": output.stdout.len(),
"stderr_bytes": output.stderr.len(),
"stdout_tail": &output.stdout,
"stderr_tail": &output.stderr,
"stdout_truncated": output.stdout_truncated,
@ -85,8 +85,8 @@ pub(crate) fn record_completed_task(
"node": &args.node,
"task": &task.task,
"status_code": output.status_code,
"stdout_bytes": output.stdout_source_bytes,
"stderr_bytes": output.stderr_source_bytes,
"stdout_bytes": output.stdout.len(),
"stderr_bytes": output.stderr.len(),
"stdout_tail": &output.stdout,
"stderr_tail": &output.stderr,
"stdout_truncated": output.stdout_truncated,
@ -123,11 +123,8 @@ pub(crate) fn record_failed_task(
debug_command: Value,
node_private_key: &str,
error: &str,
stdout_source_bytes: u64,
command_stderr_source_bytes: u64,
) -> Result<Value, Box<dyn std::error::Error>> {
let error = bounded_runtime_error(error);
let stderr_source_bytes = command_stderr_source_bytes.saturating_add(error.len() as u64);
let log_event = session.request(signed_node_request_json(
args,
node_private_key,
@ -139,8 +136,8 @@ pub(crate) fn record_failed_task(
"process": &task.process,
"node": &args.node,
"task": &task.task,
"stdout_bytes": stdout_source_bytes,
"stderr_bytes": stderr_source_bytes,
"stdout_bytes": 0,
"stderr_bytes": error.len(),
"stdout_tail": "",
"stderr_tail": &error,
"stdout_truncated": false,
@ -178,8 +175,8 @@ pub(crate) fn record_failed_task(
"task": &task.task,
"terminal_state": "failed",
"status_code": -1,
"stdout_bytes": stdout_source_bytes,
"stderr_bytes": stderr_source_bytes,
"stdout_bytes": 0,
"stderr_bytes": error.len(),
"stdout_tail": "",
"stderr_tail": &error,
"stdout_truncated": false,
@ -192,8 +189,6 @@ pub(crate) fn record_failed_task(
Ok(failed_node_report(
task,
&error,
stdout_source_bytes,
stderr_source_bytes,
registration,
heartbeat,
capability_report,
@ -215,8 +210,6 @@ pub(crate) fn record_cancelled_task(
capability_report: Value,
debug_command: Value,
node_private_key: &str,
stdout_source_bytes: u64,
stderr_source_bytes: u64,
) -> Result<Value, Box<dyn std::error::Error>> {
let recorded = session.request(signed_node_request_json(
args,
@ -231,12 +224,12 @@ pub(crate) fn record_cancelled_task(
"task": &task.task,
"terminal_state": "cancelled",
"status_code": null,
"stdout_bytes": stdout_source_bytes,
"stderr_bytes": stderr_source_bytes,
"stdout_bytes": 0,
"stderr_bytes": 0,
"stdout_tail": "",
"stderr_tail": "",
"stdout_truncated": stdout_source_bytes > 0,
"stderr_truncated": stderr_source_bytes > 0,
"stdout_truncated": false,
"stderr_truncated": false,
"artifact_path": null,
"artifact_digest": null,
"artifact_size_bytes": null,
@ -245,8 +238,6 @@ pub(crate) fn record_cancelled_task(
)?)?;
Ok(cancelled_node_report(
task,
stdout_source_bytes,
stderr_source_bytes,
registration,
heartbeat,
capability_report,
@ -290,8 +281,8 @@ pub(crate) fn completed_node_report(
"virtual_thread": output.virtual_thread,
"terminal_state": if output.status_code == Some(0) { "completed" } else { "failed" },
"status_code": output.status_code,
"stdout_bytes": output.stdout_source_bytes,
"stderr_bytes": output.stderr_source_bytes,
"stdout_bytes": output.stdout.len(),
"stderr_bytes": output.stderr.len(),
"stdout_tail": &output.stdout,
"stderr_tail": &output.stderr,
"stdout_truncated": output.stdout_truncated,
@ -314,8 +305,6 @@ pub(crate) fn completed_node_report(
#[allow(clippy::too_many_arguments)]
pub(crate) fn cancelled_node_report(
task: &RuntimeTask,
stdout_source_bytes: u64,
stderr_source_bytes: u64,
registration_response: Value,
heartbeat_response: Value,
capability_response: Value,
@ -329,12 +318,12 @@ pub(crate) fn cancelled_node_report(
"virtual_thread": &task.task,
"terminal_state": "cancelled",
"status_code": null,
"stdout_bytes": stdout_source_bytes,
"stderr_bytes": stderr_source_bytes,
"stdout_bytes": 0,
"stderr_bytes": 0,
"stdout_tail": "",
"stderr_tail": "",
"stdout_truncated": stdout_source_bytes > 0,
"stderr_truncated": stderr_source_bytes > 0,
"stdout_truncated": false,
"stderr_truncated": false,
"log_backpressured": false,
"staged_artifact": null,
"large_bytes_uploaded": false,
@ -354,8 +343,6 @@ pub(crate) fn cancelled_node_report(
pub(crate) fn failed_node_report(
task: &RuntimeTask,
error: &str,
stdout_source_bytes: u64,
stderr_source_bytes: u64,
registration_response: Value,
heartbeat_response: Value,
capability_response: Value,
@ -370,8 +357,8 @@ pub(crate) fn failed_node_report(
"virtual_thread": &task.task,
"terminal_state": "failed",
"status_code": -1,
"stdout_bytes": stdout_source_bytes,
"stderr_bytes": stderr_source_bytes,
"stdout_bytes": 0,
"stderr_bytes": error.len(),
"stdout_tail": "",
"stderr_tail": error,
"stdout_truncated": false,
@ -390,40 +377,3 @@ pub(crate) fn failed_node_report(
"coordinator_response": coordinator_response,
})
}
#[cfg(test)]
mod tests {
use super::*;
use clusterflux_core::TaskInstanceId;
#[test]
fn completed_report_uses_source_counts_instead_of_bounded_tail_lengths() {
let report = completed_node_report(
CommandOutput {
virtual_thread: TaskInstanceId::from("task"),
status_code: Some(0),
stdout: "bounded tail".to_owned(),
stderr: String::new(),
stdout_source_bytes: 519,
stderr_source_bytes: 0,
stdout_truncated: true,
stderr_truncated: false,
log_backpressured: false,
staged_artifact: None,
},
false,
Value::Null,
Value::Null,
Value::Null,
Value::Null,
Value::Null,
Value::Null,
Value::Null,
Value::Null,
1,
);
assert_eq!(report["stdout_bytes"], 519);
assert_eq!(report["stdout_tail"], "bounded tail");
}
}

View file

@ -34,5 +34,5 @@ tenant, project, process, artifact, and policy context.
The reverse stream counts framing, base64 expansion, failed bytes, and abandoned
transfers. The CLI verifies the final digest. Hosted policy may impose
per-project, per-tenant, and per-account size, concurrency, and period limits.
There is no hosted global circuit breaker, and one tenant's period usage does
not consume a shared customer byte quota.
Operators may disable relay traffic as an emergency safety control; one tenant's
period usage does not consume a shared customer byte quota.

View file

@ -11,9 +11,6 @@ cargo install --path crates/clusterflux-node --bin clusterflux-node
cargo install --path crates/clusterflux-dap --bin clusterflux-debug-dap
~~~
On NixOS or another system with Nix, the equivalent package is available with
`nix profile install .#clusterflux-tools`.
Install rootless Podman on each Linux node that will execute container-backed
environments.

View file

@ -9,18 +9,6 @@
forAllSystems = nixpkgs.lib.genAttrs systems;
in
{
packages = forAllSystems (system:
let
pkgs = import nixpkgs { inherit system; };
publicPackages = import ./packages.nix { inherit pkgs self; };
privatePackages =
if builtins.pathExists ./web/packages.nix then
import ./web/packages.nix { inherit pkgs self; }
else
{ };
in
publicPackages // privatePackages);
devShells = forAllSystems (system:
let
pkgs = import nixpkgs { inherit system; };

View file

@ -1,74 +0,0 @@
{ pkgs, self }:
let
clusterflux-tools = pkgs.rustPlatform.buildRustPackage {
pname = "clusterflux-tools";
version = "0.1.0";
src = self;
cargoLock.lockFile = ./Cargo.lock;
nativeBuildInputs = [
pkgs.git
pkgs.lld
pkgs.makeWrapper
];
cargoBuildFlags = [
"--package"
"clusterflux-cli"
"--package"
"clusterflux-node"
"--package"
"clusterflux-coordinator"
"--package"
"clusterflux-dap"
];
cargoTestFlags = [
"--package"
"clusterflux-cli"
"--package"
"clusterflux-node"
"--package"
"clusterflux-coordinator"
"--package"
"clusterflux-dap"
];
NIX_BUILD_CORES = "2";
RUST_MIN_STACK = "1073741824";
postInstall = ''
test -x "$out/bin/clusterflux"
test -x "$out/bin/clusterflux-node"
test -x "$out/bin/clusterflux-coordinator"
test -x "$out/bin/clusterflux-debug-dap"
for command in \
clusterflux \
clusterflux-node \
clusterflux-coordinator \
clusterflux-debug-dap
do
${pkgs.coreutils}/bin/timeout 5 "$out/bin/$command" --version >/dev/null
${pkgs.coreutils}/bin/timeout 5 "$out/bin/$command" --help >/dev/null
done
'';
postFixup =
let
runtimePath = pkgs.lib.makeBinPath [
pkgs.cargo
pkgs.git
pkgs.lld
pkgs.rustc
];
in
''
wrapProgram "$out/bin/clusterflux" --prefix PATH : ${runtimePath}
wrapProgram "$out/bin/clusterflux-node" --prefix PATH : ${runtimePath}
wrapProgram "$out/bin/clusterflux-debug-dap" --prefix PATH : ${runtimePath}
'';
meta = {
description = "Clusterflux CLI, node, coordinator, and debugger adapter";
mainProgram = "clusterflux";
};
};
in
{
inherit clusterflux-tools;
clusterflux = clusterflux-tools;
default = clusterflux-tools;
}