Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
31 changes: 22 additions & 9 deletions Cargo.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

3 changes: 2 additions & 1 deletion Cargo.toml
Original file line number Diff line number Diff line change
Expand Up @@ -29,7 +29,7 @@ sha256 = "1.6"
tokio = { version = "1", features = ["macros", "rt"] }
tokio-tar = { package = "astral-tokio-tar", version = "0.6.2" }
tokio-util = "0.7.18"
md5 = "0.8"
crc-fast = { version = "1.9.0", default-features = false, features = ["std"] }
base64 = "0.22.1"
async-compression = { version = "0.4.42", features = ["tokio", "gzip"] }
schemars = "1.2.1"
Expand Down Expand Up @@ -69,6 +69,7 @@ uuid = { version = "1.23.1", features = ["v4"] }
which = "8.0.2"
crc32fast = "1.5.0"
samply = { path = "crates/samply-codspeed/samply" }
bytes = "1"

# Memory profiling (memtrack) and the capability handling around it are Linux-only.
[target.'cfg(target_os = "linux")'.dependencies]
Expand Down
3 changes: 1 addition & 2 deletions src/run_environment/provider.rs
Original file line number Diff line number Diff line change
Expand Up @@ -136,8 +136,7 @@ pub trait RunEnvironmentProvider {
tokenless: api_client.token().is_none(),
repository_provider: self.get_repository_provider(),
run_environment_metadata,
profile_md5: profile_archive.hash.clone(),
profile_encoding: profile_archive.content.encoding(),
profile_archive: profile_archive.metadata.clone(),
commit_hash,
allow_empty: config.allow_empty,
runner: Runner {
Expand Down
82 changes: 78 additions & 4 deletions src/upload/interfaces.rs
Original file line number Diff line number Diff line change
@@ -1,20 +1,21 @@
use std::collections::BTreeMap;

use serde::{Deserialize, Serialize};

use crate::executor::ExecutorName;
use crate::instruments::InstrumentName;
use crate::run_environment::{RepositoryProvider, RunEnvironment, RunEnvironmentMetadata, RunPart};
use crate::system::SystemInfo;

pub const LATEST_UPLOAD_METADATA_VERSION: u32 = 11;
pub const LATEST_UPLOAD_METADATA_VERSION: u32 = 12;

#[derive(Serialize, Debug)]
#[serde(rename_all = "camelCase")]
pub struct UploadMetadata {
pub repository_provider: RepositoryProvider,
pub version: Option<u32>,
pub tokenless: bool,
pub profile_md5: String,
pub profile_encoding: Option<String>,
pub profile_archive: ProfileMetadata,
pub runner: Runner,
pub run_environment: RunEnvironment,
pub run_part: Option<RunPart>,
Expand All @@ -24,6 +25,22 @@ pub struct UploadMetadata {
pub run_environment_metadata: RunEnvironmentMetadata,
}

/// Profile archive, uploaded as an S3 multipart upload in consecutive `part_size`
/// chunks, the last one holding the remainder. S3 requires parts of 5 MiB to 5 GiB
/// (the last one excepted from the minimum), and at most 10,000 of them.
#[derive(Serialize, Debug, Clone, PartialEq)]
#[serde(rename_all = "camelCase")]
pub struct ProfileMetadata {
/// `Content-Encoding` of the archive, such as `gzip`
pub encoding: Option<String>,
pub size: u64,
/// Base64 big-endian CRC64NVME of the whole archive
pub crc64nvme: String,
pub part_size: u64,
Comment on lines +36 to +39

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

potential nitpick: wouldn't usize here make more sense?

/// Base64 big-endian CRC64NVME of each part, in upload order
pub part_crc64nvmes: Vec<String>,
}

#[derive(Serialize, Debug)]
#[serde(rename_all = "camelCase")]
pub struct Runner {
Expand All @@ -46,12 +63,69 @@ pub struct Runner {
#[serde(rename_all = "camelCase")]
pub struct UploadData {
pub status: String,
pub upload_url: String,
pub multipart_upload: MultipartUpload,
pub run_id: String,
}

#[derive(Deserialize, Serialize, Debug)]
#[serde(rename_all = "camelCase")]
pub struct MultipartUpload {
/// Presigned S3 `UploadPart` requests, one per part, in upload order
pub parts: Vec<PresignedRequest>,
/// Presigned S3 `CompleteMultipartUpload` request
pub complete: PresignedRequest,
}

/// Request presigned by the API, to send to `url` with `headers` as is: they are
/// part of the signature
#[derive(Deserialize, Serialize, Debug)]
pub struct PresignedRequest {
pub url: String,
pub headers: BTreeMap<String, String>,
}

#[derive(Deserialize, Debug)]
#[serde(rename_all = "camelCase")]
pub struct UploadError {
pub error: String,
}

#[cfg(test)]
mod tests {
use super::*;

#[test]
fn parses_multipart_upload_response() {
let upload_data: UploadData = serde_json::from_str(
r#"{
"status": "success",
"runId": "run-id",
"multipartUpload": {
"parts": [
{ "url": "https://part/1", "headers": { "x-amz-checksum-crc64nvme": "nq48tdaL2no=" } },
{ "url": "https://part/2", "headers": { "x-amz-checksum-crc64nvme": "XMXoclwBfLo=" } }
],
"complete": {
"url": "https://complete",
"headers": { "x-amz-checksum-type": "FULL_OBJECT" }
}
}
}"#,
)
.unwrap();

assert_eq!(upload_data.run_id, "run-id");
let upload = upload_data.multipart_upload;
let part_urls: Vec<_> = upload.parts.iter().map(|part| part.url.as_str()).collect();
assert_eq!(part_urls, ["https://part/1", "https://part/2"]);
assert_eq!(
upload.parts[1].headers,
BTreeMap::from([("x-amz-checksum-crc64nvme".into(), "XMXoclwBfLo=".into())])
);
assert_eq!(upload.complete.url, "https://complete");
assert_eq!(
upload.complete.headers,
BTreeMap::from([("x-amz-checksum-type".into(), "FULL_OBJECT".into())])
);
}
}
1 change: 1 addition & 0 deletions src/upload/mod.rs
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,7 @@ mod interfaces;
pub mod poll_results;
mod profile_archive;
mod run_index_state;
mod s3;
mod upload_metadata;
mod uploader;

Expand Down
Loading
Loading