perf(format): parallel cached decode and fewer copies on full reads
Same-moment A/B on a 64 MB f64 dataset: chunked+deflate 110 -> 69 ms, chunked 72 -> 60 ms, contiguous 56 -> 30 ms. - read_chunked_data_cached — the path the facade uses — decompressed chunks one at a time; only the uncached reader was parallel. Cache misses are now decoded in bounded batches (128), in parallel with the `parallel` feature. - Every chunk was pushed into the 16 MiB chunk cache, which a larger dataset just churns (insert, evict moments later). Chunks are cached only when the whole dataset fits (new ChunkCache::max_bytes). - Unfiltered chunks went file -> Vec -> aligned cache buffer -> output. They are copied straight from the file bytes. - The facade's typed reads convert a contiguous dataset straight from the borrowed file bytes instead of copying it into a Vec first. - The native little-endian fast paths allocated vec![0; n] and then overwrote it; they now fill an uninitialised buffer in one copy (native_le_to_vec). alloc_output requests zeroed memory from the allocator instead of reserving and filling. The unit test that expected unfiltered chunks to land in the decompressed cache now asserts the new design (index reused, cache not involved). Co-Authored-By: Claude Fable 5.1 <[email protected]>
This commit is contained in:
co-authored by
Claude Fable 5.1
parent
d668e45ab5
commit
0addf328bc
@@ -876,6 +876,30 @@ fn get_size(dt: &Datatype) -> usize {
|
||||
dt.type_size() as usize
|
||||
}
|
||||
|
||||
/// Reinterpret little-endian bytes as `count` native values of `T` on a
|
||||
/// little-endian target, in one copy.
|
||||
///
|
||||
/// The buffer is allocated uninitialised and filled by the copy. It used to be
|
||||
/// `vec![0; count]` first, which for a large dataset meant writing every page
|
||||
/// twice (zero it, then overwrite it) — about as expensive as the copy itself.
|
||||
#[cfg(target_endian = "little")]
|
||||
fn native_le_to_vec<T: Copy>(raw: &[u8], count: usize) -> Vec<T> {
|
||||
let bytes = count * core::mem::size_of::<T>();
|
||||
debug_assert!(bytes <= raw.len());
|
||||
let mut result: Vec<T> = Vec::with_capacity(count);
|
||||
// SAFETY: `result` has capacity for `count` values of `T`, i.e. `bytes`
|
||||
// bytes; `raw` holds at least `bytes` bytes (callers derive `count` from
|
||||
// `raw.len() / size_of::<T>()`); the regions cannot overlap because
|
||||
// `result` was just allocated. Every `T` used here (f32/f64/i32/i64) is
|
||||
// valid for any bit pattern, so after the copy all `count` values are
|
||||
// initialised and `set_len` is sound.
|
||||
unsafe {
|
||||
core::ptr::copy_nonoverlapping(raw.as_ptr(), result.as_mut_ptr().cast::<u8>(), bytes);
|
||||
result.set_len(count);
|
||||
}
|
||||
result
|
||||
}
|
||||
|
||||
/// Convert raw bytes to `f64` values.
|
||||
pub fn read_as_f64(raw: &[u8], datatype: &Datatype) -> Result<Vec<f64>, FormatError> {
|
||||
// Array datatypes (e.g. an array-typed compound member) are read as a flat
|
||||
@@ -903,14 +927,7 @@ pub fn read_as_f64(raw: &[u8], datatype: &Datatype) -> Result<Vec<f64>, FormatEr
|
||||
..
|
||||
}
|
||||
) {
|
||||
let mut result = vec![0.0f64; count];
|
||||
// SAFETY: On LE platforms, f64 in-memory representation matches LE bytes.
|
||||
// We copy raw bytes directly into the f64 buffer.
|
||||
// SAFETY: The byte slice is properly aligned for this type and the length is divisible by size_of::<T>().
|
||||
unsafe {
|
||||
core::ptr::copy_nonoverlapping(raw.as_ptr(), result.as_mut_ptr() as *mut u8, raw.len());
|
||||
}
|
||||
return Ok(result);
|
||||
return Ok(native_le_to_vec::<f64>(raw, count));
|
||||
}
|
||||
|
||||
let order = get_byte_order(datatype);
|
||||
@@ -993,12 +1010,7 @@ pub fn read_as_i64(raw: &[u8], datatype: &Datatype) -> Result<Vec<i64>, FormatEr
|
||||
}
|
||||
)
|
||||
{
|
||||
let mut result = vec![0i64; count];
|
||||
// SAFETY: The byte slice is properly aligned for this type and the length is divisible by size_of::<T>().
|
||||
unsafe {
|
||||
core::ptr::copy_nonoverlapping(raw.as_ptr(), result.as_mut_ptr() as *mut u8, raw.len());
|
||||
}
|
||||
return Ok(result);
|
||||
return Ok(native_le_to_vec::<i64>(raw, count));
|
||||
}
|
||||
|
||||
let order = get_byte_order(datatype);
|
||||
@@ -1062,12 +1074,7 @@ pub fn read_as_f32(raw: &[u8], datatype: &Datatype) -> Result<Vec<f32>, FormatEr
|
||||
..
|
||||
}
|
||||
) {
|
||||
let mut result = vec![0.0f32; count];
|
||||
// SAFETY: The byte slice is properly aligned for this type and the length is divisible by size_of::<T>().
|
||||
unsafe {
|
||||
core::ptr::copy_nonoverlapping(raw.as_ptr(), result.as_mut_ptr() as *mut u8, raw.len());
|
||||
}
|
||||
return Ok(result);
|
||||
return Ok(native_le_to_vec::<f32>(raw, count));
|
||||
}
|
||||
|
||||
let order = get_byte_order(datatype);
|
||||
@@ -1144,12 +1151,7 @@ pub fn read_as_i32(raw: &[u8], datatype: &Datatype) -> Result<Vec<i32>, FormatEr
|
||||
}
|
||||
)
|
||||
{
|
||||
let mut result = vec![0i32; count];
|
||||
// SAFETY: The byte slice is properly aligned for this type and the length is divisible by size_of::<T>().
|
||||
unsafe {
|
||||
core::ptr::copy_nonoverlapping(raw.as_ptr(), result.as_mut_ptr() as *mut u8, raw.len());
|
||||
}
|
||||
return Ok(result);
|
||||
return Ok(native_le_to_vec::<i32>(raw, count));
|
||||
}
|
||||
|
||||
let order = get_byte_order(datatype);
|
||||
|
||||
Reference in New Issue
Block a user