Skip to main content

fxfs/object_store/
data_object_handle.rs

1// Copyright 2021 The Fuchsia Authors. All rights reserved.
2// Use of this source code is governed by a BSD-style license that can be
3// found in the LICENSE file.
4
5use crate::errors::FxfsError;
6use crate::log::*;
7use crate::lsm_tree::Query;
8use crate::lsm_tree::types::{ItemRef, LayerIterator};
9use crate::object_handle::{
10    LayerObject, ObjectHandle, ObjectProperties, ReadObjectHandle, WriteBytes, WriteObjectHandle,
11};
12use crate::object_store::extent_record::{ExtentMode, ExtentValue};
13use crate::object_store::object_manager::ObjectManager;
14use crate::object_store::object_record::{
15    AttributeKey, DirType, FsverityMetadata, ObjectAttributes, ObjectItem, ObjectKey,
16    ObjectKeyData, ObjectKind, ObjectValue, Timestamp,
17};
18use crate::object_store::store_object_handle::{MaybeChecksums, NeedsTrim};
19use crate::object_store::transaction::{
20    self, AssocObj, AssociatedObject, LockKey, Mutation, ObjectStoreMutation, Operation, Options,
21    ReadGuard, Transaction, lock_keys,
22};
23use crate::object_store::{
24    AttributeId, Extent, HandleOptions, HandleOwner, RootDigest, StoreObjectHandle,
25    TRANSACTION_MUTATION_THRESHOLD, TrimMode, TrimResult,
26};
27use crate::range::RangeExt;
28use anyhow::{Context, Error, anyhow, bail, ensure};
29use fidl_fuchsia_io as fio;
30use fsverity_merkle::{
31    FsVerityDescriptor, FsVerityDescriptorRaw, FsVerityHash, FsVerityHasher, FsVerityHasherOptions,
32    MerkleTree, MerkleTreeBuilder, Sha256Hash, Sha512Hash,
33};
34use fuchsia_sync::Mutex;
35use futures::TryStreamExt;
36use futures::stream::FuturesOrdered;
37use fxfs_trace::trace;
38use std::cmp::min;
39use std::future::Future;
40use std::ops::{Deref, Range};
41use std::pin::Pin;
42use std::sync::Arc;
43use std::sync::atomic::{self, AtomicU64, Ordering};
44use storage_device::WriteFlags;
45use storage_device::buffer::{Buffer, BufferFuture, BufferRef, MutableBufferRef};
46use storage_ptr_slice::PtrByteSlice;
47use storage_units::BlockSize;
48use zerocopy::FromBytes;
49
50mod allocated_ranges;
51pub use allocated_ranges::{AllocatedRanges, RangeType};
52
53/// How much data each transaction will cover when writing an attribute across batches. Pulled from
54/// `FLUSH_BATCH_SIZE` in paged_object_handle.rs.
55pub const WRITE_ATTR_BATCH_SIZE: usize = 524_288;
56
57/// DataObjectHandle is a typed handle for file-like objects that store data in the default data
58/// attribute. In addition to traditional files, this means things like the journal, superblocks,
59/// and layer files.
60///
61/// It caches the content size of the data attribute it was configured for, and has helpers for
62/// complex extent manipulation, as well as implementations of ReadObjectHandle and
63/// WriteObjectHandle.
64pub struct DataObjectHandle<S: HandleOwner> {
65    handle: StoreObjectHandle<S>,
66    attribute_id: AttributeId,
67    content_size: AtomicU64,
68    state: Mutex<DataObjectState>,
69}
70
71/// Represents the mapping of a file's contents to the physical storage backing it.
72#[derive(Debug, Clone)]
73pub struct FileExtent {
74    logical_offset: u64,
75    device_range: Range<u64>,
76}
77
78impl FileExtent {
79    pub fn new(logical_offset: u64, device_range: Range<u64>) -> Result<Self, Error> {
80        // Ensure `device_range` is valid.
81        let length = device_range.length()?;
82        // Ensure no overflow when we calculate the end of the logical range.
83        let _ = logical_offset.checked_add(length).ok_or(FxfsError::OutOfRange)?;
84        Ok(Self { logical_offset, device_range })
85    }
86}
87
88impl FileExtent {
89    pub fn length(&self) -> u64 {
90        // SAFETY: We verified that the device_range's length is valid in Self::new.
91        unsafe { self.device_range.unchecked_length() }
92    }
93
94    pub fn logical_offset(&self) -> u64 {
95        self.logical_offset
96    }
97
98    pub fn logical_range(&self) -> Range<u64> {
99        // SAFETY: We verified logical_offset plus device_range length won't overflow in Self::new.
100        unsafe { self.logical_offset..self.logical_offset.unchecked_add(self.length()) }
101    }
102
103    pub fn device_range(&self) -> &Range<u64> {
104        &self.device_range
105    }
106}
107
108#[derive(Debug)]
109pub enum DataObjectState {
110    Standard(AllocatedRanges),
111    VerityStarted,
112    VerityPending(FsverityStateInner),
113    Verity(FsverityStateInner),
114}
115
116#[derive(Debug)]
117pub struct FsverityStateInner {
118    root_digest: RootDigest,
119    salt: Vec<u8>,
120    // TODO(b/309656632): This should store the entire merkle tree and not just the leaf nodes.
121    // Potentially store a pager-backed vmo instead of passing around a boxed array.
122    merkle_tree: Box<[u8]>,
123}
124
125#[derive(Debug, Default)]
126pub struct OverwriteOptions {
127    // If false, then all the extents for the overwrite range must have been preallocated using
128    // preallocate_range or from existing writes.
129    pub allow_allocations: bool,
130    pub barrier_on_first_write: bool,
131}
132
133impl FsverityStateInner {
134    pub fn new(root_digest: RootDigest, salt: Vec<u8>, merkle_tree: Box<[u8]>) -> Self {
135        FsverityStateInner { root_digest, salt, merkle_tree }
136    }
137
138    fn get_hasher_for_block_size(&self, block_size: BlockSize) -> FsVerityHasher {
139        match self.root_digest {
140            RootDigest::Sha256(_) => FsVerityHasher::Sha256(FsVerityHasherOptions::new(
141                self.salt.clone(),
142                block_size.get() as usize,
143            )),
144            RootDigest::Sha512(_) => FsVerityHasher::Sha512(FsVerityHasherOptions::new(
145                self.salt.clone(),
146                block_size.get() as usize,
147            )),
148        }
149    }
150
151    fn from_ptr_slice(
152        data: PtrByteSlice<'_>,
153        block_size: BlockSize,
154    ) -> Result<(Self, FsVerityHasher), Error> {
155        let descriptor = FsVerityDescriptor::new(data, block_size.get() as usize)
156            .map_err(|e| anyhow!(FxfsError::IntegrityError).context(e))?;
157
158        let root_digest = match descriptor.digest_algorithm() {
159            fio::HashAlgorithm::Sha256 => {
160                RootDigest::Sha256(descriptor.root_digest().try_into().unwrap())
161            }
162            fio::HashAlgorithm::Sha512 => RootDigest::Sha512(descriptor.root_digest().to_vec()),
163            _ => return Err(anyhow!(FxfsError::NotSupported).context("Unsupported hash algorithm")),
164        };
165        let hasher = descriptor.hasher();
166        let leaves =
167            descriptor.leaf_digests().map_err(|e| anyhow!(FxfsError::IntegrityError).context(e))?;
168
169        Ok((Self::new(root_digest, descriptor.salt().to_vec(), leaves.into_boxed_slice()), hasher))
170    }
171}
172
173impl<S: HandleOwner> Deref for DataObjectHandle<S> {
174    type Target = StoreObjectHandle<S>;
175    fn deref(&self) -> &Self::Target {
176        &self.handle
177    }
178}
179
180impl<S: HandleOwner> DataObjectHandle<S> {
181    pub fn new(
182        owner: Arc<S>,
183        object_id: u64,
184        permanent_keys: bool,
185        attribute_id: AttributeId,
186        size: u64,
187        options: HandleOptions,
188        trace: bool,
189        overwrite_ranges: &[Range<u64>],
190    ) -> Self {
191        Self {
192            handle: StoreObjectHandle::new(owner, object_id, permanent_keys, options, trace),
193            attribute_id,
194            content_size: AtomicU64::new(size),
195            state: Mutex::new(DataObjectState::Standard(AllocatedRanges::new(overwrite_ranges))),
196        }
197    }
198
199    pub fn attribute_id(&self) -> AttributeId {
200        self.attribute_id
201    }
202
203    /// Consumes the `DataObjectHandle` and returns the `StoreObjectHandle` that it contained.
204    pub fn into_store_object_handle(self) -> StoreObjectHandle<S> {
205        self.handle
206    }
207
208    pub fn overwrite_ranges_is_empty(&self) -> bool {
209        match &*self.state.lock() {
210            DataObjectState::Standard(ranges) => ranges.is_empty(),
211            _ => true,
212        }
213    }
214
215    pub fn with_overwrite_ranges<R>(&self, f: impl FnOnce(Option<&AllocatedRanges>) -> R) -> R {
216        let state = self.state.lock();
217        match &*state {
218            DataObjectState::Standard(ranges) => f(Some(ranges)),
219            _ => f(None),
220        }
221    }
222
223    pub fn with_overwrite_ranges_mut<R>(
224        &self,
225        f: impl FnOnce(Option<&mut AllocatedRanges>) -> R,
226    ) -> R {
227        let mut state = self.state.lock();
228        match &mut *state {
229            DataObjectState::Standard(ranges) => f(Some(ranges)),
230            _ => f(None),
231        }
232    }
233
234    pub fn is_verified_file(&self) -> bool {
235        matches!(*self.state.lock(), DataObjectState::Verity(_))
236    }
237
238    /// Sets `self.state` to DataObjectState::VerityStarted. Called at the top of `enable_verity`.
239    /// If another caller has already started but not completed `enable_verity`, returns
240    /// FxfsError::AlreadyBound. If another caller has already completed `enable_verity`, returns
241    /// FxfsError::AlreadyExists.
242    ///
243    /// Note: This is called before `enable_verity` acquires its object transaction lock. Any
244    /// ongoing transaction holding the object lock (such as `allocate`) will proceed to commit
245    /// its mutations to disk while `enable_verity` waits for the lock, and `will_apply_mutation`
246    /// will gracefully discard in-memory range updates for the discarded `AllocatedRanges`.
247    pub fn set_fsverity_state_started(&self) -> Result<(), Error> {
248        let mut state = self.state.lock();
249        match *state {
250            DataObjectState::Standard(_) => {
251                *state = DataObjectState::VerityStarted;
252                Ok(())
253            }
254            DataObjectState::VerityStarted | DataObjectState::VerityPending(_) => {
255                Err(anyhow!(FxfsError::Unavailable))
256            }
257            DataObjectState::Verity(_) => Err(anyhow!(FxfsError::AlreadyExists)),
258        }
259    }
260
261    /// Sets `self.state` to VerityPending. Must be called before `finalize_fsverity_state()`.
262    ///
263    /// # Panics
264    ///
265    /// Panics if the prior state was not `DataObjectState::VerityStarted`.
266    pub fn set_fsverity_state_pending(&self, descriptor: FsverityStateInner) {
267        let mut state = self.state.lock();
268        assert!(matches!(*state, DataObjectState::VerityStarted));
269        *state = DataObjectState::VerityPending(descriptor);
270    }
271
272    /// Sets `self.state` to Verity.
273    ///
274    /// # Panics
275    ///
276    /// Panics if the prior state was not `DataObjectState::VerityPending(_)`.
277    pub fn finalize_fsverity_state(&self) {
278        let mut state = self.state.lock();
279        let old_state =
280            std::mem::replace(&mut *state, DataObjectState::Standard(AllocatedRanges::empty()));
281        match old_state {
282            DataObjectState::VerityPending(inner) => *state = DataObjectState::Verity(inner),
283            _ => panic!("Cannot finalize verity state from {old_state:?}"),
284        }
285    }
286
287    /// Sets `self.state` directly to Verity without going through the entire state machine.
288    /// Used to set `self.state` on open of a verified file. The merkle tree data is
289    /// verified against the root digest here, and will return an error if the tree is not correct.
290    ///
291    /// # Panics
292    ///
293    /// Panics if the prior state was not `DataObjectState::Standard(_)`.
294    pub async fn set_fsverity_state_some(&self, descriptor: FsverityMetadata) -> Result<(), Error> {
295        let (metadata, hasher) = match descriptor {
296            FsverityMetadata::Internal(root_digest, salt) => {
297                let merkle_tree = self
298                    .read_attr(AttributeId::FSVERITY_MERKLE)
299                    .await?
300                    .ok_or_else(|| anyhow!(FxfsError::Inconsistent))?;
301                let metadata = FsverityStateInner { root_digest, salt, merkle_tree };
302                let hasher = metadata.get_hasher_for_block_size(self.block_size());
303                (metadata, hasher)
304            }
305            FsverityMetadata::F2fs(verity_range) => {
306                let expected_length = verity_range.length()? as usize;
307                let mut buffer = self
308                    .allocate_buffer(
309                        self.block_size().align_up(expected_length as u64).unwrap() as usize
310                    )
311                    .await;
312                let read = self
313                    .handle
314                    .read_aligned(AttributeId::FSVERITY_MERKLE, verity_range.start, buffer.as_mut())
315                    .await?;
316                ensure!(expected_length == read, FxfsError::Inconsistent);
317                let data = buffer.as_ptr_slice().subslice(0..expected_length);
318                FsverityStateInner::from_ptr_slice(data, self.block_size())?
319            }
320        };
321        // Validate the merkle tree data against the root before applying it.
322        ensure!(metadata.merkle_tree.len() % hasher.hash_size() == 0, FxfsError::Inconsistent);
323        let leaf_chunks = metadata.merkle_tree.chunks_exact(hasher.hash_size());
324
325        let root_hash = match &metadata.root_digest {
326            RootDigest::Sha256(root_hash) => root_hash.as_slice(),
327            RootDigest::Sha512(root_hash) => root_hash.as_slice(),
328        };
329
330        let tree = match hasher {
331            FsVerityHasher::Sha256(_) => {
332                let mut builder = MerkleTreeBuilder::<Sha256Hash>::new(hasher);
333                for leaf in leaf_chunks {
334                    let hash = Sha256Hash::read_from_bytes(leaf).unwrap();
335                    builder.push_data_hash(hash);
336                }
337                builder.finish()
338            }
339            FsVerityHasher::Sha512(_) => {
340                let mut builder = MerkleTreeBuilder::<Sha512Hash>::new(hasher);
341                for leaf in leaf_chunks {
342                    let hash = Sha512Hash::read_from_bytes(leaf).unwrap();
343                    builder.push_data_hash(hash);
344                }
345                builder.finish()
346            }
347        };
348
349        ensure!(root_hash == tree.root(), FxfsError::IntegrityError);
350
351        let mut state = self.state.lock();
352        assert!(matches!(*state, DataObjectState::Standard(_)));
353        *state = DataObjectState::Verity(metadata);
354
355        Ok(())
356    }
357
358    /// Verifies contents of `buffer` against the corresponding hashes in the stored merkle tree.
359    /// `offset` is the logical offset in the file that `buffer` starts at. Fails on non
360    /// fsverity-enabled files.
361    ///
362    /// # Panics
363    ///
364    /// Panics if `offset` is not block-aligned.
365    fn verify_data(&self, mut offset: usize, buffer: PtrByteSlice<'_>) -> Result<(), Error> {
366        let block_size = self.block_size();
367        assert!(block_size.is_aligned(offset as u64));
368        let state = self.state.lock();
369        match &*state {
370            DataObjectState::Standard(_) => {
371                Err(anyhow!("Tried to verify read on a non verity-enabled file"))
372            }
373            DataObjectState::VerityStarted | DataObjectState::VerityPending(_) => {
374                Err(anyhow!("Enable verity has not yet completed, state: {state:?}"))
375            }
376            DataObjectState::Verity(metadata) => {
377                let hasher = metadata.get_hasher_for_block_size(block_size);
378                let leaf_nodes: Vec<&[u8]> =
379                    metadata.merkle_tree.chunks(hasher.hash_size()).collect();
380                fxfs_trace::duration!("fsverity-verify", "len" => buffer.len());
381                // TODO(b/318880297): Consider parallelizing computation.
382                for chunk in buffer.chunks(block_size.get() as usize) {
383                    // SAFETY: Ideally we wouldn't be creating references here as technically this is
384                    // Rust undefined behaviour, but it's difficult for us to fix and mitigated
385                    // because these pointers end up being passed directly to non-Rust code (Mundane).
386                    let b = unsafe { &*chunk.as_raw_slice_ptr() };
387
388                    ensure!(
389                        hasher.hash_block(b) == leaf_nodes[((offset as u64) / block_size) as usize],
390                        anyhow!(FxfsError::Inconsistent).context("Hash mismatch")
391                    );
392                    offset += block_size.get() as usize;
393                }
394                Ok(())
395            }
396        }
397    }
398
399    /// Extend the file with the given extent.  The only use case for this right now is for files
400    /// that must exist at certain offsets on the device, such as super-blocks.
401    pub async fn extend<'a>(
402        &'a self,
403        transaction: &mut Transaction<'a>,
404        device_range: Range<u64>,
405    ) -> Result<(), Error> {
406        let old_end =
407            self.block_size().align_up(self.txn_get_size(transaction)).ok_or(FxfsError::TooBig)?;
408        let new_size = old_end + device_range.end - device_range.start;
409        self.store().allocator().mark_allocated(
410            transaction,
411            self.store().store_object_id(),
412            device_range.clone(),
413        )?;
414        self.txn_update_size(transaction, new_size, None).await?;
415        let key_id = self.get_key(None).await?.0;
416        transaction.add(
417            self.store().store_object_id,
418            Mutation::merge_object(
419                ObjectKey::extent(self.object_id(), self.attribute_id(), old_end..new_size),
420                ObjectValue::Extent(ExtentValue::new_raw(device_range.start, key_id)),
421            ),
422        );
423        self.update_allocated_size(transaction, device_range.end - device_range.start, 0).await
424    }
425
426    // Returns a new aligned buffer (reading the head and tail blocks if necessary) with a copy of
427    // the data from `buf`.
428    async fn align_buffer(
429        &self,
430        offset: u64,
431        buf: BufferRef<'_>,
432    ) -> Result<(std::ops::Range<u64>, Buffer<'_>), Error> {
433        self.handle.align_buffer(self.attribute_id(), offset, buf).await
434    }
435
436    // Writes potentially unaligned data at `device_offset` and returns checksums if requested. The
437    // data will be encrypted if necessary.
438    // `buf` is mutable as an optimization, since the write may require encryption, we can encrypt
439    // the buffer in-place rather than copying to another buffer if the write is already aligned.
440    // `flags` are forwarded to the underlying device write.
441    async fn write_at(
442        &self,
443        offset: u64,
444        buf: MutableBufferRef<'_>,
445        device_offset: u64,
446        flags: WriteFlags,
447    ) -> Result<MaybeChecksums, Error> {
448        self.handle
449            .write_at_with_flags(self.attribute_id(), offset, buf, None, device_offset, flags)
450            .await
451    }
452
453    /// Verifies that the entire range in the file is zeroes, as either uninitialized overwrite
454    /// range, or no extent at all. If a single allocated and written extent is found, this returns
455    /// false.
456    pub async fn check_unwritten_zero(&self, range: Range<u64>) -> Result<bool, Error> {
457        let tree = &self.store().tree();
458        let layer_set = tree.layer_set();
459        let key = Extent(range);
460        let lower_bound = ObjectKey::attribute(
461            self.object_id(),
462            self.attribute_id,
463            AttributeKey::Extent(key.search_key()),
464        );
465        let mut merger = layer_set.merger();
466        let mut iter = merger.query(Query::FullRange(&lower_bound)).await?;
467        while let Some(ItemRef {
468            key:
469                ObjectKey {
470                    object_id,
471                    data: ObjectKeyData::Attribute(attr_id, AttributeKey::Extent(extent_key)),
472                },
473            value: ObjectValue::Extent(value),
474            ..
475        }) = iter.get()
476            && *object_id == self.object_id()
477            && *attr_id == self.attribute_id
478        {
479            if let ExtentValue::Some { mode, .. } = value {
480                if let Some(overlap) = key.overlap(extent_key) {
481                    if let ExtentMode::OverwritePartial(bits) = mode {
482                        let starting_index = (overlap.start - extent_key.start) / self.block_size();
483                        for initialized in bits
484                            .iter()
485                            .skip(starting_index as usize)
486                            .take((overlap.length().unwrap() / self.block_size()) as usize)
487                        {
488                            if initialized {
489                                return Ok(false);
490                            }
491                        }
492                    } else {
493                        return Ok(false);
494                    }
495                } else {
496                    break;
497                }
498            }
499            iter.advance().await?;
500        }
501        Ok(true)
502    }
503
504    /// Zeroes the given range.  The range must be aligned.  Returns the amount of data deallocated.
505    pub async fn zero(
506        &self,
507        transaction: &mut Transaction<'_>,
508        range: Range<u64>,
509    ) -> Result<(), Error> {
510        self.handle.zero(transaction, self.attribute_id(), range).await
511    }
512
513    /// The cached value for `self.fsverity_state` is set either in `open_object` or on
514    /// `enable_verity`. If set, translates `self.fsverity_state.descriptor` into an
515    /// fio::VerificationOptions instance and a root hash. Otherwise, returns None.
516    pub fn get_descriptor(&self) -> Option<(fio::VerificationOptions, Vec<u8>)> {
517        let state = self.state.lock();
518        match &*state {
519            DataObjectState::Verity(metadata) => {
520                let (options, root_hash) = match &metadata.root_digest {
521                    RootDigest::Sha256(root_hash) => (
522                        fio::VerificationOptions {
523                            hash_algorithm: Some(fio::HashAlgorithm::Sha256),
524                            salt: Some(metadata.salt.clone()),
525                            ..Default::default()
526                        },
527                        root_hash.to_vec(),
528                    ),
529                    RootDigest::Sha512(root_hash) => (
530                        fio::VerificationOptions {
531                            hash_algorithm: Some(fio::HashAlgorithm::Sha512),
532                            salt: Some(metadata.salt.clone()),
533                            ..Default::default()
534                        },
535                        root_hash.clone(),
536                    ),
537                };
538                Some((options, root_hash))
539            }
540            _ => None,
541        }
542    }
543
544    async fn build_verity_tree(
545        &self,
546        hasher: FsVerityHasher,
547        hash_alg: fio::HashAlgorithm,
548        salt: &[u8],
549    ) -> Result<(MerkleTree, Vec<u8>), Error> {
550        match hasher {
551            FsVerityHasher::Sha256(_) => {
552                self.build_verity_tree_impl::<Sha256Hash>(hasher, hash_alg, salt).await
553            }
554            FsVerityHasher::Sha512(_) => {
555                self.build_verity_tree_impl::<Sha512Hash>(hasher, hash_alg, salt).await
556            }
557        }
558    }
559
560    async fn build_verity_tree_impl<D: FsVerityHash>(
561        &self,
562        hasher: FsVerityHasher,
563        hash_alg: fio::HashAlgorithm,
564        salt: &[u8],
565    ) -> Result<(MerkleTree, Vec<u8>), Error> {
566        let hash_len = hasher.hash_size();
567        let mut builder = MerkleTreeBuilder::<D>::new(hasher);
568        let mut offset = 0;
569        let size = self.get_size();
570        // TODO(b/314836822): Consider further tuning the buffer size to optimize
571        // performance. Experimentally, most verity-enabled files are <256K.
572        let mut buf = self.allocate_buffer(64 * self.block_size().get() as usize).await;
573        while offset < size {
574            // TODO(b/314842875): Consider optimizations for sparse files.
575            let read = self.read_aligned(offset, buf.as_mut()).await? as u64;
576            assert!(offset + read <= size);
577            let slice = buf.as_ptr_slice().subslice(0..read as usize);
578
579            // SAFETY: Ideally we wouldn't be creating references here as technically this is
580            // Rust undefined behaviour, but it's difficult for us to fix and mitigated
581            // because these pointers end up being passed directly to non-Rust code (Mundane).
582            let chunk = unsafe { &*slice.as_raw_slice_ptr() };
583
584            builder.write(chunk);
585            offset += read;
586        }
587        let tree = builder.finish();
588        // This will include a block for the root layer, which will be used to house the descriptor.
589        let tree_data_len = tree
590            .levels()
591            .iter()
592            .map(|layer| self.block_size().align_up(layer.len() as u64).unwrap() as usize)
593            .sum();
594        let mut merkle_tree_data = Vec::<u8>::with_capacity(tree_data_len);
595        // Iterating from the top layers down to the leaves.
596        for layer in tree.levels().iter().rev() {
597            // Skip the root layer.
598            if layer.len() <= hash_len {
599                continue;
600            }
601            merkle_tree_data.extend_from_slice(layer);
602            // Pad to the end of the block.
603            let padded_size =
604                self.block_size().align_up(merkle_tree_data.len() as u64).unwrap() as usize;
605            merkle_tree_data.resize(padded_size, 0);
606        }
607
608        // Zero the last block, then write the descriptor to the start of it.
609        let descriptor_offset = merkle_tree_data.len();
610        merkle_tree_data.resize(descriptor_offset + self.block_size().get() as usize, 0);
611        let descriptor = FsVerityDescriptorRaw::new(
612            hash_alg,
613            self.block_size().get(),
614            self.get_size(),
615            tree.root(),
616            salt,
617        )?;
618        descriptor.write_to_slice(&mut merkle_tree_data[descriptor_offset..])?;
619
620        Ok((tree, merkle_tree_data))
621    }
622
623    /// Reads the data attribute and computes a merkle tree from the data. The values of the
624    /// parameters required to build the merkle tree are supplied by `descriptor` (i.e. salt,
625    /// hash_algorithm, etc.) Writes the leaf nodes of the merkle tree to an attribute with id
626    /// `AttributeId::FSVERITY_MERKLE`. Updates the root_hash of the `descriptor` according to the
627    /// computed merkle tree and then replaces the ObjectValue of the data attribute with
628    /// ObjectValue::VerifiedAttribute, which stores the `descriptor` inline.
629    #[trace]
630    pub async fn enable_verity(&self, options: fio::VerificationOptions) -> Result<(), Error> {
631        self.set_fsverity_state_started()?;
632        // If the merkle attribute was tombstoned in the last attempt of `enable_verity`, flushing
633        // the graveyard should process the tombstone before we start rewriting the attribute.
634        if self
635            .store()
636            .tree()
637            .exists(&ObjectKey::graveyard_attribute_entry(
638                self.store().graveyard_directory_object_id(),
639                self.object_id(),
640                AttributeId::FSVERITY_MERKLE,
641            ))
642            .await?
643        {
644            self.store().filesystem().graveyard().flush().await;
645        }
646        let mut transaction = self.new_transaction().await?;
647        let hash_alg =
648            options.hash_algorithm.ok_or_else(|| anyhow!("No hash algorithm provided"))?;
649        let salt = options.salt.ok_or_else(|| anyhow!("No salt provided"))?;
650        let (root_digest, merkle_tree) = match hash_alg {
651            fio::HashAlgorithm::Sha256 => {
652                let hasher = FsVerityHasher::Sha256(FsVerityHasherOptions::new(
653                    salt.clone(),
654                    self.block_size().get() as usize,
655                ));
656                let (tree, merkle_tree_data) =
657                    self.build_verity_tree(hasher, hash_alg, &salt).await?;
658                let root: [u8; 32] = tree.root().try_into().unwrap();
659                (RootDigest::Sha256(root), merkle_tree_data)
660            }
661            fio::HashAlgorithm::Sha512 => {
662                let hasher = FsVerityHasher::Sha512(FsVerityHasherOptions::new(
663                    salt.clone(),
664                    self.block_size().get() as usize,
665                ));
666                let (tree, merkle_tree_data) =
667                    self.build_verity_tree(hasher, hash_alg, &salt).await?;
668                (RootDigest::Sha512(tree.root().to_vec()), merkle_tree_data)
669            }
670            _ => {
671                bail!(
672                    anyhow!(FxfsError::NotSupported)
673                        .context(format!("hash algorithm not supported"))
674                );
675            }
676        };
677        // TODO(b/314194485): Eventually want streaming writes.
678        // The merkle tree attribute should not require trimming because it should not
679        // exist.
680        self.handle
681            .write_new_attr_in_batches(
682                &mut transaction,
683                AttributeId::FSVERITY_MERKLE,
684                &merkle_tree,
685                WRITE_ATTR_BATCH_SIZE,
686            )
687            .await?;
688        if merkle_tree.len() > WRITE_ATTR_BATCH_SIZE {
689            self.store().remove_attribute_from_graveyard(
690                &mut transaction,
691                self.object_id(),
692                AttributeId::FSVERITY_MERKLE,
693            );
694        };
695        let descriptor_decoded =
696            FsVerityDescriptor::new(&merkle_tree[..], self.block_size().get() as usize)?;
697        let descriptor = FsverityStateInner {
698            root_digest,
699            salt,
700            merkle_tree: descriptor_decoded.leaf_digests()?.into(),
701        };
702        self.set_fsverity_state_pending(descriptor);
703        transaction.add_with_object(
704            self.store().store_object_id(),
705            Mutation::replace_or_insert_object(
706                ObjectKey::attribute(self.object_id(), AttributeId::DATA, AttributeKey::Attribute),
707                ObjectValue::verified_attribute(
708                    self.get_size(),
709                    FsverityMetadata::F2fs(0..merkle_tree.len() as u64),
710                ),
711            ),
712            AssocObj::Borrowed(self),
713        );
714        transaction.commit().await?;
715        Ok(())
716    }
717
718    /// Pre-allocate disk space for the given logical file range. If any part of the allocation
719    /// range is beyond the end of the file, the file size is updated.
720    pub async fn allocate(&self, range: Range<u64>) -> Result<(), Error> {
721        debug_assert!(range.start < range.end);
722
723        // It's not required that callers of allocate use block aligned ranges, but we need to make
724        // the extents block aligned. Luckily, fallocate in posix is allowed to allocate more than
725        // what was asked for for block alignment purposes. We just need to make sure that the size
726        // NB: FxfsError::TooBig turns into EFBIG when passed through starnix, which is the
727        // required error code when the requested range is larger than the file size.
728        let mut new_range =
729            self.block_size().align_range_outwards(&range).ok_or(FxfsError::TooBig)?;
730
731        let mut transaction = self.new_transaction().await?;
732        // It's safe to check state after acquiring the transaction lock. Note that `enable_verity`
733        // calls `set_fsverity_state_started` before creating its own transaction, which could
734        // transition `self.state` to `VerityStarted` concurrently while this transaction is held.
735        // If that happens, `enable_verity`'s subsequent `new_transaction` call will block until
736        // this allocation transaction finishes, and `will_apply_mutation` will gracefully discard
737        // updates to the in-memory allocated ranges.
738        {
739            let state = self.state.lock();
740            match &*state {
741                DataObjectState::Standard(_) => {}
742                _ => bail!(
743                    anyhow!(FxfsError::AccessDenied).context("Cannot allocate on verity file")
744                ),
745            }
746        }
747        let mut to_allocate = Vec::new();
748        let mut to_switch = Vec::new();
749        let key_id = self.get_key(None).await?.0;
750
751        {
752            let tree = &self.store().tree;
753            let layer_set = tree.layer_set();
754            let offset_key = ObjectKey::attribute(
755                self.object_id(),
756                self.attribute_id(),
757                AttributeKey::Extent(Extent::search_key_from_offset(new_range.start)),
758            );
759            let mut merger = layer_set.merger();
760            let mut iter = merger.query(Query::FullRange(&offset_key)).await?;
761
762            loop {
763                match iter.get() {
764                    Some(ItemRef {
765                        key:
766                            ObjectKey {
767                                object_id,
768                                data:
769                                    ObjectKeyData::Attribute(
770                                        attribute_id,
771                                        AttributeKey::Extent(extent_key),
772                                    ),
773                            },
774                        value: ObjectValue::Extent(extent_value),
775                        ..
776                    }) if *object_id == self.object_id()
777                        && *attribute_id == self.attribute_id() =>
778                    {
779                        // If the start of this extent is beyond the end of the range we are
780                        // allocating, we don't have any more work to do.
781                        if new_range.end <= extent_key.start {
782                            break;
783                        }
784                        // Add any prefix we might need to allocate.
785                        if new_range.start < extent_key.start {
786                            to_allocate.push(new_range.start..extent_key.start);
787                            new_range.start = extent_key.start;
788                        }
789                        let device_offset = match extent_value {
790                            ExtentValue::None => {
791                                // If the extent value is None, it indicates a deleted extent. In
792                                // that case, we just skip it entirely. By keeping the new_range
793                                // where it is, this section will get included in the new
794                                // allocations.
795                                iter.advance().await?;
796                                continue;
797                            }
798                            ExtentValue::Some { mode: ExtentMode::OverwritePartial(_), .. }
799                            | ExtentValue::Some { mode: ExtentMode::Overwrite, .. } => {
800                                // If this extent is already in overwrite mode, we can skip it.
801                                if extent_key.end < new_range.end {
802                                    new_range.start = extent_key.end;
803                                    iter.advance().await?;
804                                    continue;
805                                } else {
806                                    new_range.start = new_range.end;
807                                    break;
808                                }
809                            }
810                            ExtentValue::Some { device_offset, .. } => *device_offset,
811                        };
812
813                        // Figure out how we have to break up the ranges.
814                        let device_offset = device_offset + (new_range.start - extent_key.start);
815                        if extent_key.end < new_range.end {
816                            to_switch.push((new_range.start..extent_key.end, device_offset));
817                            new_range.start = extent_key.end;
818                        } else {
819                            to_switch.push((new_range.start..new_range.end, device_offset));
820                            new_range.start = new_range.end;
821                            break;
822                        }
823                    }
824                    // The records are sorted so if we find something that isn't an extent or
825                    // doesn't match the object id then there are no more extent records for this
826                    // object.
827                    _ => break,
828                }
829                iter.advance().await?;
830            }
831        }
832
833        if new_range.start < new_range.end {
834            to_allocate.push(new_range.clone());
835        }
836
837        // We can update the size in the first transaction because even if subsequent transactions
838        // don't get replayed, the data between the current and new end of the file will be zero
839        // (either sparse zero or allocated zero). On the other hand, if we don't update the size
840        // in the first transaction, overwrite extents may be written past the end of the file
841        // which is an fsck error.
842        //
843        // The potential new size needs to be the non-block-aligned range end - we round up to the
844        // nearest block size for the actual allocation, but shouldn't do that for the file size.
845        let new_size = std::cmp::max(range.end, self.get_size());
846        // Make sure the mutation that flips the has_overwrite_extents advisory flag is in the
847        // first transaction, in case we split transactions. This makes it okay to only replay the
848        // first transaction if power loss occurs - the file will be in an unusual state, but not
849        // an invalid one, if only part of the allocate goes through.
850        transaction.add_with_object(
851            self.store().store_object_id(),
852            Mutation::replace_or_insert_object(
853                ObjectKey::attribute(
854                    self.object_id(),
855                    self.attribute_id(),
856                    AttributeKey::Attribute,
857                ),
858                ObjectValue::Attribute { size: new_size, has_overwrite_extents: true },
859            ),
860            AssocObj::Borrowed(self),
861        );
862
863        // The maximum number of mutations we are going to allow per transaction in allocate. This
864        // is probably quite a bit lower than the actual limit, but it should be large enough to
865        // handle most non-edge-case versions of allocate without splitting the transaction.
866        const MAX_TRANSACTION_SIZE: usize = 256;
867        for (switch_range, device_offset) in to_switch {
868            transaction.add_with_object(
869                self.store().store_object_id(),
870                Mutation::merge_object(
871                    ObjectKey::extent(self.object_id(), self.attribute_id(), switch_range),
872                    ObjectValue::Extent(ExtentValue::initialized_overwrite_extent(
873                        device_offset,
874                        key_id,
875                    )),
876                ),
877                AssocObj::Borrowed(self),
878            );
879            if transaction.mutations().len() >= MAX_TRANSACTION_SIZE {
880                transaction.commit_and_continue().await?;
881            }
882        }
883
884        let mut allocated = 0;
885        let allocator = self.store().allocator();
886        for mut allocate_range in to_allocate {
887            while allocate_range.start < allocate_range.end {
888                let device_range = allocator
889                    .allocate(
890                        &mut transaction,
891                        self.store().store_object_id(),
892                        allocate_range.end - allocate_range.start,
893                    )
894                    .await
895                    .context("allocation failed")?;
896                let device_range_len = device_range.end - device_range.start;
897
898                transaction.add_with_object(
899                    self.store().store_object_id(),
900                    Mutation::merge_object(
901                        ObjectKey::extent(
902                            self.object_id(),
903                            self.attribute_id(),
904                            allocate_range.start..allocate_range.start + device_range_len,
905                        ),
906                        ObjectValue::Extent(ExtentValue::blank_overwrite_extent(
907                            device_range.start,
908                            (device_range_len / self.block_size()) as usize,
909                            key_id,
910                        )),
911                    ),
912                    AssocObj::Borrowed(self),
913                );
914
915                allocate_range.start += device_range_len;
916                allocated += device_range_len;
917
918                if transaction.mutations().len() >= MAX_TRANSACTION_SIZE {
919                    self.update_allocated_size(&mut transaction, allocated, 0).await?;
920                    transaction.commit_and_continue().await?;
921                    allocated = 0;
922                }
923            }
924        }
925
926        self.update_allocated_size(&mut transaction, allocated, 0).await?;
927        transaction.commit().await?;
928
929        Ok(())
930    }
931
932    /// Return information on a contiguous set of extents that has the same allocation status,
933    /// starting from `start_offset`. The information returned is if this set of extents are marked
934    /// allocated/not allocated and also the size of this set (in bytes). This is used when
935    /// querying slices for volumes.
936    /// This function expects `start_offset` to be aligned to block size
937    pub async fn is_allocated(&self, start_offset: u64) -> Result<(bool, u64), Error> {
938        let block_size = self.block_size();
939        assert_eq!(start_offset % block_size, 0);
940
941        if start_offset > self.get_size() {
942            bail!(FxfsError::OutOfRange)
943        }
944
945        if start_offset == self.get_size() {
946            return Ok((false, 0));
947        }
948
949        let tree = &self.store().tree;
950        let layer_set = tree.layer_set();
951        let offset_key = ObjectKey::attribute(
952            self.object_id(),
953            self.attribute_id(),
954            AttributeKey::Extent(Extent::search_key_from_offset(start_offset)),
955        );
956        let mut merger = layer_set.merger();
957        let mut iter = merger.query(Query::FullRange(&offset_key)).await?;
958
959        let mut allocated = None;
960        let mut end = start_offset;
961
962        loop {
963            // Iterate through the extents, each time setting `end` as the end of the previous
964            // extent
965            match iter.get() {
966                Some(ItemRef {
967                    key:
968                        ObjectKey {
969                            object_id,
970                            data:
971                                ObjectKeyData::Attribute(attribute_id, AttributeKey::Extent(extent_key)),
972                        },
973                    value: ObjectValue::Extent(extent_value),
974                    ..
975                }) => {
976                    // Equivalent of getting no extents back
977                    if *object_id != self.object_id() || *attribute_id != self.attribute_id() {
978                        if allocated == Some(false) || allocated.is_none() {
979                            end = self.get_size();
980                            allocated = Some(false);
981                        }
982                        break;
983                    }
984                    ensure!(block_size.is_aligned(extent_key), FxfsError::Inconsistent);
985                    if extent_key.start > end {
986                        // If a previous extent has already been visited and we are tracking an
987                        // allocated set, we are only interested in an extent where the range of the
988                        // current extent follows immediately after the previous one.
989                        if allocated == Some(true) {
990                            break;
991                        } else {
992                            // The gap between the previous `end` and this extent is not allocated
993                            end = extent_key.start;
994                            allocated = Some(false);
995                            // Continue this iteration, except now the `end` is set to the end of
996                            // the "previous" extent which is this gap between the start_offset
997                            // and the current extent
998                        }
999                    }
1000
1001                    // We can assume that from here, the `end` points to the end of a previous
1002                    // extent.
1003                    match extent_value {
1004                        // The current extent has been allocated
1005                        ExtentValue::Some { .. } => {
1006                            // Stop searching if previous extent was marked deleted
1007                            if allocated == Some(false) {
1008                                break;
1009                            }
1010                            allocated = Some(true);
1011                        }
1012                        // This extent has been marked deleted
1013                        ExtentValue::None => {
1014                            // Stop searching if previous extent was marked allocated
1015                            if allocated == Some(true) {
1016                                break;
1017                            }
1018                            allocated = Some(false);
1019                        }
1020                    }
1021                    end = extent_key.end;
1022                }
1023                // This occurs when there are no extents left
1024                None => {
1025                    if allocated == Some(false) || allocated.is_none() {
1026                        end = self.get_size();
1027                        allocated = Some(false);
1028                    }
1029                    // Otherwise, we were monitoring extents that were allocated, so just exit.
1030                    break;
1031                }
1032                // Non-extent records (Object, Child, GraveyardEntry) are ignored.
1033                Some(_) => {}
1034            }
1035            iter.advance().await?;
1036        }
1037
1038        Ok((allocated.unwrap(), end - start_offset))
1039    }
1040
1041    pub async fn txn_write<'a>(
1042        &'a self,
1043        transaction: &mut Transaction<'a>,
1044        offset: u64,
1045        buf: BufferRef<'_>,
1046    ) -> Result<(), Error> {
1047        if buf.is_empty() {
1048            return Ok(());
1049        }
1050        let (aligned, mut transfer_buf) = self.align_buffer(offset, buf).await?;
1051        self.multi_write(
1052            transaction,
1053            self.attribute_id(),
1054            std::slice::from_ref(&aligned),
1055            transfer_buf.as_mut(),
1056        )
1057        .await?;
1058        if offset + buf.len() as u64 > self.txn_get_size(transaction) {
1059            self.txn_update_size(transaction, offset + buf.len() as u64, None).await?;
1060        }
1061        Ok(())
1062    }
1063
1064    // Writes to multiple ranges with data provided in `buf`.  The buffer can be modified in place
1065    // if encryption takes place.  The ranges must all be aligned and no change to content size is
1066    // applied; the caller is responsible for updating size if required.
1067    pub async fn multi_write<'a>(
1068        &'a self,
1069        transaction: &mut Transaction<'a>,
1070        attribute_id: AttributeId,
1071        ranges: &[Range<u64>],
1072        buf: MutableBufferRef<'_>,
1073    ) -> Result<(), Error> {
1074        self.handle.multi_write(transaction, attribute_id, None, ranges, buf).await
1075    }
1076
1077    // `buf` is mutable as an optimization, since the write may require encryption, we can
1078    // encrypt the buffer in-place rather than copying to another buffer if the write is
1079    // already aligned.
1080    //
1081    // Note: in the event of power failure during an overwrite() call, it is possible that
1082    // old data (which hasn't been overwritten with new bytes yet) may be exposed to the user.
1083    // Since the old data should be encrypted, it is probably safe to expose, although not ideal.
1084    pub async fn overwrite(
1085        &self,
1086        mut offset: u64,
1087        mut buf: MutableBufferRef<'_>,
1088        options: OverwriteOptions,
1089    ) -> Result<(), Error> {
1090        ensure!((buf.len() as u32) % self.store().device.block_size() == 0, FxfsError::InvalidArgs);
1091        let end = offset + buf.len() as u64;
1092
1093        let key_id = self.get_key(None).await?.0;
1094
1095        // The transaction only ends up being used if allow_allocations is true
1096        let mut transaction =
1097            if options.allow_allocations { Some(self.new_transaction().await?) } else { None };
1098
1099        // We build up a list of writes to perform later
1100        let mut writes = FuturesOrdered::new();
1101        let mut first_write = options.barrier_on_first_write;
1102
1103        // We create a new scope here, so that the merger iterator will get dropped before we try to
1104        // commit our transaction. Otherwise the transaction commit would block.
1105        {
1106            let store = self.store();
1107            let store_object_id = store.store_object_id;
1108            let allocator = store.allocator();
1109            let tree = &store.tree;
1110            let layer_set = tree.layer_set();
1111            let mut merger = layer_set.merger();
1112            let mut iter = merger
1113                .query(Query::FullRange(&ObjectKey::attribute(
1114                    self.object_id(),
1115                    self.attribute_id(),
1116                    AttributeKey::Extent(Extent::search_key_from_offset(offset)),
1117                )))
1118                .await?;
1119            let block_size = self.block_size();
1120
1121            loop {
1122                let (device_offset, bytes_to_write, should_advance) = match iter.get() {
1123                    Some(ItemRef {
1124                        key:
1125                            ObjectKey {
1126                                object_id,
1127                                data:
1128                                    ObjectKeyData::Attribute(attribute_id, AttributeKey::Extent(extent)),
1129                            },
1130                        value: ObjectValue::Extent(ExtentValue::Some { .. }),
1131                        ..
1132                    }) if *object_id == self.object_id()
1133                        && *attribute_id == self.attribute_id()
1134                        && extent.end == offset =>
1135                    {
1136                        iter.advance().await?;
1137                        continue;
1138                    }
1139                    Some(ItemRef {
1140                        key:
1141                            ObjectKey {
1142                                object_id,
1143                                data:
1144                                    ObjectKeyData::Attribute(attribute_id, AttributeKey::Extent(extent)),
1145                            },
1146                        value,
1147                        ..
1148                    }) if *object_id == self.object_id()
1149                        && *attribute_id == self.attribute_id()
1150                        && extent.start <= offset =>
1151                    {
1152                        match value {
1153                            ObjectValue::Extent(ExtentValue::Some {
1154                                device_offset,
1155                                mode: ExtentMode::Raw,
1156                                ..
1157                            }) => {
1158                                ensure!(
1159                                    block_size.is_aligned(extent)
1160                                        && block_size.is_aligned(device_offset),
1161                                    FxfsError::Inconsistent
1162                                );
1163                                let offset_within_extent = offset - extent.start;
1164                                let remaining_length_of_extent = (extent
1165                                    .end
1166                                    .checked_sub(offset)
1167                                    .ok_or(FxfsError::Inconsistent)?)
1168                                    as usize;
1169                                // Yields (device_offset, bytes_to_write, should_advance)
1170                                (
1171                                    device_offset + offset_within_extent,
1172                                    min(buf.len(), remaining_length_of_extent),
1173                                    true,
1174                                )
1175                            }
1176                            ObjectValue::Extent(ExtentValue::Some { .. }) => {
1177                                // TODO(https://fxbug.dev/42066056): Maybe we should create
1178                                // a new extent without checksums?
1179                                bail!(
1180                                    "extent from ({},{}) which overlaps offset \
1181                                        {} has the wrong extent mode",
1182                                    extent.start,
1183                                    extent.end,
1184                                    offset
1185                                )
1186                            }
1187                            _ => {
1188                                bail!(
1189                                    "overwrite failed: extent overlapping offset {} has \
1190                                      unexpected ObjectValue",
1191                                    offset
1192                                )
1193                            }
1194                        }
1195                    }
1196                    maybe_item_ref => {
1197                        if let Some(transaction) = transaction.as_mut() {
1198                            assert_eq!(options.allow_allocations, true);
1199                            assert_eq!(offset % self.block_size(), 0);
1200
1201                            // We are going to make a new extent, but let's check if there is an
1202                            // extent after us. If there is an extent after us, then we don't want
1203                            // our new extent to bump into it...
1204                            let mut bytes_to_allocate = self
1205                                .block_size()
1206                                .align_up(buf.len() as u64)
1207                                .ok_or(FxfsError::TooBig)?;
1208                            if let Some(ItemRef {
1209                                key:
1210                                    ObjectKey {
1211                                        object_id,
1212                                        data:
1213                                            ObjectKeyData::Attribute(
1214                                                attribute_id,
1215                                                AttributeKey::Extent(extent),
1216                                            ),
1217                                    },
1218                                ..
1219                            }) = maybe_item_ref
1220                            {
1221                                if *object_id == self.object_id()
1222                                    && *attribute_id == self.attribute_id()
1223                                    && offset < extent.start
1224                                {
1225                                    let bytes_until_next_extent = extent.start - offset;
1226                                    bytes_to_allocate =
1227                                        min(bytes_to_allocate, bytes_until_next_extent);
1228                                }
1229                            }
1230
1231                            let device_range = allocator
1232                                .allocate(transaction, store_object_id, bytes_to_allocate)
1233                                .await?;
1234                            let device_range_len = device_range.end - device_range.start;
1235                            transaction.add(
1236                                store_object_id,
1237                                Mutation::insert_object(
1238                                    ObjectKey::extent(
1239                                        self.object_id(),
1240                                        self.attribute_id(),
1241                                        offset..offset + device_range_len,
1242                                    ),
1243                                    ObjectValue::Extent(ExtentValue::new_raw(
1244                                        device_range.start,
1245                                        key_id,
1246                                    )),
1247                                ),
1248                            );
1249
1250                            self.update_allocated_size(transaction, device_range_len, 0).await?;
1251
1252                            // Yields (device_offset, bytes_to_write, should_advance)
1253                            (device_range.start, min(buf.len(), device_range_len as usize), false)
1254                        } else {
1255                            bail!(
1256                                "no extent overlapping offset {}, \
1257                                and new allocations are not allowed",
1258                                offset
1259                            )
1260                        }
1261                    }
1262                };
1263                let (current_buf, remaining_buf) = buf.split_at_mut(bytes_to_write);
1264                let flags = if first_write {
1265                    first_write = false;
1266                    WriteFlags::PRE_BARRIER
1267                } else {
1268                    WriteFlags::empty()
1269                };
1270                writes.push_back(self.write_at(offset, current_buf, device_offset, flags));
1271                if remaining_buf.len() == 0 {
1272                    break;
1273                } else {
1274                    buf = remaining_buf;
1275                    offset += bytes_to_write as u64;
1276                    if should_advance {
1277                        iter.advance().await?;
1278                    }
1279                }
1280            }
1281        }
1282
1283        self.store().logical_write_ops.fetch_add(1, Ordering::Relaxed);
1284        // The checksums are being ignored here, but we don't need to know them
1285        writes.try_collect::<Vec<MaybeChecksums>>().await?;
1286
1287        if let Some(mut transaction) = transaction {
1288            assert_eq!(options.allow_allocations, true);
1289            if !transaction.is_empty() {
1290                if end > self.get_size() {
1291                    self.grow(&mut transaction, self.get_size(), end).await?;
1292                }
1293                transaction.commit().await?;
1294            }
1295        }
1296
1297        Ok(())
1298    }
1299
1300    // Within a transaction, the size of the object might have changed, so get the size from there
1301    // if it exists, otherwise, fall back on the cached size.
1302    fn txn_get_size(&self, transaction: &Transaction<'_>) -> u64 {
1303        transaction
1304            .get_object_mutation(
1305                self.store().store_object_id,
1306                ObjectKey::attribute(
1307                    self.object_id(),
1308                    self.attribute_id(),
1309                    AttributeKey::Attribute,
1310                ),
1311            )
1312            .and_then(|m| {
1313                if let ObjectItem { value: ObjectValue::Attribute { size, .. }, .. } = m.item {
1314                    Some(size)
1315                } else {
1316                    None
1317                }
1318            })
1319            .unwrap_or_else(|| self.get_size())
1320    }
1321
1322    pub async fn txn_update_size<'a>(
1323        &'a self,
1324        transaction: &mut Transaction<'a>,
1325        new_size: u64,
1326        // Allow callers to update the has_overwrite_extents metadata if they want. If this is
1327        // Some it is set to the value, if None it is left unchanged.
1328        update_has_overwrite_extents: Option<bool>,
1329    ) -> Result<(), Error> {
1330        let key =
1331            ObjectKey::attribute(self.object_id(), self.attribute_id(), AttributeKey::Attribute);
1332        let mut mutation = if let Some(mutation) =
1333            transaction.get_object_mutation(self.store().store_object_id(), key.clone())
1334        {
1335            mutation.clone()
1336        } else {
1337            ObjectStoreMutation {
1338                item: self.store().tree().find(&key).await?.ok_or(FxfsError::NotFound)?,
1339                op: Operation::ReplaceOrInsert,
1340            }
1341        };
1342        if let ObjectValue::Attribute { size, has_overwrite_extents } = &mut mutation.item.value {
1343            *size = new_size;
1344            if let Some(update_has_overwrite_extents) = update_has_overwrite_extents {
1345                *has_overwrite_extents = update_has_overwrite_extents;
1346            }
1347        } else {
1348            bail!(anyhow!(FxfsError::Inconsistent).context("Unexpected object value"));
1349        }
1350        transaction.add_with_object(
1351            self.store().store_object_id(),
1352            Mutation::ObjectStore(mutation),
1353            AssocObj::Borrowed(self),
1354        );
1355        Ok(())
1356    }
1357
1358    async fn update_allocated_size(
1359        &self,
1360        transaction: &mut Transaction<'_>,
1361        allocated: u64,
1362        deallocated: u64,
1363    ) -> Result<(), Error> {
1364        self.handle.update_allocated_size(transaction, allocated, deallocated).await
1365    }
1366
1367    pub fn truncate_overwrite_ranges(&self, size: u64) -> Result<Option<bool>, Error> {
1368        let cutoff = self.block_size().align_up(size).ok_or(FxfsError::TooBig)?;
1369        if self.with_overwrite_ranges_mut(|ranges| ranges.map_or(false, |r| r.truncate(cutoff))) {
1370            // This returns true if there were ranges, but this truncate removed them all, which
1371            // indicates that we need to flip the has_overwrite_extents metadata flag to false.
1372            Ok(Some(false))
1373        } else {
1374            Ok(None)
1375        }
1376    }
1377
1378    pub async fn shrink<'a>(
1379        &'a self,
1380        transaction: &mut Transaction<'a>,
1381        size: u64,
1382        update_has_overwrite_extents: Option<bool>,
1383    ) -> Result<NeedsTrim, Error> {
1384        let needs_trim = self.handle.shrink(transaction, self.attribute_id(), size).await?;
1385        self.txn_update_size(transaction, size, update_has_overwrite_extents).await?;
1386        Ok(needs_trim)
1387    }
1388
1389    pub async fn grow<'a>(
1390        &'a self,
1391        transaction: &mut Transaction<'a>,
1392        old_size: u64,
1393        size: u64,
1394    ) -> Result<(), Error> {
1395        // Before growing the file, we must make sure that a previous trim has completed.
1396        let store = self.store();
1397        while matches!(
1398            store
1399                .trim_some(
1400                    transaction,
1401                    self.object_id(),
1402                    self.attribute_id(),
1403                    TrimMode::FromOffset(old_size)
1404                )
1405                .await?,
1406            TrimResult::Incomplete
1407        ) {
1408            transaction.commit_and_continue().await?;
1409        }
1410        // We might need to zero out the tail of the old last block.
1411        let block_size = self.block_size();
1412        if !block_size.is_aligned(old_size) {
1413            let layer_set = store.tree.layer_set();
1414            let mut merger = layer_set.merger();
1415            let aligned_old_size = block_size.align_down(old_size);
1416            let iter = merger
1417                .query(Query::FullRange(&ObjectKey::attribute(
1418                    self.object_id(),
1419                    self.attribute_id(),
1420                    AttributeKey::Extent(Extent::search_key_from_offset(aligned_old_size)),
1421                )))
1422                .await?;
1423            if let Some(ItemRef {
1424                key:
1425                    ObjectKey {
1426                        object_id,
1427                        data:
1428                            ObjectKeyData::Attribute(attribute_id, AttributeKey::Extent(extent_key)),
1429                    },
1430                value: ObjectValue::Extent(ExtentValue::Some { device_offset, key_id, .. }),
1431                ..
1432            }) = iter.get()
1433            {
1434                if *object_id == self.object_id() && *attribute_id == self.attribute_id() {
1435                    let device_offset = device_offset
1436                        .checked_add(aligned_old_size - extent_key.start)
1437                        .ok_or(FxfsError::Inconsistent)?;
1438                    ensure!(block_size.is_aligned(device_offset), FxfsError::Inconsistent);
1439                    let mut buf = self.allocate_buffer(block_size.get() as usize).await;
1440                    // In the case that this extent is in OverwritePartial mode, there is a
1441                    // possibility that the last block is allocated, but not initialized yet, in
1442                    // which case we don't actually need to bother zeroing out the tail. However,
1443                    // it's not strictly incorrect to change uninitialized data, so we skip the
1444                    // check and blindly do it to keep it simpler here.
1445                    self.read_and_decrypt(device_offset, aligned_old_size, buf.as_mut(), *key_id)
1446                        .await?;
1447                    buf.subslice_mut((old_size % block_size) as usize..buf.len()).fill(0);
1448                    self.multi_write(
1449                        transaction,
1450                        *attribute_id,
1451                        &[aligned_old_size..aligned_old_size + block_size],
1452                        buf.as_mut(),
1453                    )
1454                    .await?;
1455                }
1456            }
1457        }
1458        self.txn_update_size(transaction, size, None).await?;
1459        Ok(())
1460    }
1461
1462    /// Attempts to pre-allocate a `file_range` of bytes for this object.
1463    /// Returns a set of device ranges (i.e. potentially multiple extents).
1464    ///
1465    /// It may not be possible to preallocate the entire requested range in one request
1466    /// due to limitations on transaction size. In such cases, we will preallocate as much as
1467    /// we can up to some (arbitrary, internal) limit on transaction size.
1468    ///
1469    /// `file_range.start` is modified to point at the end of the logical range
1470    /// that was preallocated such that repeated calls to `preallocate_range` with new
1471    /// transactions can be used to preallocate ranges of any size.
1472    ///
1473    /// Requested range must be a multiple of block size.
1474    pub async fn preallocate_range<'a>(
1475        &'a self,
1476        transaction: &mut Transaction<'a>,
1477        file_range: &mut Range<u64>,
1478    ) -> Result<Vec<Range<u64>>, Error> {
1479        let block_size = self.block_size();
1480        ensure!(block_size.is_aligned(&*file_range), FxfsError::InvalidArgs);
1481        ensure!(!self.handle.is_encrypted(), FxfsError::NotSupported);
1482        let mut ranges = Vec::new();
1483        let tree = &self.store().tree;
1484        let layer_set = tree.layer_set();
1485        let mut merger = layer_set.merger();
1486        let mut iter = merger
1487            .query(Query::FullRange(&ObjectKey::attribute(
1488                self.object_id(),
1489                self.attribute_id(),
1490                AttributeKey::Extent(Extent::search_key_from_offset(file_range.start)),
1491            )))
1492            .await?;
1493        let mut allocated = 0;
1494        let key_id = self.get_key(None).await?.0;
1495        'outer: while file_range.start < file_range.end {
1496            let allocate_end = loop {
1497                match iter.get() {
1498                    // Case for allocated extents for the same object that overlap with file_range.
1499                    Some(ItemRef {
1500                        key:
1501                            ObjectKey {
1502                                object_id,
1503                                data:
1504                                    ObjectKeyData::Attribute(attribute_id, AttributeKey::Extent(extent)),
1505                            },
1506                        value: ObjectValue::Extent(ExtentValue::Some { device_offset, .. }),
1507                        ..
1508                    }) if *object_id == self.object_id()
1509                        && *attribute_id == self.attribute_id()
1510                        && extent.start < file_range.end =>
1511                    {
1512                        ensure!(
1513                            extent.is_valid()
1514                                && block_size.is_aligned(extent)
1515                                && block_size.is_aligned(device_offset),
1516                            FxfsError::Inconsistent
1517                        );
1518                        // If the start of the requested file_range overlaps with an existing extent...
1519                        if extent.start <= file_range.start {
1520                            // Record the existing extent and move on.
1521                            let device_range = device_offset
1522                                .checked_add(file_range.start - extent.start)
1523                                .ok_or(FxfsError::Inconsistent)?
1524                                ..device_offset
1525                                    .checked_add(min(extent.end, file_range.end) - extent.start)
1526                                    .ok_or(FxfsError::Inconsistent)?;
1527                            file_range.start += device_range.end - device_range.start;
1528                            ranges.push(device_range);
1529                            if file_range.start >= file_range.end {
1530                                break 'outer;
1531                            }
1532                            iter.advance().await?;
1533                            continue;
1534                        } else {
1535                            // There's nothing allocated between file_range.start and the beginning
1536                            // of this extent.
1537                            break extent.start;
1538                        }
1539                    }
1540                    // Case for deleted extents eclipsed by file_range.
1541                    Some(ItemRef {
1542                        key:
1543                            ObjectKey {
1544                                object_id,
1545                                data:
1546                                    ObjectKeyData::Attribute(attribute_id, AttributeKey::Extent(extent)),
1547                            },
1548                        value: ObjectValue::Extent(ExtentValue::None),
1549                        ..
1550                    }) if *object_id == self.object_id()
1551                        && *attribute_id == self.attribute_id()
1552                        && extent.end < file_range.end =>
1553                    {
1554                        iter.advance().await?;
1555                    }
1556                    _ => {
1557                        // We can just preallocate the rest.
1558                        break file_range.end;
1559                    }
1560                }
1561            };
1562            let device_range = self
1563                .store()
1564                .allocator()
1565                .allocate(
1566                    transaction,
1567                    self.store().store_object_id(),
1568                    allocate_end - file_range.start,
1569                )
1570                .await
1571                .context("Allocation failed")?;
1572            allocated += device_range.end - device_range.start;
1573            let this_file_range =
1574                file_range.start..file_range.start + device_range.end - device_range.start;
1575            file_range.start = this_file_range.end;
1576            transaction.add(
1577                self.store().store_object_id,
1578                Mutation::merge_object(
1579                    ObjectKey::extent(self.object_id(), self.attribute_id(), this_file_range),
1580                    ObjectValue::Extent(ExtentValue::new_raw(device_range.start, key_id)),
1581                ),
1582            );
1583            ranges.push(device_range);
1584            // If we didn't allocate all that we requested, we'll loop around and try again.
1585            // ... unless we have filled the transaction. The caller should check file_range.
1586            if transaction.mutations().len() > TRANSACTION_MUTATION_THRESHOLD {
1587                break;
1588            }
1589        }
1590        // Update the file size if it changed.
1591        if file_range.start > block_size.align_up(self.txn_get_size(transaction)).unwrap() {
1592            self.txn_update_size(transaction, file_range.start, None).await?;
1593        }
1594        self.update_allocated_size(transaction, allocated, 0).await?;
1595        Ok(ranges)
1596    }
1597
1598    pub async fn update_attributes<'a>(
1599        &self,
1600        transaction: &mut Transaction<'a>,
1601        node_attributes: Option<&fio::MutableNodeAttributes>,
1602        change_time: Option<Timestamp>,
1603    ) -> Result<(), Error> {
1604        // This codepath is only called by files, whose wrapping key id users cannot directly set
1605        // as per fscrypt.
1606        ensure!(
1607            !matches!(
1608                node_attributes,
1609                Some(fio::MutableNodeAttributes { encryption_policy: Some(_), .. })
1610            ),
1611            FxfsError::BadPath
1612        );
1613        self.handle.update_attributes(transaction, node_attributes, change_time).await
1614    }
1615
1616    /// Get the default set of transaction options for this object. This is mostly the overall
1617    /// default, modified by any [`HandleOptions`] held by this handle.
1618    pub fn default_transaction_options<'b>(&self) -> Options<'b> {
1619        self.handle.default_transaction_options()
1620    }
1621
1622    pub async fn new_transaction<'b>(&self) -> Result<Transaction<'b>, Error> {
1623        self.new_transaction_with_options(self.default_transaction_options()).await
1624    }
1625
1626    pub async fn new_transaction_with_options<'b>(
1627        &self,
1628        options: Options<'b>,
1629    ) -> Result<Transaction<'b>, Error> {
1630        self.handle.new_transaction_with_options(self.attribute_id(), options).await
1631    }
1632
1633    /// Flushes the underlying device.  This is expensive and should be used sparingly.
1634    pub async fn flush_device(&self) -> Result<(), Error> {
1635        self.handle.flush_device().await
1636    }
1637
1638    /// Reads an entire attribute.
1639    pub async fn read_attr(&self, attribute_id: AttributeId) -> Result<Option<Box<[u8]>>, Error> {
1640        self.handle.read_attr(attribute_id).await
1641    }
1642
1643    /// Writes an entire attribute.  This *always* uses the volume data key.
1644    pub async fn write_attr(&self, attribute_id: AttributeId, data: &[u8]) -> Result<(), Error> {
1645        // Must be different attribute otherwise cached size gets out of date.
1646        assert_ne!(attribute_id, self.attribute_id());
1647        let store = self.store();
1648        let mut transaction = self.new_transaction().await?;
1649        if self.handle.write_attr(&mut transaction, attribute_id, data).await?.0 {
1650            transaction.commit_and_continue().await?;
1651            while matches!(
1652                store
1653                    .trim_some(
1654                        &mut transaction,
1655                        self.object_id(),
1656                        attribute_id,
1657                        TrimMode::FromOffset(data.len() as u64),
1658                    )
1659                    .await?,
1660                TrimResult::Incomplete
1661            ) {
1662                transaction.commit_and_continue().await?;
1663            }
1664        }
1665        transaction.commit().await?;
1666        Ok(())
1667    }
1668
1669    async fn read_and_decrypt(
1670        &self,
1671        device_offset: u64,
1672        file_offset: u64,
1673        buffer: MutableBufferRef<'_>,
1674        key_id: u64,
1675    ) -> Result<(), Error> {
1676        self.handle
1677            .read_and_decrypt(self.attribute_id, device_offset, file_offset, buffer, key_id)
1678            .await
1679    }
1680
1681    /// Truncates a file to a given size (growing/shrinking as required).
1682    ///
1683    /// Nb: Most code will want to call truncate() instead. This method is used
1684    /// to update the super block -- a case where we must borrow metadata space.
1685    pub async fn truncate_with_options(
1686        &self,
1687        options: Options<'_>,
1688        size: u64,
1689    ) -> Result<(), Error> {
1690        let mut transaction = self.new_transaction_with_options(options).await?;
1691        {
1692            let state = self.state.lock();
1693            match &*state {
1694                DataObjectState::Standard(_) => {}
1695                _ => bail!(anyhow!(FxfsError::AccessDenied).context("Cannot truncate verity file")),
1696            }
1697        }
1698        let old_size = self.get_size();
1699        if size == old_size {
1700            return Ok(());
1701        }
1702        if size < old_size {
1703            let update_has_overwrite_ranges = self.truncate_overwrite_ranges(size)?;
1704            if self.shrink(&mut transaction, size, update_has_overwrite_ranges).await?.0 {
1705                // The file needs to be trimmed.
1706                transaction.commit_and_continue().await?;
1707                let store = self.store();
1708                while matches!(
1709                    store
1710                        .trim_some(
1711                            &mut transaction,
1712                            self.object_id(),
1713                            self.attribute_id(),
1714                            TrimMode::FromOffset(size)
1715                        )
1716                        .await?,
1717                    TrimResult::Incomplete
1718                ) {
1719                    if let Err(error) = transaction.commit_and_continue().await {
1720                        warn!(error:?; "Failed to trim after truncate");
1721                        return Ok(());
1722                    }
1723                }
1724                if let Err(error) = transaction.commit().await {
1725                    warn!(error:?; "Failed to trim after truncate");
1726                }
1727                return Ok(());
1728            }
1729        } else {
1730            self.grow(&mut transaction, old_size, size).await?;
1731        }
1732        transaction.commit().await?;
1733        Ok(())
1734    }
1735
1736    pub async fn get_properties(&self) -> Result<ObjectProperties, Error> {
1737        // We don't take a read guard here since the object properties are contained in a single
1738        // object, which cannot be inconsistent with itself. The LSM tree does not return
1739        // intermediate states for a single object.
1740        let value = self
1741            .store()
1742            .tree
1743            .find_value(&ObjectKey::object(self.object_id()))
1744            .await?
1745            .expect("Unable to find object record");
1746        match value {
1747            ObjectValue::Object {
1748                kind: ObjectKind::File { refs, .. },
1749                attributes:
1750                    ObjectAttributes {
1751                        creation_time,
1752                        modification_time,
1753                        posix_attributes,
1754                        allocated_size,
1755                        access_time,
1756                        change_time,
1757                        ..
1758                    },
1759            } => Ok(ObjectProperties {
1760                refs,
1761                allocated_size,
1762                data_attribute_size: self.get_size(),
1763                creation_time,
1764                modification_time,
1765                access_time,
1766                change_time,
1767                sub_dirs: 0,
1768                posix_attributes,
1769                dir_type: DirType::Normal,
1770            }),
1771            _ => bail!(FxfsError::NotFile),
1772        }
1773    }
1774
1775    /// Returns the set of file_offset->extent mappings for this file. The extents will be sorted by
1776    /// their logical offset within the file.
1777    ///
1778    /// *NOTE*: This operation is potentially expensive and should generally be avoided.
1779    pub async fn device_extents(&self) -> Result<Vec<FileExtent>, Error> {
1780        let tree = &self.store().tree;
1781        let layer_set = tree.layer_set();
1782        let mut merger = layer_set.merger();
1783        let stream = self.handle.extent_stream(&mut merger, self.attribute_id()).await?;
1784        let extents: Vec<FileExtent> = stream.try_collect().await?;
1785        Ok(extents)
1786    }
1787
1788    /// Returns the contents of this object. This object must be < |limit| bytes in size.
1789    pub async fn contents(&self, limit: usize) -> Result<Box<[u8]>, Error> {
1790        let size = self.get_size();
1791        if size > limit as u64 {
1792            bail!("Object too big ({} > {})", size, limit);
1793        }
1794        self.read_bytes(0..size).await
1795    }
1796
1797    /// Reads the contents of this object within `range`. `range` does not need to be block aligned.
1798    ///
1799    /// This method should be avoided in high-performance contexts. Use `read_aligned` instead.
1800    pub async fn read_bytes(&self, range: Range<u64>) -> Result<Box<[u8]>, Error> {
1801        const MAX_READ_BYTES_CHUNK_SIZE: usize = 2 * 1024 * 1024; // 2 MiB
1802
1803        ensure!(range.start <= range.end, FxfsError::InvalidArgs);
1804        if range.is_empty() {
1805            return Ok(Box::default());
1806        }
1807        let fs = self.store().filesystem();
1808        let guard = fs
1809            .lock_manager()
1810            .read_lock(lock_keys![LockKey::object_attribute(
1811                self.store().store_object_id,
1812                self.object_id(),
1813                self.attribute_id(),
1814            )])
1815            .await;
1816
1817        let size = self.get_size();
1818        if range.start >= size {
1819            return Ok(Box::default());
1820        }
1821        let end = std::cmp::min(range.end, size);
1822        let total_to_read = (end - range.start) as usize;
1823        let block_size = self.block_size();
1824        let aligned_start = block_size.align_down(range.start);
1825        let aligned_end = block_size.align_up(end).ok_or(FxfsError::TooBig)?;
1826        let total_aligned_len = aligned_end - aligned_start;
1827
1828        let buf_size = std::cmp::min(total_aligned_len, MAX_READ_BYTES_CHUNK_SIZE as u64) as usize;
1829        let mut buf = self.allocate_buffer(buf_size).await;
1830        let mut out = Vec::with_capacity(total_to_read);
1831        let mut current_block_offset = aligned_start;
1832
1833        while current_block_offset < end {
1834            let bytes_read =
1835                self.read_aligned_locked(current_block_offset, buf.as_mut(), &guard).await?;
1836            let chunk_start = current_block_offset;
1837            let chunk_end = current_block_offset + bytes_read as u64;
1838            let slice_start = std::cmp::max(range.start, chunk_start);
1839            let slice_end = std::cmp::min(end, chunk_end);
1840            let buf_offset = (slice_start - chunk_start) as usize;
1841            let to_copy = (slice_end - slice_start) as usize;
1842            buf.subslice(buf_offset..buf_offset + to_copy).append_to(&mut out);
1843            current_block_offset = current_block_offset.saturating_add(buf_size as u64);
1844        }
1845        Ok(out.into_boxed_slice())
1846    }
1847
1848    /// Reads up to `buf.len()` bytes from `offset` while holding `guard`.
1849    ///
1850    /// Both `offset` and `buf.len()` must be aligned to the object's `block_size()`.
1851    ///
1852    /// `guard` must be an active read lock for this object's attribute.
1853    ///
1854    /// Returns the number of bytes read up to the object's size (or 0 if `offset >= size`). Holes/
1855    /// sparse extents within the read range are zero-filled. Callers should not make any
1856    /// assumptions about the contents of the buffer past the returned read amount. If this is a
1857    /// verified file (fs-verity), the data is validated against the Merkle tree before returning.
1858    async fn read_aligned_locked(
1859        &self,
1860        offset: u64,
1861        mut buf: MutableBufferRef<'_>,
1862        guard: &ReadGuard<'_>,
1863    ) -> Result<usize, Error> {
1864        let block_size = self.block_size();
1865        debug_assert!(block_size.is_aligned(offset));
1866        debug_assert!(block_size.is_aligned(buf.len() as u64));
1867
1868        let size = self.get_size();
1869        if offset >= size {
1870            return Ok(0);
1871        }
1872        let length = min(buf.len() as u64, size - offset) as usize;
1873        let aligned_length =
1874            block_size.align_up(length as u64).ok_or(FxfsError::Inconsistent)? as usize;
1875        buf = buf.subslice_mut(0..aligned_length);
1876
1877        self.handle
1878            .read_aligned_unchecked(self.attribute_id(), offset, buf.reborrow(), guard)
1879            .await?;
1880        if self.is_verified_file() {
1881            self.verify_data(offset as usize, buf.subslice(0..length).as_ptr_slice())?;
1882        }
1883        Ok(length)
1884    }
1885
1886    /// Fills `buf` with bytes read from `offset` on the underlying device.
1887    ///
1888    /// Both `offset` and `buf.len()` must be aligned to the object's `block_size()`.
1889    ///
1890    /// Returns the number of bytes read. If `offset >= size`, returns 0. Holes/sparse extents
1891    /// within the read range are zero-filled. Callers should not make any assumptions about the
1892    /// contents of the buffer past the returned read amount.
1893    ///
1894    /// If the object is a verified file (fs-verity), the read data is validated against the Merkle
1895    /// tree before returning.
1896    ///
1897    /// This is an inherent version of the `ReadObjectHandle::read_aligned` trait method which
1898    /// avoids boxing the returned `Future`.
1899    pub async fn read_aligned(
1900        &self,
1901        offset: u64,
1902        buf: MutableBufferRef<'_>,
1903    ) -> Result<usize, Error> {
1904        let block_size = self.block_size();
1905        ensure!(block_size.is_aligned(offset), FxfsError::InvalidArgs);
1906        ensure!(block_size.is_aligned(buf.len() as u64), FxfsError::InvalidArgs);
1907        let fs = self.store().filesystem();
1908        let guard = fs
1909            .lock_manager()
1910            .read_lock(lock_keys![LockKey::object_attribute(
1911                self.store().store_object_id,
1912                self.object_id(),
1913                self.attribute_id(),
1914            )])
1915            .await;
1916        self.read_aligned_locked(offset, buf, &guard).await
1917    }
1918}
1919
1920impl<S: HandleOwner> AssociatedObject for DataObjectHandle<S> {
1921    fn will_apply_mutation(&self, mutation: &Mutation, _object_id: u64, _manager: &ObjectManager) {
1922        match mutation {
1923            Mutation::ObjectStore(ObjectStoreMutation {
1924                item: ObjectItem { value: ObjectValue::Attribute { size, .. }, .. },
1925                ..
1926            }) => self.content_size.store(*size, atomic::Ordering::Relaxed),
1927            Mutation::ObjectStore(ObjectStoreMutation {
1928                item: ObjectItem { value: ObjectValue::VerifiedAttribute { size, .. }, .. },
1929                ..
1930            }) => {
1931                debug_assert_eq!(
1932                    self.get_size(),
1933                    *size,
1934                    "size should be set when verity is enabled and must not change"
1935                );
1936                self.finalize_fsverity_state()
1937            }
1938            Mutation::ObjectStore(ObjectStoreMutation {
1939                item:
1940                    ObjectItem {
1941                        key:
1942                            ObjectKey {
1943                                object_id,
1944                                data:
1945                                    ObjectKeyData::Attribute(attr_id, AttributeKey::Extent(extent)),
1946                            },
1947                        value: ObjectValue::Extent(ExtentValue::Some { mode, .. }),
1948                        ..
1949                    },
1950                ..
1951            }) if self.object_id() == *object_id && self.attribute_id() == *attr_id => match mode {
1952                ExtentMode::Overwrite | ExtentMode::OverwritePartial(_) => {
1953                    // If `enable_verity` transitioned state to `VerityStarted` concurrently while a
1954                    // transaction was in progress, `with_overwrite_ranges_mut` will return `None`
1955                    // and safely discard in-memory range updates, since verity files do not track
1956                    // overwrite ranges.
1957                    self.with_overwrite_ranges_mut(|ranges| {
1958                        if let Some(ranges) = ranges {
1959                            ranges.apply_range(extent.clone().into());
1960                        }
1961                    });
1962                }
1963                ExtentMode::Raw | ExtentMode::Cow(_) => (),
1964            },
1965            _ => {}
1966        }
1967    }
1968}
1969
1970impl<S: HandleOwner> ObjectHandle for DataObjectHandle<S> {
1971    fn set_trace(&self, v: bool) {
1972        self.handle.set_trace(v)
1973    }
1974
1975    fn object_id(&self) -> u64 {
1976        self.handle.object_id()
1977    }
1978
1979    fn allocate_buffer(&self, size: usize) -> BufferFuture<'_> {
1980        self.handle.allocate_buffer(size)
1981    }
1982
1983    fn block_size(&self) -> BlockSize {
1984        self.handle.block_size()
1985    }
1986}
1987
1988impl<S: HandleOwner> ReadObjectHandle for DataObjectHandle<S> {
1989    fn read_aligned<'a, 'b, 'c>(
1990        &'a self,
1991        offset: u64,
1992        buf: MutableBufferRef<'b>,
1993    ) -> Pin<Box<dyn Future<Output = Result<usize, Error>> + Send + 'c>>
1994    where
1995        'a: 'c,
1996        'b: 'c,
1997        Self: 'c,
1998    {
1999        Box::pin(DataObjectHandle::read_aligned(self, offset, buf))
2000    }
2001
2002    fn get_size(&self) -> u64 {
2003        self.content_size.load(atomic::Ordering::Relaxed)
2004    }
2005}
2006
2007impl<S: HandleOwner> LayerObject for DataObjectHandle<S> {}
2008
2009impl<S: HandleOwner> WriteObjectHandle for DataObjectHandle<S> {
2010    async fn write_or_append(&self, offset: Option<u64>, buf: BufferRef<'_>) -> Result<u64, Error> {
2011        let offset = offset.unwrap_or_else(|| self.get_size());
2012        let mut transaction = self.new_transaction().await?;
2013        self.txn_write(&mut transaction, offset, buf).await?;
2014        let new_size = self.txn_get_size(&transaction);
2015        transaction.commit().await?;
2016        Ok(new_size)
2017    }
2018
2019    async fn truncate(&self, size: u64) -> Result<(), Error> {
2020        self.truncate_with_options(self.default_transaction_options(), size).await
2021    }
2022
2023    async fn flush(&self) -> Result<(), Error> {
2024        Ok(())
2025    }
2026}
2027
2028/// Like object_handle::Writer, but allows custom transaction options to be set, and makes every
2029/// write go directly to the handle in a transaction.
2030pub struct DirectWriter<'a, S: HandleOwner> {
2031    handle: &'a DataObjectHandle<S>,
2032    options: transaction::Options<'a>,
2033    buffer: Buffer<'a>,
2034    offset: u64,
2035    buf_offset: usize,
2036}
2037
2038const BUFFER_SIZE: usize = 1_048_576;
2039
2040impl<S: HandleOwner> Drop for DirectWriter<'_, S> {
2041    fn drop(&mut self) {
2042        if self.buf_offset != 0 {
2043            warn!("DirectWriter: dropping data, did you forget to call complete?");
2044        }
2045    }
2046}
2047
2048impl<'a, S: HandleOwner> DirectWriter<'a, S> {
2049    pub async fn new(
2050        handle: &'a DataObjectHandle<S>,
2051        options: transaction::Options<'a>,
2052    ) -> DirectWriter<'a, S> {
2053        Self {
2054            handle,
2055            options,
2056            buffer: handle.allocate_buffer(BUFFER_SIZE).await,
2057            offset: 0,
2058            buf_offset: 0,
2059        }
2060    }
2061
2062    async fn flush(&mut self) -> Result<(), Error> {
2063        let mut transaction = self.handle.new_transaction_with_options(self.options).await?;
2064        self.handle
2065            .txn_write(&mut transaction, self.offset, self.buffer.subslice(0..self.buf_offset))
2066            .await?;
2067        transaction.commit().await?;
2068        self.offset += self.buf_offset as u64;
2069        self.buf_offset = 0;
2070        Ok(())
2071    }
2072}
2073
2074impl<'a, S: HandleOwner> WriteBytes for DirectWriter<'a, S> {
2075    fn block_size(&self) -> BlockSize {
2076        self.handle.block_size()
2077    }
2078
2079    async fn write_bytes(&mut self, mut buf: &[u8]) -> Result<(), Error> {
2080        while buf.len() > 0 {
2081            let to_do = std::cmp::min(buf.len(), BUFFER_SIZE - self.buf_offset);
2082            self.buffer
2083                .subslice_mut(self.buf_offset..self.buf_offset + to_do)
2084                .copy_from_slice(&buf[..to_do]);
2085            self.buf_offset += to_do;
2086            if self.buf_offset == BUFFER_SIZE {
2087                self.flush().await?;
2088            }
2089            buf = &buf[to_do..];
2090        }
2091        Ok(())
2092    }
2093
2094    async fn complete(mut self) -> Result<u64, Error> {
2095        self.flush().await?;
2096        Ok(self.offset + self.buf_offset as u64)
2097    }
2098
2099    async fn skip(&mut self, amount: u64) -> Result<(), Error> {
2100        if (BUFFER_SIZE - self.buf_offset) as u64 > amount {
2101            self.buffer.subslice_mut(self.buf_offset..self.buf_offset + amount as usize).fill(0);
2102            self.buf_offset += amount as usize;
2103        } else {
2104            self.flush().await?;
2105            self.offset += amount;
2106        }
2107        Ok(())
2108    }
2109}
2110
2111#[cfg(test)]
2112mod tests {
2113    use crate::errors::FxfsError;
2114    use crate::filesystem::{FxFilesystem, FxFilesystemBuilder, OpenFxFilesystem, SyncOptions};
2115    use crate::fsck::{
2116        FsckOptions, fsck, fsck_volume, fsck_volume_with_options, fsck_with_options,
2117    };
2118    use crate::lsm_tree::Query;
2119    use crate::lsm_tree::types::{ItemRef, LayerIterator};
2120    use crate::object_handle::{
2121        ObjectHandle, ObjectProperties, ReadObjectHandle, WriteObjectHandle,
2122    };
2123    use crate::object_store::data_object_handle::{OverwriteOptions, WRITE_ATTR_BATCH_SIZE};
2124    use crate::object_store::directory::replace_child;
2125    use crate::object_store::object_record::{FsverityMetadata, ObjectKey, ObjectValue, Timestamp};
2126    use crate::object_store::transaction::{Mutation, Options, ReservationOptions, lock_keys};
2127    use crate::object_store::volume::root_volume;
2128    use crate::object_store::{
2129        AttributeId, AttributeKey, DataObjectHandle, DirType, Directory, Extent, ExtentMode,
2130        ExtentValue, HandleOptions, LockKey, NewChildStoreOptions, ObjectKeyData, ObjectStore,
2131        PosixAttributes, StoreOptions, TRANSACTION_MUTATION_THRESHOLD,
2132    };
2133    use crate::range::RangeExt;
2134    use crate::round::{round_down, round_up};
2135    use assert_matches::assert_matches;
2136    use bit_vec::BitVec;
2137    use fidl_fuchsia_io as fio;
2138    use fsverity_merkle::{FsVerityDescriptor, FsVerityDescriptorRaw};
2139    use fuchsia_async as fasync;
2140    use fuchsia_sync::Mutex;
2141    use futures::FutureExt;
2142    use futures::channel::oneshot::channel;
2143    use futures::stream::{FuturesUnordered, StreamExt};
2144    use fxfs_crypto::{Crypt, EncryptionKey, KeyPurpose};
2145    use fxfs_insecure_crypto::new_insecure_crypt;
2146    use std::ops::Range;
2147    use std::sync::Arc;
2148    use std::time::Duration;
2149    use storage_device::DeviceHolder;
2150    use storage_device::fake_device::FakeDevice;
2151
2152    const TEST_DEVICE_BLOCK_SIZE: u32 = 512;
2153
2154    // Some tests (the preallocate_range ones) currently assume that the data only occupies a single
2155    // device block.
2156    const TEST_DATA_OFFSET: u64 = 5000;
2157    const TEST_DATA: &[u8] = b"hello";
2158    const TEST_OBJECT_SIZE: u64 = 5678;
2159    const TEST_OBJECT_ALLOCATED_SIZE: u64 = 4096;
2160    const TEST_OBJECT_NAME: &str = "foo";
2161
2162    async fn test_filesystem() -> OpenFxFilesystem {
2163        let device = DeviceHolder::new(FakeDevice::new(8192, TEST_DEVICE_BLOCK_SIZE));
2164        FxFilesystem::new_empty(device).await.expect("new_empty failed")
2165    }
2166
2167    async fn create_object_with_key(
2168        fs: Arc<FxFilesystem>,
2169        crypt: Option<&dyn Crypt>,
2170        write_object_test_data: bool,
2171    ) -> DataObjectHandle<ObjectStore> {
2172        let store = fs.root_store();
2173        let object;
2174
2175        let mut transaction = fs
2176            .root_store()
2177            .new_transaction(
2178                lock_keys![LockKey::object(
2179                    store.store_object_id(),
2180                    store.root_directory_object_id()
2181                )],
2182                Options::default(),
2183            )
2184            .await
2185            .expect("new_transaction failed");
2186
2187        object = if let Some(crypt) = crypt {
2188            let object_id = store.get_next_object_id(&transaction).await.unwrap();
2189            let (key, unwrapped_key) =
2190                crypt.create_key(object_id.get(), KeyPurpose::Data).await.unwrap();
2191            ObjectStore::create_object_with_key(
2192                &store,
2193                &mut transaction,
2194                object_id,
2195                HandleOptions::default(),
2196                EncryptionKey::Fxfs(key),
2197                unwrapped_key,
2198            )
2199            .await
2200            .expect("create_object failed")
2201        } else {
2202            ObjectStore::create_object(&store, &mut transaction, HandleOptions::default(), None)
2203                .await
2204                .expect("create_object failed")
2205        };
2206
2207        let root_directory =
2208            Directory::open(&store, store.root_directory_object_id()).await.expect("open failed");
2209        root_directory
2210            .add_child_file(&mut transaction, TEST_OBJECT_NAME, &object)
2211            .await
2212            .expect("add_child_file failed");
2213
2214        if write_object_test_data {
2215            let align = TEST_DATA_OFFSET as usize % TEST_DEVICE_BLOCK_SIZE as usize;
2216            let mut buf = object.allocate_buffer(align + TEST_DATA.len()).await;
2217            buf.subslice_mut(align..buf.len()).copy_from_slice(TEST_DATA);
2218            object
2219                .txn_write(&mut transaction, TEST_DATA_OFFSET, buf.subslice(align..buf.len()))
2220                .await
2221                .expect("write failed");
2222        }
2223        transaction.commit().await.expect("commit failed");
2224        object.truncate(TEST_OBJECT_SIZE).await.expect("truncate failed");
2225        object
2226    }
2227
2228    async fn test_filesystem_and_object_with_key(
2229        crypt: Option<&dyn Crypt>,
2230        write_object_test_data: bool,
2231    ) -> (OpenFxFilesystem, DataObjectHandle<ObjectStore>) {
2232        let fs = test_filesystem().await;
2233        let object = create_object_with_key(fs.clone(), crypt, write_object_test_data).await;
2234        (fs, object)
2235    }
2236
2237    async fn test_filesystem_and_object() -> (OpenFxFilesystem, DataObjectHandle<ObjectStore>) {
2238        test_filesystem_and_object_with_key(Some(&new_insecure_crypt()), true).await
2239    }
2240
2241    async fn test_filesystem_and_empty_object() -> (OpenFxFilesystem, DataObjectHandle<ObjectStore>)
2242    {
2243        test_filesystem_and_object_with_key(Some(&new_insecure_crypt()), false).await
2244    }
2245
2246    #[fuchsia::test]
2247    async fn test_zero_buf_len_read() {
2248        let (fs, object) = test_filesystem_and_object().await;
2249        let mut buf = object.allocate_buffer(0).await;
2250        assert_eq!(object.read_aligned(0u64, buf.as_mut()).await.expect("read failed"), 0);
2251        fs.close().await.expect("Close failed");
2252    }
2253
2254    #[fuchsia::test]
2255    async fn test_beyond_eof_read() {
2256        let (fs, object) = test_filesystem_and_object().await;
2257        let offset = TEST_OBJECT_SIZE as usize - 2;
2258        let align = (offset as u64 % fs.block_size()) as usize;
2259        let len: usize = 2;
2260        let block_size = fs.block_size().get() as usize;
2261
2262        // Unaligned buffer should fail.
2263        let mut unaligned_buf = object.allocate_buffer(align + len + 1).await;
2264        assert_matches!(
2265            object.read_aligned((offset - align) as u64, unaligned_buf.as_mut()).await,
2266            Err(e) if FxfsError::InvalidArgs.matches(&e)
2267        );
2268
2269        let mut buf = object.allocate_buffer(block_size).await;
2270        buf.fill(123u8);
2271        assert_eq!(
2272            object.read_aligned((offset - align) as u64, buf.as_mut()).await.expect("read failed"),
2273            align + len
2274        );
2275        assert_eq!(&buf.as_ptr_slice().subslice(align..align + len).to_vec()[..], &vec![0u8; len]);
2276
2277        // Unaligned offset should fail.
2278        assert_matches!(
2279            object.read_aligned((offset - align + 1) as u64, buf.as_mut()).await,
2280            Err(e) if FxfsError::InvalidArgs.matches(&e)
2281        );
2282
2283        // Reading starting at or after EOF should return 0.
2284        let aligned_eof = fs.block_size().align_up(TEST_OBJECT_SIZE).unwrap();
2285        assert_eq!(object.read_aligned(aligned_eof, buf.as_mut()).await.expect("read failed"), 0);
2286
2287        // Also test read_bytes past EOF.
2288        let data = object
2289            .read_bytes(offset as u64..offset as u64 + len as u64 + 1)
2290            .await
2291            .expect("read_bytes failed");
2292        assert_eq!(&data[..], &vec![0u8; len]);
2293
2294        fs.close().await.expect("Close failed");
2295    }
2296
2297    #[fuchsia::test]
2298    async fn test_beyond_eof_read_from() {
2299        let (fs, object) = test_filesystem_and_object().await;
2300        let handle = &*object;
2301        let offset = TEST_OBJECT_SIZE as usize - 2;
2302        let align = (offset as u64 % fs.block_size()) as usize;
2303        let len: usize = 2;
2304        let block_size = fs.block_size().get() as usize;
2305
2306        // Unaligned buffer should fail.
2307        let mut unaligned_buf = object.allocate_buffer(align + len + 1).await;
2308        assert_matches!(
2309            handle
2310                .read_aligned(AttributeId::DATA, (offset - align) as u64, unaligned_buf.as_mut())
2311                .await,
2312            Err(e) if FxfsError::InvalidArgs.matches(&e)
2313        );
2314
2315        let mut buf = object.allocate_buffer(block_size).await;
2316        buf.fill(123u8);
2317        assert_eq!(
2318            handle
2319                .read_aligned(AttributeId::DATA, (offset - align) as u64, buf.as_mut())
2320                .await
2321                .expect("read failed"),
2322            align + len
2323        );
2324        assert_eq!(&buf.as_ptr_slice().subslice(align..align + len).to_vec()[..], &vec![0u8; len]);
2325
2326        // Unaligned offset should fail.
2327        assert_matches!(
2328            handle
2329                .read_aligned(AttributeId::DATA, (offset - align + 1) as u64, buf.as_mut())
2330                .await,
2331            Err(e) if FxfsError::InvalidArgs.matches(&e)
2332        );
2333
2334        // Reading starting at or after EOF should return 0.
2335        let aligned_eof = fs.block_size().align_up(TEST_OBJECT_SIZE).unwrap();
2336        assert_eq!(
2337            handle
2338                .read_aligned(AttributeId::DATA, aligned_eof, buf.as_mut())
2339                .await
2340                .expect("read failed"),
2341            0
2342        );
2343
2344        fs.close().await.expect("Close failed");
2345    }
2346
2347    #[fuchsia::test]
2348    async fn test_beyond_eof_read_aligned_unchecked() {
2349        let (fs, object) = test_filesystem_and_object().await;
2350        let offset = TEST_OBJECT_SIZE as usize - 2;
2351        let align = (offset as u64 % fs.block_size()) as usize;
2352        let block_size = fs.block_size().get() as usize;
2353        let mut buf = object.allocate_buffer(block_size).await;
2354        buf.fill(123u8);
2355        let guard = fs
2356            .lock_manager()
2357            .read_lock(lock_keys![LockKey::object_attribute(
2358                object.store().store_object_id,
2359                object.object_id(),
2360                AttributeId::DATA,
2361            )])
2362            .await;
2363        object
2364            .read_aligned_unchecked(
2365                AttributeId::DATA,
2366                (offset - align) as u64,
2367                buf.as_mut(),
2368                &guard,
2369            )
2370            .await
2371            .expect("read failed");
2372        assert_eq!(
2373            &buf.as_ptr_slice().subslice(align..block_size).to_vec()[..],
2374            &vec![0u8; block_size - align]
2375        );
2376
2377        // Reading entirely past EOF with read_aligned_unchecked fills with zeros.
2378        let aligned_eof = fs.block_size().align_up(TEST_OBJECT_SIZE).unwrap();
2379        buf.fill(123u8);
2380        object
2381            .read_aligned_unchecked(AttributeId::DATA, aligned_eof, buf.as_mut(), &guard)
2382            .await
2383            .expect("read failed");
2384        assert_eq!(buf.to_vec(), vec![0u8; block_size]);
2385        fs.close().await.expect("Close failed");
2386    }
2387
2388    #[fuchsia::test]
2389    async fn test_read_sparse() {
2390        let (fs, object) = test_filesystem_and_object().await;
2391        // Deliberately read not right to eof.
2392        let len = TEST_OBJECT_SIZE as usize - 1;
2393        let data = object.read_bytes(0..len as u64).await.expect("read failed");
2394        let mut expected = vec![0; len];
2395        let offset = TEST_DATA_OFFSET as usize;
2396        expected[offset..offset + TEST_DATA.len()].copy_from_slice(TEST_DATA);
2397        assert_eq!(&data[..], &expected[..]);
2398
2399        let aligned_len = fs.block_size().align_up(TEST_OBJECT_SIZE).unwrap() as usize;
2400        let mut buf = object.allocate_buffer(aligned_len).await;
2401        buf.fill(123u8);
2402        assert_eq!(
2403            object.read_aligned(0, buf.as_mut()).await.expect("read failed"),
2404            TEST_OBJECT_SIZE as usize
2405        );
2406        let mut expected_aligned = vec![0; TEST_OBJECT_SIZE as usize];
2407        expected_aligned[offset..offset + TEST_DATA.len()].copy_from_slice(TEST_DATA);
2408        assert_eq!(
2409            &buf.as_ptr_slice().subslice(0..TEST_OBJECT_SIZE as usize).to_vec()[..],
2410            &expected_aligned[..]
2411        );
2412        fs.close().await.expect("Close failed");
2413    }
2414
2415    #[fuchsia::test]
2416    async fn test_read_after_writes_interspersed_with_flush() {
2417        let (fs, object) = test_filesystem_and_object().await;
2418
2419        object.owner().flush().await.expect("flush failed");
2420
2421        // Write more test data to the first block fo the file.
2422        let mut buf = object.allocate_buffer(TEST_DATA.len()).await;
2423        buf.copy_from_slice(TEST_DATA);
2424        object.write_or_append(Some(0u64), buf.as_ref()).await.expect("write failed");
2425
2426        let len = TEST_OBJECT_SIZE as usize - 1;
2427        let data = object.read_bytes(0..len as u64).await.expect("read failed");
2428
2429        let mut expected = vec![0u8; len];
2430        let offset = TEST_DATA_OFFSET as usize;
2431        expected[offset..offset + TEST_DATA.len()].copy_from_slice(TEST_DATA);
2432        expected[..TEST_DATA.len()].copy_from_slice(TEST_DATA);
2433        assert_eq!(&data[..], &expected);
2434        fs.close().await.expect("Close failed");
2435    }
2436
2437    #[fuchsia::test]
2438    async fn test_read_after_truncate_and_extend() {
2439        let (fs, object) = test_filesystem_and_object().await;
2440
2441        // Arrange for there to be <extent><deleted-extent><extent>.
2442        let mut buf = object.allocate_buffer(TEST_DATA.len()).await;
2443        buf.copy_from_slice(TEST_DATA);
2444        // This adds an extent at 0..512.
2445        object.write_or_append(Some(0), buf.as_ref()).await.expect("write failed");
2446        // This deletes 512..1024.
2447        object.truncate(3).await.expect("truncate failed");
2448        let data = b"foo";
2449        let offset = 1500u64;
2450        let align = (offset % fs.block_size()) as usize;
2451        let mut buf = object.allocate_buffer(align + data.len()).await;
2452        buf.subslice_mut(align..buf.len()).copy_from_slice(data);
2453        // This adds 1024..1536.
2454        object
2455            .write_or_append(Some(1500), buf.subslice(align..buf.len()))
2456            .await
2457            .expect("write failed");
2458
2459        const LEN1: usize = 1503;
2460        let data1 = object.read_bytes(0..LEN1 as u64).await.expect("read failed");
2461        let mut expected = [0; LEN1];
2462        expected[..3].copy_from_slice(&TEST_DATA[..3]);
2463        expected[1500..].copy_from_slice(b"foo");
2464        assert_eq!(&data1[..], &expected);
2465
2466        // Also test a read that ends midway through the deleted extent.
2467        const LEN2: usize = 601;
2468        let data2 = object.read_bytes(0..LEN2 as u64).await.expect("read failed");
2469        assert_eq!(&data2[..], &expected[..LEN2]);
2470        fs.close().await.expect("Close failed");
2471    }
2472
2473    #[fuchsia::test]
2474    async fn test_read_whole_blocks_with_multiple_objects() {
2475        let (fs, object) = test_filesystem_and_object().await;
2476        let block_size = object.block_size().get() as usize;
2477        let mut buffer = object.allocate_buffer(block_size).await;
2478        buffer.fill(0xaf);
2479        object.write_or_append(Some(0), buffer.as_ref()).await.expect("write failed");
2480
2481        let store = object.owner();
2482        let mut transaction = fs
2483            .root_store()
2484            .new_transaction(lock_keys![], Options::default())
2485            .await
2486            .expect("new_transaction failed");
2487        let object2 =
2488            ObjectStore::create_object(&store, &mut transaction, HandleOptions::default(), None)
2489                .await
2490                .expect("create_object failed");
2491        transaction.commit().await.expect("commit failed");
2492        let mut ef_buffer = object.allocate_buffer(block_size).await;
2493        ef_buffer.fill(0xef);
2494        object2.write_or_append(Some(0), ef_buffer.as_ref()).await.expect("write failed");
2495
2496        let mut buffer = object.allocate_buffer(block_size).await;
2497        buffer.fill(0xaf);
2498        object
2499            .write_or_append(Some(block_size as u64), buffer.as_ref())
2500            .await
2501            .expect("write failed");
2502        object.truncate(3 * block_size as u64).await.expect("truncate failed");
2503        object2
2504            .write_or_append(Some(block_size as u64), ef_buffer.as_ref())
2505            .await
2506            .expect("write failed");
2507
2508        let mut buffer = object.allocate_buffer(4 * block_size).await;
2509        buffer.fill(123);
2510        assert_eq!(
2511            object.read_aligned(0, buffer.as_mut()).await.expect("read failed"),
2512            3 * block_size
2513        );
2514        assert_eq!(
2515            &buffer.as_ptr_slice().subslice(0..2 * block_size).to_vec()[..],
2516            &vec![0xaf; 2 * block_size]
2517        );
2518        assert_eq!(
2519            &buffer.as_ptr_slice().subslice(2 * block_size..3 * block_size).to_vec()[..],
2520            &vec![0; block_size]
2521        );
2522        assert_eq!(
2523            object2.read_aligned(0, buffer.as_mut()).await.expect("read failed"),
2524            2 * block_size
2525        );
2526        assert_eq!(
2527            &buffer.as_ptr_slice().subslice(0..2 * block_size).to_vec()[..],
2528            &vec![0xef; 2 * block_size]
2529        );
2530        fs.close().await.expect("Close failed");
2531    }
2532
2533    #[fuchsia::test]
2534    async fn test_alignment() {
2535        let (fs, object) = test_filesystem_and_object().await;
2536
2537        struct AlignTest {
2538            fill: u8,
2539            object: DataObjectHandle<ObjectStore>,
2540            mirror: Vec<u8>,
2541        }
2542
2543        impl AlignTest {
2544            async fn new(object: DataObjectHandle<ObjectStore>) -> Self {
2545                let mirror =
2546                    object.read_bytes(0..object.get_size()).await.expect("read failed").into_vec();
2547                Self { fill: 0, object, mirror }
2548            }
2549
2550            // Fills |range| of self.object with a byte value (self.fill) and mirrors the same
2551            // operation to an in-memory copy of the object.
2552            // Each subsequent call bumps the value of fill.
2553            // It is expected that the object and its mirror maintain identical content.
2554            async fn test(&mut self, range: Range<u64>) {
2555                let mut buf = self.object.allocate_buffer((range.end - range.start) as usize).await;
2556                self.fill += 1;
2557                buf.fill(self.fill);
2558                self.object
2559                    .write_or_append(Some(range.start), buf.as_ref())
2560                    .await
2561                    .expect("write_or_append failed");
2562                if range.end > self.mirror.len() as u64 {
2563                    self.mirror.resize(range.end as usize, 0);
2564                }
2565                self.mirror[range.start as usize..range.end as usize].fill(self.fill);
2566                let data = self
2567                    .object
2568                    .read_bytes(0..self.mirror.len() as u64 + 1)
2569                    .await
2570                    .expect("read failed");
2571                assert_eq!(&data[..], self.mirror.as_slice());
2572            }
2573        }
2574
2575        let block_size = object.block_size().get();
2576        let mut align = AlignTest::new(object).await;
2577
2578        // Fill the object to start with (with 1).
2579        align.test(0..2 * block_size + 1).await;
2580
2581        // Unaligned head (fills with 2, overwrites that with 3).
2582        align.test(1..block_size).await;
2583        align.test(1..2 * block_size).await;
2584
2585        // Unaligned tail (fills with 4 and 5).
2586        align.test(0..block_size - 1).await;
2587        align.test(0..2 * block_size - 1).await;
2588
2589        // Both unaligned (fills with 6 and 7).
2590        align.test(1..block_size - 1).await;
2591        align.test(1..2 * block_size - 1).await;
2592
2593        fs.close().await.expect("Close failed");
2594    }
2595
2596    #[fuchsia::test]
2597    async fn test_read_bytes_chunked() {
2598        // 16MiB device size.
2599        let device = DeviceHolder::new(FakeDevice::new(32768, TEST_DEVICE_BLOCK_SIZE));
2600        let fs = FxFilesystem::new_empty(device).await.expect("new_empty failed");
2601        let object = create_object_with_key(fs.clone(), Some(&new_insecure_crypt()), false).await;
2602
2603        const FILE_SIZE: usize = 5 * 1024 * 1024;
2604        let mut data = vec![0u8; FILE_SIZE];
2605        for (i, byte) in data.iter_mut().enumerate() {
2606            *byte = (i % 251) as u8;
2607        }
2608
2609        const WRITE_CHUNK_SIZE: usize = 1024 * 1024;
2610        let mut buf = object.allocate_buffer(WRITE_CHUNK_SIZE).await;
2611        for chunk in (0..FILE_SIZE).step_by(WRITE_CHUNK_SIZE) {
2612            let end = std::cmp::min(chunk + WRITE_CHUNK_SIZE, FILE_SIZE);
2613            buf.subslice_mut(..end - chunk).copy_from_slice(&data[chunk..end]);
2614            object
2615                .write_or_append(Some(chunk as u64), buf.subslice(..end - chunk))
2616                .await
2617                .expect("write failed");
2618        }
2619
2620        // Read full file (> 2 MiB, requires multiple 2 MiB chunks).
2621        let read_data = object.read_bytes(0..FILE_SIZE as u64).await.expect("read_bytes failed");
2622        assert_eq!(&read_data[..], &data[..]);
2623
2624        // Read an unaligned range that spans multiple 2 MiB chunks.
2625        let range = 123_456..4_718_592;
2626        let read_data = object.read_bytes(range.clone()).await.expect("read_bytes failed");
2627        assert_eq!(&read_data[..], &data[range.start as usize..range.end as usize]);
2628
2629        // Read an unaligned range that straddles the 2 MiB boundary.
2630        let range = (2 * 1024 * 1024 - 500)..(2 * 1024 * 1024 + 500);
2631        let read_data = object.read_bytes(range.clone()).await.expect("read_bytes failed");
2632        assert_eq!(&read_data[..], &data[range.start as usize..range.end as usize]);
2633
2634        // Read past EOF.
2635        let range = (FILE_SIZE as u64 - 100)..(FILE_SIZE as u64 + 1000);
2636        let read_data = object.read_bytes(range).await.expect("read_bytes failed");
2637        assert_eq!(&read_data[..], &data[(FILE_SIZE - 100)..]);
2638
2639        // Read entirely past EOF.
2640        let range = (FILE_SIZE as u64 + 10)..(FILE_SIZE as u64 + 100);
2641        let read_data = object.read_bytes(range).await.expect("read_bytes failed");
2642        assert!(read_data.is_empty());
2643
2644        // Test contents() method.
2645        assert!(object.contents(FILE_SIZE - 1).await.is_err());
2646        let contents = object.contents(usize::MAX).await.expect("contents failed");
2647        assert_eq!(&contents[..], &data[..]);
2648
2649        // Test read_bytes with empty ranges.
2650        let empty = object.read_bytes(0..0).await.expect("read_bytes failed");
2651        assert!(empty.is_empty());
2652        let empty = object.read_bytes(100..100).await.expect("read_bytes failed");
2653        assert!(empty.is_empty());
2654
2655        fs.close().await.expect("Close failed");
2656    }
2657
2658    async fn test_preallocate_common(fs: &FxFilesystem, object: DataObjectHandle<ObjectStore>) {
2659        let allocator = fs.allocator();
2660        let allocated_before = allocator.get_allocated_bytes();
2661        let mut transaction = object.new_transaction().await.expect("new_transaction failed");
2662        object
2663            .preallocate_range(&mut transaction, &mut (0..fs.block_size().get()))
2664            .await
2665            .expect("preallocate_range failed");
2666        transaction.commit().await.expect("commit failed");
2667        assert!(object.get_size() < 1048576);
2668        let mut transaction = object.new_transaction().await.expect("new_transaction failed");
2669        object
2670            .preallocate_range(&mut transaction, &mut (0..1048576))
2671            .await
2672            .expect("preallocate_range failed");
2673        transaction.commit().await.expect("commit failed");
2674        assert_eq!(object.get_size(), 1048576);
2675        // Check that it didn't reallocate the space for the existing extent
2676        let allocated_after = allocator.get_allocated_bytes();
2677        assert_eq!(allocated_after - allocated_before, 1048576 - fs.block_size());
2678
2679        let mut buf = object
2680            .allocate_buffer(fs.block_size().align_up(TEST_DATA_OFFSET).unwrap() as usize)
2681            .await;
2682        buf.fill(47);
2683        object
2684            .write_or_append(Some(0), buf.subslice(0..TEST_DATA_OFFSET as usize))
2685            .await
2686            .expect("write failed");
2687        buf.fill(95);
2688        let offset = fs.block_size().align_up(TEST_OBJECT_SIZE).unwrap();
2689        object
2690            .overwrite(offset, buf.as_mut(), OverwriteOptions::default())
2691            .await
2692            .expect("write failed");
2693
2694        // Make sure there were no more allocations.
2695        assert_eq!(allocator.get_allocated_bytes(), allocated_after);
2696
2697        // Read back the data and make sure it is what we expect.
2698        let mut buf = object.allocate_buffer(1048576).await;
2699        assert_eq!(object.read_aligned(0, buf.as_mut()).await.expect("read failed"), buf.len());
2700        assert_eq!(
2701            &buf.as_ptr_slice().subslice(0..TEST_DATA_OFFSET as usize).to_vec()[..],
2702            &[47; TEST_DATA_OFFSET as usize]
2703        );
2704        assert_eq!(
2705            &buf.as_ptr_slice()
2706                .subslice(TEST_DATA_OFFSET as usize..TEST_DATA_OFFSET as usize + TEST_DATA.len())
2707                .to_vec()[..],
2708            TEST_DATA
2709        );
2710        assert_eq!(
2711            &buf.as_ptr_slice().subslice(offset as usize..offset as usize + 2048).to_vec()[..],
2712            &[95; 2048]
2713        );
2714    }
2715
2716    #[fuchsia::test]
2717    async fn test_unaligned_overwrite_returns_error() {
2718        let (fs, object) = test_filesystem_and_object().await;
2719        let mut buf = object.allocate_buffer(100).await;
2720        let res = object.overwrite(0, buf.as_mut(), OverwriteOptions::default()).await;
2721        assert!(matches!(res, Err(e) if FxfsError::InvalidArgs.matches(&e)));
2722        fs.close().await.expect("Close failed");
2723    }
2724
2725    #[fuchsia::test]
2726    async fn test_unaligned_preallocate_returns_error() {
2727        let (fs, object) = test_filesystem_and_object().await;
2728        let mut transaction = fs
2729            .root_store()
2730            .new_transaction(lock_keys![], Options::default())
2731            .await
2732            .expect("new failed");
2733        let res = object.preallocate_range(&mut transaction, &mut (0..100)).await;
2734        assert!(matches!(res, Err(e) if FxfsError::InvalidArgs.matches(&e)));
2735        fs.close().await.expect("Close failed");
2736    }
2737
2738    #[fuchsia::test]
2739    async fn test_encrypted_preallocate_returns_error() {
2740        let (fs, object) = test_filesystem_and_object().await;
2741        let mut transaction = fs
2742            .root_store()
2743            .new_transaction(lock_keys![], Options::default())
2744            .await
2745            .expect("new failed");
2746        let bs = fs.block_size();
2747        let res = object.preallocate_range(&mut transaction, &mut (0..bs.get())).await;
2748        assert!(matches!(res, Err(e) if FxfsError::NotSupported.matches(&e)));
2749        fs.close().await.expect("Close failed");
2750    }
2751
2752    #[fuchsia::test]
2753    async fn test_preallocate_range() {
2754        let (fs, object) = test_filesystem_and_object_with_key(None, true).await;
2755        test_preallocate_common(&fs, object).await;
2756        fs.close().await.expect("Close failed");
2757    }
2758
2759    // This is identical to the previous test except that we flush so that extents end up in
2760    // different layers.
2761    #[fuchsia::test]
2762    async fn test_preallocate_succeeds_when_extents_are_in_different_layers() {
2763        let (fs, object) = test_filesystem_and_object_with_key(None, true).await;
2764        object.owner().flush().await.expect("flush failed");
2765        test_preallocate_common(&fs, object).await;
2766        fs.close().await.expect("Close failed");
2767    }
2768
2769    #[fuchsia::test]
2770    async fn test_already_preallocated() {
2771        let (fs, object) = test_filesystem_and_object_with_key(None, true).await;
2772        let allocator = fs.allocator();
2773        let allocated_before = allocator.get_allocated_bytes();
2774        let mut transaction = object.new_transaction().await.expect("new_transaction failed");
2775        let offset = fs.block_size().align_down(TEST_DATA_OFFSET);
2776        object
2777            .preallocate_range(&mut transaction, &mut (offset..offset + fs.block_size()))
2778            .await
2779            .expect("preallocate_range failed");
2780        transaction.commit().await.expect("commit failed");
2781        // Check that it didn't reallocate any new space.
2782        assert_eq!(allocator.get_allocated_bytes(), allocated_before);
2783        fs.close().await.expect("Close failed");
2784    }
2785
2786    #[fuchsia::test]
2787    async fn test_overwrite_when_preallocated_at_start_of_file() {
2788        // The standard test data we put in the test object would cause an extent with checksums
2789        // to be created, which overwrite() doesn't support. So we create an empty object instead.
2790        let (fs, object) = test_filesystem_and_empty_object().await;
2791
2792        let object = ObjectStore::open_object(
2793            object.owner(),
2794            object.object_id(),
2795            HandleOptions::default(),
2796            None,
2797        )
2798        .await
2799        .expect("open_object failed");
2800
2801        assert_eq!(fs.block_size(), 4096);
2802
2803        let mut write_buf = object.allocate_buffer(4096).await;
2804        write_buf.fill(95);
2805
2806        // First try to overwrite without allowing allocations
2807        // We expect this to fail, since nothing is allocated yet
2808        object
2809            .overwrite(0, write_buf.as_mut(), OverwriteOptions::default())
2810            .await
2811            .expect_err("overwrite succeeded");
2812
2813        // Now preallocate some space (exactly one block)
2814        let mut transaction = object.new_transaction().await.expect("new_transaction failed");
2815        object
2816            .preallocate_range(&mut transaction, &mut (0..4096 as u64))
2817            .await
2818            .expect("preallocate_range failed");
2819        transaction.commit().await.expect("commit failed");
2820
2821        // Now try the same overwrite command as before, it should work this time,
2822        // even with allocations disabled...
2823        {
2824            let mut read_buf = object.allocate_buffer(4096).await;
2825            object.read_aligned(0, read_buf.as_mut()).await.expect("read failed");
2826            assert_eq!(&read_buf.to_vec()[..], &[0; 4096]);
2827        }
2828        object
2829            .overwrite(0, write_buf.as_mut(), OverwriteOptions::default())
2830            .await
2831            .expect("overwrite failed");
2832        {
2833            let mut read_buf = object.allocate_buffer(4096).await;
2834            object.read_aligned(0, read_buf.as_mut()).await.expect("read failed");
2835            assert_eq!(&read_buf.to_vec()[..], &[95; 4096]);
2836        }
2837
2838        // Now try to overwrite at offset 4096. We expect this to fail, since we only preallocated
2839        // one block earlier at offset 0
2840        object
2841            .overwrite(4096, write_buf.as_mut(), OverwriteOptions::default())
2842            .await
2843            .expect_err("overwrite succeeded");
2844
2845        // We can't assert anything about the existing bytes, because they haven't been allocated
2846        // yet and they could contain any values
2847        object
2848            .overwrite(
2849                4096,
2850                write_buf.as_mut(),
2851                OverwriteOptions { allow_allocations: true, ..Default::default() },
2852            )
2853            .await
2854            .expect("overwrite failed");
2855        {
2856            let mut read_buf = object.allocate_buffer(4096).await;
2857            object.read_aligned(4096, read_buf.as_mut()).await.expect("read failed");
2858            assert_eq!(&read_buf.to_vec()[..], &[95; 4096]);
2859        }
2860
2861        // Check that the overwrites haven't messed up the filesystem state
2862        let fsck_options = FsckOptions {
2863            fail_on_warning: true,
2864            no_lock: true,
2865            on_error: Box::new(|err| println!("fsck error: {:?}", err)),
2866            ..Default::default()
2867        };
2868        fsck_with_options(fs.clone(), &fsck_options).await.expect("fsck failed");
2869
2870        fs.close().await.expect("Close failed");
2871    }
2872
2873    #[fuchsia::test]
2874    async fn test_overwrite_large_buffer_and_file_with_many_holes() {
2875        // The standard test data we put in the test object would cause an extent with checksums
2876        // to be created, which overwrite() doesn't support. So we create an empty object instead.
2877        let (fs, object) = test_filesystem_and_empty_object().await;
2878
2879        let object = ObjectStore::open_object(
2880            object.owner(),
2881            object.object_id(),
2882            HandleOptions::default(),
2883            None,
2884        )
2885        .await
2886        .expect("open_object failed");
2887
2888        assert_eq!(fs.block_size(), 4096);
2889        assert_eq!(object.get_size(), TEST_OBJECT_SIZE);
2890
2891        // Let's create some non-holes
2892        let mut transaction = object.new_transaction().await.expect("new_transaction failed");
2893        object
2894            .preallocate_range(&mut transaction, &mut (4096..8192 as u64))
2895            .await
2896            .expect("preallocate_range failed");
2897        object
2898            .preallocate_range(&mut transaction, &mut (16384..32768 as u64))
2899            .await
2900            .expect("preallocate_range failed");
2901        object
2902            .preallocate_range(&mut transaction, &mut (65536..131072 as u64))
2903            .await
2904            .expect("preallocate_range failed");
2905        object
2906            .preallocate_range(&mut transaction, &mut (262144..524288 as u64))
2907            .await
2908            .expect("preallocate_range failed");
2909        transaction.commit().await.expect("commit failed");
2910
2911        assert_eq!(object.get_size(), 524288);
2912
2913        let mut write_buf = object.allocate_buffer(4096).await;
2914        write_buf.fill(95);
2915
2916        // We shouldn't be able to overwrite in the holes if new allocations aren't enabled
2917        object
2918            .overwrite(0, write_buf.as_mut(), OverwriteOptions::default())
2919            .await
2920            .expect_err("overwrite succeeded");
2921        object
2922            .overwrite(8192, write_buf.as_mut(), OverwriteOptions::default())
2923            .await
2924            .expect_err("overwrite succeeded");
2925        object
2926            .overwrite(32768, write_buf.as_mut(), OverwriteOptions::default())
2927            .await
2928            .expect_err("overwrite succeeded");
2929        object
2930            .overwrite(131072, write_buf.as_mut(), OverwriteOptions::default())
2931            .await
2932            .expect_err("overwrite succeeded");
2933
2934        // But we should be able to overwrite in the prealloc'd areas without needing allocations
2935        {
2936            let mut read_buf = object.allocate_buffer(4096).await;
2937            object.read_aligned(4096, read_buf.as_mut()).await.expect("read failed");
2938            assert_eq!(&read_buf.to_vec()[..], &[0; 4096]);
2939        }
2940        object
2941            .overwrite(4096, write_buf.as_mut(), OverwriteOptions::default())
2942            .await
2943            .expect("overwrite failed");
2944        {
2945            let mut read_buf = object.allocate_buffer(4096).await;
2946            object.read_aligned(4096, read_buf.as_mut()).await.expect("read failed");
2947            assert_eq!(&read_buf.to_vec()[..], &[95; 4096]);
2948        }
2949        {
2950            let mut read_buf = object.allocate_buffer(4096).await;
2951            object.read_aligned(16384, read_buf.as_mut()).await.expect("read failed");
2952            assert_eq!(&read_buf.to_vec()[..], &[0; 4096]);
2953        }
2954        object
2955            .overwrite(16384, write_buf.as_mut(), OverwriteOptions::default())
2956            .await
2957            .expect("overwrite failed");
2958        {
2959            let mut read_buf = object.allocate_buffer(4096).await;
2960            object.read_aligned(16384, read_buf.as_mut()).await.expect("read failed");
2961            assert_eq!(&read_buf.to_vec()[..], &[95; 4096]);
2962        }
2963        {
2964            let mut read_buf = object.allocate_buffer(4096).await;
2965            object.read_aligned(65536, read_buf.as_mut()).await.expect("read failed");
2966            assert_eq!(&read_buf.to_vec()[..], &[0; 4096]);
2967        }
2968        object
2969            .overwrite(65536, write_buf.as_mut(), OverwriteOptions::default())
2970            .await
2971            .expect("overwrite failed");
2972        {
2973            let mut read_buf = object.allocate_buffer(4096).await;
2974            object.read_aligned(65536, read_buf.as_mut()).await.expect("read failed");
2975            assert_eq!(&read_buf.to_vec()[..], &[95; 4096]);
2976        }
2977        {
2978            let mut read_buf = object.allocate_buffer(4096).await;
2979            object.read_aligned(262144, read_buf.as_mut()).await.expect("read failed");
2980            assert_eq!(&read_buf.to_vec()[..], &[0; 4096]);
2981        }
2982        object
2983            .overwrite(262144, write_buf.as_mut(), OverwriteOptions::default())
2984            .await
2985            .expect("overwrite failed");
2986        {
2987            let mut read_buf = object.allocate_buffer(4096).await;
2988            object.read_aligned(262144, read_buf.as_mut()).await.expect("read failed");
2989            assert_eq!(&read_buf.to_vec()[..], &[95; 4096]);
2990        }
2991
2992        // Now let's try to do a huge overwrite, that spans over many holes and non-holes
2993        let mut huge_write_buf = object.allocate_buffer(524288).await;
2994        huge_write_buf.fill(96);
2995
2996        // With allocations disabled, the big overwrite should fail...
2997        object
2998            .overwrite(0, huge_write_buf.as_mut(), OverwriteOptions::default())
2999            .await
3000            .expect_err("overwrite succeeded");
3001        // ... but it should work when allocations are enabled
3002        object
3003            .overwrite(
3004                0,
3005                huge_write_buf.as_mut(),
3006                OverwriteOptions { allow_allocations: true, ..Default::default() },
3007            )
3008            .await
3009            .expect("overwrite failed");
3010        {
3011            let mut read_buf = object.allocate_buffer(524288).await;
3012            object.read_aligned(0, read_buf.as_mut()).await.expect("read failed");
3013            assert_eq!(&read_buf.to_vec()[..], &[96; 524288]);
3014        }
3015
3016        // Check that the overwrites haven't messed up the filesystem state
3017        let fsck_options = FsckOptions {
3018            fail_on_warning: true,
3019            no_lock: true,
3020            on_error: Box::new(|err| println!("fsck error: {:?}", err)),
3021            ..Default::default()
3022        };
3023        fsck_with_options(fs.clone(), &fsck_options).await.expect("fsck failed");
3024
3025        fs.close().await.expect("Close failed");
3026    }
3027
3028    #[fuchsia::test]
3029    async fn test_overwrite_when_unallocated_at_start_of_file() {
3030        // The standard test data we put in the test object would cause an extent with checksums
3031        // to be created, which overwrite() doesn't support. So we create an empty object instead.
3032        let (fs, object) = test_filesystem_and_empty_object().await;
3033
3034        let object = ObjectStore::open_object(
3035            object.owner(),
3036            object.object_id(),
3037            HandleOptions::default(),
3038            None,
3039        )
3040        .await
3041        .expect("open_object failed");
3042
3043        assert_eq!(fs.block_size(), 4096);
3044
3045        let mut write_buf = object.allocate_buffer(4096).await;
3046        write_buf.fill(95);
3047
3048        // First try to overwrite without allowing allocations
3049        // We expect this to fail, since nothing is allocated yet
3050        object
3051            .overwrite(0, write_buf.as_mut(), OverwriteOptions::default())
3052            .await
3053            .expect_err("overwrite succeeded");
3054
3055        // Now try the same overwrite command as before, but allow allocations
3056        object
3057            .overwrite(
3058                0,
3059                write_buf.as_mut(),
3060                OverwriteOptions { allow_allocations: true, ..Default::default() },
3061            )
3062            .await
3063            .expect("overwrite failed");
3064        {
3065            let mut read_buf = object.allocate_buffer(4096).await;
3066            object.read_aligned(0, read_buf.as_mut()).await.expect("read failed");
3067            assert_eq!(&read_buf.to_vec()[..], &[95; 4096]);
3068        }
3069
3070        // Now try to overwrite at the next block. This should fail if allocations are disabled
3071        object
3072            .overwrite(4096, write_buf.as_mut(), OverwriteOptions::default())
3073            .await
3074            .expect_err("overwrite succeeded");
3075
3076        // ... but it should work if allocations are enabled
3077        object
3078            .overwrite(
3079                4096,
3080                write_buf.as_mut(),
3081                OverwriteOptions { allow_allocations: true, ..Default::default() },
3082            )
3083            .await
3084            .expect("overwrite failed");
3085        {
3086            let mut read_buf = object.allocate_buffer(4096).await;
3087            object.read_aligned(4096, read_buf.as_mut()).await.expect("read failed");
3088            assert_eq!(&read_buf.to_vec()[..], &[95; 4096]);
3089        }
3090
3091        // Check that the overwrites haven't messed up the filesystem state
3092        let fsck_options = FsckOptions {
3093            fail_on_warning: true,
3094            no_lock: true,
3095            on_error: Box::new(|err| println!("fsck error: {:?}", err)),
3096            ..Default::default()
3097        };
3098        fsck_with_options(fs.clone(), &fsck_options).await.expect("fsck failed");
3099
3100        fs.close().await.expect("Close failed");
3101    }
3102
3103    #[fuchsia::test]
3104    async fn test_overwrite_can_extend_a_file() {
3105        // The standard test data we put in the test object would cause an extent with checksums
3106        // to be created, which overwrite() doesn't support. So we create an empty object instead.
3107        let (fs, object) = test_filesystem_and_empty_object().await;
3108
3109        let object = ObjectStore::open_object(
3110            object.owner(),
3111            object.object_id(),
3112            HandleOptions::default(),
3113            None,
3114        )
3115        .await
3116        .expect("open_object failed");
3117
3118        assert_eq!(fs.block_size(), 4096);
3119        assert_eq!(object.get_size(), TEST_OBJECT_SIZE);
3120
3121        let mut write_buf = object.allocate_buffer(4096).await;
3122        write_buf.fill(95);
3123
3124        // Let's try to fill up the last block, and increase the file size in doing so
3125        let last_block_offset = round_down(TEST_OBJECT_SIZE, 4096 as u32);
3126
3127        // Expected to fail with allocations disabled
3128        object
3129            .overwrite(last_block_offset, write_buf.as_mut(), OverwriteOptions::default())
3130            .await
3131            .expect_err("overwrite succeeded");
3132        // ... but expected to succeed with allocations enabled
3133        object
3134            .overwrite(
3135                last_block_offset,
3136                write_buf.as_mut(),
3137                OverwriteOptions { allow_allocations: true, ..Default::default() },
3138            )
3139            .await
3140            .expect("overwrite failed");
3141        {
3142            let mut read_buf = object.allocate_buffer(4096).await;
3143            object.read_aligned(last_block_offset, read_buf.as_mut()).await.expect("read failed");
3144            assert_eq!(&read_buf.to_vec()[..], &[95; 4096]);
3145        }
3146
3147        assert_eq!(object.get_size(), 8192);
3148
3149        // Let's try to write at the next block, too
3150        let next_block_offset = round_up(TEST_OBJECT_SIZE, 4096 as u32).unwrap();
3151
3152        // Expected to fail with allocations disabled
3153        object
3154            .overwrite(next_block_offset, write_buf.as_mut(), OverwriteOptions::default())
3155            .await
3156            .expect_err("overwrite succeeded");
3157        // ... but expected to succeed with allocations enabled
3158        object
3159            .overwrite(
3160                next_block_offset,
3161                write_buf.as_mut(),
3162                OverwriteOptions { allow_allocations: true, ..Default::default() },
3163            )
3164            .await
3165            .expect("overwrite failed");
3166        {
3167            let mut read_buf = object.allocate_buffer(4096).await;
3168            object.read_aligned(next_block_offset, read_buf.as_mut()).await.expect("read failed");
3169            assert_eq!(&read_buf.to_vec()[..], &[95; 4096]);
3170        }
3171
3172        assert_eq!(object.get_size(), 12288);
3173
3174        // Check that the overwrites haven't messed up the filesystem state
3175        let fsck_options = FsckOptions {
3176            fail_on_warning: true,
3177            no_lock: true,
3178            on_error: Box::new(|err| println!("fsck error: {:?}", err)),
3179            ..Default::default()
3180        };
3181        fsck_with_options(fs.clone(), &fsck_options).await.expect("fsck failed");
3182
3183        fs.close().await.expect("Close failed");
3184    }
3185
3186    #[fuchsia::test]
3187    async fn test_enable_verity() {
3188        let fs: OpenFxFilesystem = test_filesystem().await;
3189        let mut transaction = fs
3190            .root_store()
3191            .new_transaction(lock_keys![], Options::default())
3192            .await
3193            .expect("new_transaction failed");
3194        let store = fs.root_store();
3195        let object = Arc::new(
3196            ObjectStore::create_object(&store, &mut transaction, HandleOptions::default(), None)
3197                .await
3198                .expect("create_object failed"),
3199        );
3200
3201        transaction.commit().await.unwrap();
3202
3203        object
3204            .enable_verity(fio::VerificationOptions {
3205                hash_algorithm: Some(fio::HashAlgorithm::Sha256),
3206                salt: Some(vec![]),
3207                ..Default::default()
3208            })
3209            .await
3210            .expect("set verified file metadata failed");
3211
3212        let handle =
3213            ObjectStore::open_object(&store, object.object_id(), HandleOptions::default(), None)
3214                .await
3215                .expect("open_object failed");
3216
3217        assert!(handle.is_verified_file());
3218
3219        fs.close().await.expect("Close failed");
3220    }
3221
3222    #[fuchsia::test]
3223    async fn test_enable_verity_large_file() {
3224        // Need to make a large FakeDevice to create space for a 67 MB file.
3225        let device = DeviceHolder::new(FakeDevice::new(262144, TEST_DEVICE_BLOCK_SIZE));
3226        let fs = FxFilesystem::new_empty(device).await.expect("new_empty failed");
3227        let root_store = fs.root_store();
3228        let mut transaction = fs
3229            .root_store()
3230            .new_transaction(lock_keys![], Options::default())
3231            .await
3232            .expect("new_transaction failed");
3233
3234        let handle = ObjectStore::create_object(
3235            &root_store,
3236            &mut transaction,
3237            HandleOptions::default(),
3238            None,
3239        )
3240        .await
3241        .expect("failed to create object");
3242        transaction.commit().await.expect("commit failed");
3243        let mut offset = 0;
3244
3245        // Write a file big enough to trigger multiple transactions on enable_verity().
3246        let mut buf = handle.allocate_buffer(WRITE_ATTR_BATCH_SIZE).await;
3247        buf.fill(1);
3248        for _ in 0..130 {
3249            handle.write_or_append(Some(offset), buf.as_ref()).await.expect("write failed");
3250            offset += WRITE_ATTR_BATCH_SIZE as u64;
3251        }
3252
3253        handle
3254            .enable_verity(fio::VerificationOptions {
3255                hash_algorithm: Some(fio::HashAlgorithm::Sha256),
3256                salt: Some(vec![]),
3257                ..Default::default()
3258            })
3259            .await
3260            .expect("set verified file metadata failed");
3261
3262        let mut buf = handle.allocate_buffer(WRITE_ATTR_BATCH_SIZE).await;
3263        offset = 0;
3264        for _ in 0..130 {
3265            handle
3266                .read_aligned(offset, buf.as_mut())
3267                .await
3268                .expect("verification during read should fail");
3269            assert_eq!(buf.to_vec(), &[1; WRITE_ATTR_BATCH_SIZE]);
3270            offset += WRITE_ATTR_BATCH_SIZE as u64;
3271        }
3272
3273        fsck(fs.clone()).await.expect("fsck failed");
3274        fs.close().await.expect("Close failed");
3275    }
3276
3277    #[fuchsia::test]
3278    async fn test_retry_enable_verity_on_reboot() {
3279        let device = DeviceHolder::new(FakeDevice::new(8192, TEST_DEVICE_BLOCK_SIZE));
3280        let fs = FxFilesystem::new_empty(device).await.expect("new_empty failed");
3281        let root_store = fs.root_store();
3282        let mut transaction = fs
3283            .root_store()
3284            .new_transaction(lock_keys![], Options::default())
3285            .await
3286            .expect("new_transaction failed");
3287
3288        let handle = ObjectStore::create_object(
3289            &root_store,
3290            &mut transaction,
3291            HandleOptions::default(),
3292            None,
3293        )
3294        .await
3295        .expect("failed to create object");
3296        transaction.commit().await.expect("commit failed");
3297
3298        let object_id = {
3299            let mut transaction = handle.new_transaction().await.expect("new_transaction failed");
3300            transaction.add(
3301                root_store.store_object_id(),
3302                Mutation::replace_or_insert_object(
3303                    ObjectKey::graveyard_attribute_entry(
3304                        root_store.graveyard_directory_object_id(),
3305                        handle.object_id(),
3306                        AttributeId::FSVERITY_MERKLE,
3307                    ),
3308                    ObjectValue::Some,
3309                ),
3310            );
3311
3312            // This write should span three transactions. This test mimics the behavior when the
3313            // last transaction gets interrupted by a filesystem.close().
3314            handle
3315                .write_new_attr_in_batches(
3316                    &mut transaction,
3317                    AttributeId::FSVERITY_MERKLE,
3318                    &vec![0; 2 * WRITE_ATTR_BATCH_SIZE],
3319                    WRITE_ATTR_BATCH_SIZE,
3320                )
3321                .await
3322                .expect("failed to write merkle attribute");
3323
3324            handle.object_id()
3325            // Drop the transaction to simulate interrupting the merkle tree creation as well as to
3326            // release the transaction locks.
3327        };
3328
3329        fs.close().await.expect("failed to close filesystem");
3330        let device = fs.take_device().await;
3331        device.reopen(false);
3332
3333        let fs =
3334            FxFilesystemBuilder::new().read_only(true).open(device).await.expect("open failed");
3335        fsck(fs.clone()).await.expect("fsck failed");
3336        fs.close().await.expect("failed to close filesystem");
3337        let device = fs.take_device().await;
3338        device.reopen(false);
3339
3340        // On open, the filesystem will call initial_reap which will call queue_tombstone().
3341        let fs = FxFilesystem::open(device).await.expect("open failed");
3342        let root_store = fs.root_store();
3343        let handle =
3344            ObjectStore::open_object(&root_store, object_id, HandleOptions::default(), None)
3345                .await
3346                .expect("open_object failed");
3347        handle
3348            .enable_verity(fio::VerificationOptions {
3349                hash_algorithm: Some(fio::HashAlgorithm::Sha256),
3350                salt: Some(vec![]),
3351                ..Default::default()
3352            })
3353            .await
3354            .expect("set verified file metadata failed");
3355
3356        // `flush` will ensure that initial reap fully processes all the graveyard entries. This
3357        // isn't strictly necessary for the test to pass (the graveyard marker was already
3358        // processed during `enable_verity`), but it does help catch bugs, such as the attribute
3359        // graveyard entry not being removed upon processing.
3360        fs.graveyard().flush().await;
3361        let merkle_data = handle
3362            .read_attr(AttributeId::FSVERITY_MERKLE)
3363            .await
3364            .expect("read_attr failed")
3365            .expect("No attr found");
3366        assert!(
3367            FsVerityDescriptor::new(&merkle_data[..], handle.block_size().get() as usize).is_ok()
3368        );
3369        fsck(fs.clone()).await.expect("fsck failed");
3370        fs.close().await.expect("Close failed");
3371    }
3372
3373    #[fuchsia::test]
3374    async fn test_verify_data_corrupt_file() {
3375        let fs: OpenFxFilesystem = test_filesystem().await;
3376        let mut transaction = fs
3377            .root_store()
3378            .new_transaction(lock_keys![], Options::default())
3379            .await
3380            .expect("new_transaction failed");
3381        let store = fs.root_store();
3382        let object = Arc::new(
3383            ObjectStore::create_object(&store, &mut transaction, HandleOptions::default(), None)
3384                .await
3385                .expect("create_object failed"),
3386        );
3387
3388        transaction.commit().await.unwrap();
3389
3390        let mut buf = object.allocate_buffer(5 * fs.block_size().get() as usize).await;
3391        buf.fill(123);
3392        object.write_or_append(Some(0), buf.as_ref()).await.expect("write failed");
3393
3394        object
3395            .enable_verity(fio::VerificationOptions {
3396                hash_algorithm: Some(fio::HashAlgorithm::Sha256),
3397                salt: Some(vec![]),
3398                ..Default::default()
3399            })
3400            .await
3401            .expect("set verified file metadata failed");
3402
3403        // Change file contents and ensure verification fails
3404        buf.fill(234);
3405        object.write_or_append(Some(0), buf.as_ref()).await.expect("write failed");
3406        object
3407            .read_aligned(0, buf.as_mut())
3408            .await
3409            .expect_err("verification during read should fail");
3410        object
3411            .read_bytes(0..buf.len() as u64)
3412            .await
3413            .expect_err("verification during read_bytes should fail");
3414
3415        fs.close().await.expect("Close failed");
3416    }
3417
3418    // TODO(https://fxbug.dev/450398331): More tests to be added when this can support writing the
3419    // f2fs format natively. For now, relying on tests inside of the f2fs_reader to exercise more
3420    // paths.
3421    #[fuchsia::test]
3422    async fn test_parse_f2fs_verity() {
3423        let fs: OpenFxFilesystem = test_filesystem().await;
3424        let mut transaction = fs
3425            .root_store()
3426            .new_transaction(lock_keys![], Options::default())
3427            .await
3428            .expect("new_transaction failed");
3429        let store = fs.root_store();
3430        let object = Arc::new(
3431            ObjectStore::create_object(&store, &mut transaction, HandleOptions::default(), None)
3432                .await
3433                .expect("create_object failed"),
3434        );
3435
3436        transaction.commit().await.unwrap();
3437        let file_size = fs.block_size() * 2;
3438        // Write over one block to make there be leaf hashes.
3439        {
3440            let mut buf = object.allocate_buffer(file_size as usize).await;
3441            buf.fill(64);
3442            assert_eq!(
3443                object.write_or_append(None, buf.as_ref()).await.expect("Writing to file."),
3444                file_size
3445            );
3446        }
3447
3448        // Enable verity normally, then shift the type.
3449        object
3450            .enable_verity(fio::VerificationOptions {
3451                hash_algorithm: Some(fio::HashAlgorithm::Sha256),
3452                salt: Some(vec![]),
3453                ..Default::default()
3454            })
3455            .await
3456            .expect("set verified file metadata failed");
3457        let (verity_info, root_hash) = object.get_descriptor().unwrap();
3458
3459        let mut transaction = fs
3460            .root_store()
3461            .new_transaction(
3462                lock_keys![LockKey::Object {
3463                    store_object_id: store.store_object_id(),
3464                    object_id: object.object_id()
3465                }],
3466                Options::default(),
3467            )
3468            .await
3469            .expect("new_transaction failed");
3470        transaction.add(
3471            store.store_object_id(),
3472            Mutation::replace_or_insert_object(
3473                ObjectKey::attribute(
3474                    object.object_id(),
3475                    AttributeId::DATA,
3476                    AttributeKey::Attribute,
3477                ),
3478                ObjectValue::verified_attribute(
3479                    file_size,
3480                    FsverityMetadata::F2fs(0..(fs.block_size() * 2)),
3481                ),
3482            ),
3483        );
3484        transaction.add(
3485            store.store_object_id(),
3486            Mutation::replace_or_insert_object(
3487                ObjectKey::attribute(
3488                    object.object_id(),
3489                    AttributeId::FSVERITY_MERKLE,
3490                    AttributeKey::Attribute,
3491                ),
3492                ObjectValue::attribute(fs.block_size() * 2, false),
3493            ),
3494        );
3495        {
3496            let descriptor = FsVerityDescriptorRaw::new(
3497                fio::HashAlgorithm::Sha256,
3498                fs.block_size().get(),
3499                file_size,
3500                root_hash.as_slice(),
3501                match &verity_info.salt {
3502                    Some(salt) => salt.as_slice(),
3503                    None => [0u8; 0].as_slice(),
3504                },
3505            )
3506            .expect("Creating descriptor");
3507            let mut buf = object.allocate_buffer(fs.block_size().get() as usize).await;
3508            let mut temp = vec![0u8; fs.block_size().get() as usize];
3509            descriptor.write_to_slice(&mut temp).expect("Writing descriptor to buf");
3510            buf.copy_from_slice(&temp);
3511            object
3512                .multi_write(
3513                    &mut transaction,
3514                    AttributeId::FSVERITY_MERKLE,
3515                    &[fs.block_size().get()..(fs.block_size() * 2)],
3516                    buf.as_mut(),
3517                )
3518                .await
3519                .expect("Writing descriptor");
3520        }
3521        transaction.commit().await.unwrap();
3522
3523        let handle =
3524            ObjectStore::open_object(&store, object.object_id(), HandleOptions::default(), None)
3525                .await
3526                .expect("open_object failed");
3527
3528        assert!(handle.is_verified_file());
3529
3530        let mut buf = object.allocate_buffer(file_size as usize).await;
3531        assert_eq!(
3532            handle.read_aligned(0, buf.as_mut()).await.expect("Read whole file."),
3533            file_size as usize
3534        );
3535
3536        fs.close().await.expect("Close failed");
3537    }
3538
3539    #[fuchsia::test]
3540    async fn test_verify_data_corrupt_tree() {
3541        let fs: OpenFxFilesystem = test_filesystem().await;
3542        let object_id = {
3543            let store = fs.root_store();
3544            let mut transaction = fs
3545                .root_store()
3546                .new_transaction(lock_keys![], Options::default())
3547                .await
3548                .expect("new_transaction failed");
3549            let object = Arc::new(
3550                ObjectStore::create_object(
3551                    &store,
3552                    &mut transaction,
3553                    HandleOptions::default(),
3554                    None,
3555                )
3556                .await
3557                .expect("create_object failed"),
3558            );
3559            let object_id = object.object_id();
3560
3561            transaction.commit().await.unwrap();
3562
3563            let mut buf = object.allocate_buffer(5 * fs.block_size().get() as usize).await;
3564            buf.fill(123);
3565            object.write_or_append(Some(0), buf.as_ref()).await.expect("write failed");
3566
3567            object
3568                .enable_verity(fio::VerificationOptions {
3569                    hash_algorithm: Some(fio::HashAlgorithm::Sha256),
3570                    salt: Some(vec![]),
3571                    ..Default::default()
3572                })
3573                .await
3574                .expect("set verified file metadata failed");
3575            object.read_aligned(0, buf.as_mut()).await.expect("verified read");
3576
3577            // Corrupt the merkle tree before closing.
3578            let mut merkle = object
3579                .read_attr(AttributeId::FSVERITY_MERKLE)
3580                .await
3581                .unwrap()
3582                .expect("Reading merkle tree");
3583            merkle[0] = merkle[0].wrapping_add(1);
3584            object
3585                .write_attr(AttributeId::FSVERITY_MERKLE, &*merkle)
3586                .await
3587                .expect("Overwriting merkle");
3588
3589            object_id
3590        }; // Close object.
3591
3592        // Reopening the object should complain about the corrupted merkle tree.
3593        assert!(
3594            ObjectStore::open_object(&fs.root_store(), object_id, HandleOptions::default(), None)
3595                .await
3596                .is_err()
3597        );
3598        fs.close().await.expect("Close failed");
3599    }
3600
3601    #[fuchsia::test]
3602    async fn test_allocate_verity_file() {
3603        let fs: OpenFxFilesystem = test_filesystem().await;
3604        let mut transaction = fs
3605            .root_store()
3606            .new_transaction(lock_keys![], Options::default())
3607            .await
3608            .expect("new_transaction failed");
3609        let store = fs.root_store();
3610        let object = Arc::new(
3611            ObjectStore::create_object(&store, &mut transaction, HandleOptions::default(), None)
3612                .await
3613                .expect("create_object failed"),
3614        );
3615        transaction.commit().await.unwrap();
3616
3617        let mut buf = object.allocate_buffer(8192).await;
3618        buf.fill(0xAA);
3619        object.write_or_append(Some(0), buf.as_ref()).await.expect("write failed");
3620
3621        object
3622            .enable_verity(fio::VerificationOptions {
3623                hash_algorithm: Some(fio::HashAlgorithm::Sha256),
3624                salt: Some(vec![]),
3625                ..Default::default()
3626            })
3627            .await
3628            .expect("enable_verity failed");
3629
3630        assert!(object.is_verified_file());
3631
3632        // Calling allocate on a verity-enabled file should return an error.
3633        assert!(object.allocate(0..8192).await.is_err());
3634
3635        // Even after opening the object again (simulating cache eviction), it must remain verified.
3636        let reopened =
3637            ObjectStore::open_object(&store, object.object_id(), HandleOptions::default(), None)
3638                .await
3639                .expect("open_object failed");
3640        assert!(reopened.is_verified_file());
3641
3642        fs.close().await.expect("Close failed");
3643    }
3644
3645    #[fuchsia::test]
3646    async fn test_extend() {
3647        let fs = test_filesystem().await;
3648        let handle;
3649        let mut transaction = fs
3650            .root_store()
3651            .new_transaction(lock_keys![], Options::default())
3652            .await
3653            .expect("new_transaction failed");
3654        let store = fs.root_store();
3655        handle =
3656            ObjectStore::create_object(&store, &mut transaction, HandleOptions::default(), None)
3657                .await
3658                .expect("create_object failed");
3659
3660        // As of writing, an empty filesystem has two 512kiB superblock extents and a little over
3661        // 256kiB of additional allocations (journal, etc) so we start use a 'magic' starting point
3662        // of 2MiB here.
3663        const START_OFFSET: u64 = 2048 * 1024;
3664        handle
3665            .extend(&mut transaction, START_OFFSET..START_OFFSET + 5 * fs.block_size())
3666            .await
3667            .expect("extend failed");
3668        transaction.commit().await.expect("commit failed");
3669        let mut buf = handle.allocate_buffer(5 * fs.block_size().get() as usize).await;
3670        buf.fill(123);
3671        handle.write_or_append(Some(0), buf.as_ref()).await.expect("write failed");
3672        buf.fill(67);
3673        handle.read_aligned(0, buf.as_mut()).await.expect("read failed");
3674        assert_eq!(buf.to_vec(), vec![123; 5 * fs.block_size().get() as usize]);
3675        fs.close().await.expect("Close failed");
3676    }
3677
3678    #[fuchsia::test]
3679    async fn test_truncate_deallocates_old_extents() {
3680        let (fs, object) = test_filesystem_and_object().await;
3681        let mut buf = object.allocate_buffer(5 * fs.block_size().get() as usize).await;
3682        buf.fill(0xaa);
3683        object.write_or_append(Some(0), buf.as_ref()).await.expect("write failed");
3684
3685        let allocator = fs.allocator();
3686        let allocated_before = allocator.get_allocated_bytes();
3687        object.truncate(fs.block_size().get()).await.expect("truncate failed");
3688        let allocated_after = allocator.get_allocated_bytes();
3689        assert!(
3690            allocated_after < allocated_before,
3691            "before = {} after = {}",
3692            allocated_before,
3693            allocated_after
3694        );
3695        fs.close().await.expect("Close failed");
3696    }
3697
3698    #[fuchsia::test]
3699    async fn test_truncate_zeroes_tail_block() {
3700        let (fs, object) = test_filesystem_and_object().await;
3701
3702        WriteObjectHandle::truncate(&object, TEST_DATA_OFFSET + 3).await.expect("truncate failed");
3703        WriteObjectHandle::truncate(&object, TEST_DATA_OFFSET + TEST_DATA.len() as u64)
3704            .await
3705            .expect("truncate failed");
3706
3707        let mut buf = object.allocate_buffer(fs.block_size().get() as usize).await;
3708        let offset = (TEST_DATA_OFFSET % fs.block_size()) as usize;
3709        object
3710            .read_aligned(TEST_DATA_OFFSET - offset as u64, buf.as_mut())
3711            .await
3712            .expect("read failed");
3713
3714        let mut expected = TEST_DATA.to_vec();
3715        expected[3..].fill(0);
3716        assert_eq!(
3717            &buf.as_ptr_slice().subslice(offset..offset + expected.len()).to_vec()[..],
3718            &expected
3719        );
3720    }
3721
3722    #[fuchsia::test]
3723    async fn test_trim() {
3724        // Format a new filesystem.
3725        let device = DeviceHolder::new(FakeDevice::new(8192, TEST_DEVICE_BLOCK_SIZE));
3726        let fs = FxFilesystem::new_empty(device).await.expect("new_empty failed");
3727        let block_size = fs.block_size();
3728        root_volume(fs.clone())
3729            .await
3730            .expect("root_volume failed")
3731            .new_volume("test", NewChildStoreOptions::default())
3732            .await
3733            .expect("volume failed");
3734        fs.close().await.expect("close failed");
3735        let device = fs.take_device().await;
3736        device.reopen(false);
3737
3738        // To test trim, we open the filesystem and set up a post commit hook that runs after every
3739        // transaction.  When the hook triggers, we can fsck the volume, take a snapshot of the
3740        // device and check that it gets replayed correctly on the snapshot.  We can check that the
3741        // graveyard trims the file as expected.
3742        #[derive(Default)]
3743        struct Context {
3744            store: Option<Arc<ObjectStore>>,
3745            object_id: Option<u64>,
3746        }
3747        let shared_context = Arc::new(Mutex::new(Context::default()));
3748
3749        let object_size = (TRANSACTION_MUTATION_THRESHOLD as u64 + 10) * 2 * block_size;
3750
3751        // Wait for an object to get tombstoned by the graveyard.
3752        async fn expect_tombstoned(store: &Arc<ObjectStore>, object_id: u64) {
3753            loop {
3754                if let Err(e) =
3755                    ObjectStore::open_object(store, object_id, HandleOptions::default(), None).await
3756                {
3757                    assert!(
3758                        FxfsError::NotFound.matches(&e),
3759                        "open_object didn't fail with NotFound: {:?}",
3760                        e
3761                    );
3762                    break;
3763                }
3764                // The graveyard should eventually tombstone the object.
3765                fasync::Timer::new(std::time::Duration::from_millis(100)).await;
3766            }
3767        }
3768
3769        // Checks to see if the object needs to be trimmed.
3770        async fn needs_trim(store: &Arc<ObjectStore>) -> Option<DataObjectHandle<ObjectStore>> {
3771            let root_directory = Directory::open(store, store.root_directory_object_id())
3772                .await
3773                .expect("open failed");
3774            let oid = root_directory.lookup("foo").await.expect("lookup failed");
3775            if let Some((oid, _, _)) = oid {
3776                let object = ObjectStore::open_object(store, oid, HandleOptions::default(), None)
3777                    .await
3778                    .expect("open_object failed");
3779                let props = object.get_properties().await.expect("get_properties failed");
3780                if props.allocated_size > 0 && props.data_attribute_size == 0 {
3781                    Some(object)
3782                } else {
3783                    None
3784                }
3785            } else {
3786                None
3787            }
3788        }
3789
3790        let shared_context_clone = shared_context.clone();
3791        let post_commit = move || {
3792            let store = shared_context_clone.lock().store.as_ref().cloned().unwrap();
3793            let shared_context = shared_context_clone.clone();
3794            async move {
3795                // First run fsck on the current filesystem.
3796                let options = FsckOptions {
3797                    fail_on_warning: true,
3798                    no_lock: true,
3799                    on_error: Box::new(|err| println!("fsck error: {:?}", err)),
3800                    ..Default::default()
3801                };
3802                let fs = store.filesystem();
3803
3804                fsck_with_options(fs.clone(), &options).await.expect("fsck_with_options failed");
3805                fsck_volume_with_options(fs.as_ref(), &options, store.store_object_id(), None)
3806                    .await
3807                    .expect("fsck_volume_with_options failed");
3808
3809                // Now check that we can replay this correctly.
3810                fs.sync(SyncOptions { flush_device: true, ..Default::default() })
3811                    .await
3812                    .expect("sync failed");
3813                let device = fs.device().snapshot().expect("snapshot failed");
3814
3815                let object_id = shared_context.lock().object_id.clone();
3816
3817                let fs2 = FxFilesystemBuilder::new()
3818                    .skip_initial_reap(object_id.is_none())
3819                    .open(device)
3820                    .await
3821                    .expect("open failed");
3822
3823                // If the "foo" file exists check that allocated size matches content size.
3824                let root_vol = root_volume(fs2.clone()).await.expect("root_volume failed");
3825                let store =
3826                    root_vol.volume("test", StoreOptions::default()).await.expect("volume failed");
3827
3828                if let Some(oid) = object_id {
3829                    // For the second pass, the object should get tombstoned.
3830                    expect_tombstoned(&store, oid).await;
3831                } else if let Some(object) = needs_trim(&store).await {
3832                    // Extend the file and make sure that it is correctly trimmed.
3833                    object.truncate(object_size).await.expect("truncate failed");
3834                    let mut buf = object.allocate_buffer(block_size.get() as usize).await;
3835                    object
3836                        .read_aligned(object_size - block_size * 2, buf.as_mut())
3837                        .await
3838                        .expect("read failed");
3839                    assert_eq!(buf.to_vec(), vec![0; block_size.get() as usize]);
3840
3841                    // Remount, this time with the graveyard performing an initial reap and the
3842                    // object should get trimmed.
3843                    let fs = FxFilesystem::open(fs.device().snapshot().expect("snapshot failed"))
3844                        .await
3845                        .expect("open failed");
3846                    let root_vol = root_volume(fs.clone()).await.expect("root_volume failed");
3847                    let store = root_vol
3848                        .volume("test", StoreOptions::default())
3849                        .await
3850                        .expect("volume failed");
3851                    while needs_trim(&store).await.is_some() {
3852                        // The object has been truncated, but still has some data allocated to
3853                        // it.  The graveyard should trim the object eventually.
3854                        fasync::Timer::new(std::time::Duration::from_millis(100)).await;
3855                    }
3856
3857                    // Run fsck.
3858                    fsck_with_options(fs.clone(), &options)
3859                        .await
3860                        .expect("fsck_with_options failed");
3861                    fsck_volume_with_options(fs.as_ref(), &options, store.store_object_id(), None)
3862                        .await
3863                        .expect("fsck_volume_with_options failed");
3864                    fs.close().await.expect("close failed");
3865                }
3866
3867                // Run fsck on fs2.
3868                fsck_with_options(fs2.clone(), &options).await.expect("fsck_with_options failed");
3869                fsck_volume_with_options(fs2.as_ref(), &options, store.store_object_id(), None)
3870                    .await
3871                    .expect("fsck_volume_with_options failed");
3872                fs2.close().await.expect("close failed");
3873            }
3874            .boxed()
3875        };
3876
3877        let fs = FxFilesystemBuilder::new()
3878            .post_commit_hook(post_commit)
3879            .open(device)
3880            .await
3881            .expect("open failed");
3882
3883        let root_vol = root_volume(fs.clone()).await.expect("root_volume failed");
3884        let store = root_vol.volume("test", StoreOptions::default()).await.expect("volume failed");
3885
3886        shared_context.lock().store = Some(store.clone());
3887
3888        let root_directory =
3889            Directory::open(&store, store.root_directory_object_id()).await.expect("open failed");
3890
3891        let object;
3892        let mut transaction = fs
3893            .root_store()
3894            .new_transaction(
3895                lock_keys![LockKey::object(
3896                    store.store_object_id(),
3897                    store.root_directory_object_id()
3898                )],
3899                Options::default(),
3900            )
3901            .await
3902            .expect("new_transaction failed");
3903        object = root_directory
3904            .create_child_file(&mut transaction, "foo")
3905            .await
3906            .expect("create_object failed");
3907        transaction.commit().await.expect("commit failed");
3908
3909        let mut transaction = fs
3910            .root_store()
3911            .new_transaction(
3912                lock_keys![LockKey::object(store.store_object_id(), object.object_id())],
3913                Options::default(),
3914            )
3915            .await
3916            .expect("new_transaction failed");
3917
3918        // Two passes: first with a regular object, and then with that object moved into the
3919        // graveyard.
3920        let mut pass = 0;
3921        loop {
3922            // Create enough extents in it such that when we truncate the object it will require
3923            // more than one transaction.
3924            let mut buf = object.allocate_buffer(5).await;
3925            buf.fill(1);
3926            // Write every other block.
3927            for offset in (0..object_size).into_iter().step_by((2 * block_size) as usize) {
3928                object
3929                    .txn_write(&mut transaction, offset, buf.as_ref())
3930                    .await
3931                    .expect("write failed");
3932            }
3933            transaction.commit().await.expect("commit failed");
3934            // This should take up more than one transaction.
3935            WriteObjectHandle::truncate(&object, 0).await.expect("truncate failed");
3936
3937            if pass == 1 {
3938                break;
3939            }
3940
3941            // Store the object ID so that we can make sure the object is always tombstoned
3942            // after remount (see above).
3943            shared_context.lock().object_id = Some(object.object_id());
3944
3945            transaction = fs
3946                .root_store()
3947                .new_transaction(
3948                    lock_keys![
3949                        LockKey::object(store.store_object_id(), store.root_directory_object_id()),
3950                        LockKey::object(store.store_object_id(), object.object_id()),
3951                    ],
3952                    Options::default(),
3953                )
3954                .await
3955                .expect("new_transaction failed");
3956
3957            // Move the object into the graveyard.
3958            replace_child(&mut transaction, None, (&root_directory, "foo"))
3959                .await
3960                .expect("replace_child failed");
3961            store.add_to_graveyard(&mut transaction, object.object_id());
3962
3963            pass += 1;
3964        }
3965
3966        fs.close().await.expect("Close failed");
3967    }
3968
3969    #[fuchsia::test]
3970    async fn test_adjust_refs() {
3971        let (fs, object) = test_filesystem_and_object().await;
3972        let store = object.owner();
3973        let mut transaction = fs
3974            .root_store()
3975            .new_transaction(
3976                lock_keys![LockKey::object(store.store_object_id(), object.object_id())],
3977                Options::default(),
3978            )
3979            .await
3980            .expect("new_transaction failed");
3981        assert_eq!(
3982            store
3983                .adjust_refs(&mut transaction, object.object_id(), 1)
3984                .await
3985                .expect("adjust_refs failed"),
3986            false
3987        );
3988        transaction.commit().await.expect("commit failed");
3989
3990        let allocator = fs.allocator();
3991        let allocated_before = allocator.get_allocated_bytes();
3992        let mut transaction = fs
3993            .root_store()
3994            .new_transaction(
3995                lock_keys![LockKey::object(store.store_object_id(), object.object_id())],
3996                Options::default(),
3997            )
3998            .await
3999            .expect("new_transaction failed");
4000        assert_eq!(
4001            store
4002                .adjust_refs(&mut transaction, object.object_id(), -2)
4003                .await
4004                .expect("adjust_refs failed"),
4005            true
4006        );
4007        transaction.commit().await.expect("commit failed");
4008
4009        assert_eq!(allocator.get_allocated_bytes(), allocated_before);
4010
4011        store
4012            .tombstone_object(
4013                object.object_id(),
4014                Options { reservation: ReservationOptions::BorrowedMetadata, ..Default::default() },
4015                None,
4016            )
4017            .await
4018            .expect("purge failed");
4019
4020        assert_eq!(allocated_before - allocator.get_allocated_bytes(), fs.block_size());
4021
4022        // We need to remove the directory entry, too, otherwise fsck will complain
4023        {
4024            let mut transaction = fs
4025                .root_store()
4026                .new_transaction(
4027                    lock_keys![LockKey::object(
4028                        store.store_object_id(),
4029                        store.root_directory_object_id()
4030                    )],
4031                    Options::default(),
4032                )
4033                .await
4034                .expect("new_transaction failed");
4035            let root_directory = Directory::open(&store, store.root_directory_object_id())
4036                .await
4037                .expect("open failed");
4038            transaction.add(
4039                store.store_object_id(),
4040                Mutation::replace_or_insert_object(
4041                    ObjectKey::child(root_directory.object_id(), TEST_OBJECT_NAME, DirType::Normal),
4042                    ObjectValue::None,
4043                ),
4044            );
4045            transaction.commit().await.expect("commit failed");
4046        }
4047
4048        fsck_with_options(
4049            fs.clone(),
4050            &FsckOptions {
4051                fail_on_warning: true,
4052                on_error: Box::new(|err| println!("fsck error: {:?}", err)),
4053                ..Default::default()
4054            },
4055        )
4056        .await
4057        .expect("fsck_with_options failed");
4058
4059        fs.close().await.expect("Close failed");
4060    }
4061
4062    #[fuchsia::test]
4063    async fn test_locks() {
4064        let (fs, object) = test_filesystem_and_object().await;
4065        let (send1, recv1) = channel();
4066        let (send2, recv2) = channel();
4067        let (send3, recv3) = channel();
4068        let done = Mutex::new(false);
4069        let mut futures = FuturesUnordered::new();
4070        futures.push(
4071            async {
4072                let mut t = object.new_transaction().await.expect("new_transaction failed");
4073                send1.send(()).unwrap(); // Tell the next future to continue.
4074                send3.send(()).unwrap(); // Tell the last future to continue.
4075                recv2.await.unwrap();
4076                let mut buf = object.allocate_buffer(5).await;
4077                buf.copy_from_slice(b"hello");
4078                object.txn_write(&mut t, 0, buf.as_ref()).await.expect("write failed");
4079                // This is a halting problem so all we can do is sleep.
4080                fasync::Timer::new(Duration::from_millis(100)).await;
4081                assert!(!*done.lock());
4082                t.commit().await.expect("commit failed");
4083            }
4084            .boxed(),
4085        );
4086        futures.push(
4087            async {
4088                recv1.await.unwrap();
4089                // Reads should not block.
4090                let data = object
4091                    .read_bytes(TEST_DATA_OFFSET..(TEST_DATA_OFFSET + TEST_DATA.len() as u64))
4092                    .await
4093                    .expect("read failed");
4094                assert_eq!(&*data, TEST_DATA);
4095                // Tell the first future to continue.
4096                send2.send(()).unwrap();
4097            }
4098            .boxed(),
4099        );
4100        futures.push(
4101            async {
4102                // This should block until the first future has completed.
4103                recv3.await.unwrap();
4104                let _t = object.new_transaction().await.expect("new_transaction failed");
4105                let data = object.read_bytes(0..5).await.expect("read failed");
4106                assert_eq!(&*data, b"hello");
4107            }
4108            .boxed(),
4109        );
4110        while let Some(()) = futures.next().await {}
4111        fs.close().await.expect("Close failed");
4112    }
4113
4114    #[fuchsia::test(threads = 10)]
4115    async fn test_racy_reads() {
4116        let fs = test_filesystem().await;
4117        let object;
4118        let mut transaction = fs
4119            .root_store()
4120            .new_transaction(lock_keys![], Options::default())
4121            .await
4122            .expect("new_transaction failed");
4123        let store = fs.root_store();
4124        object = Arc::new(
4125            ObjectStore::create_object(&store, &mut transaction, HandleOptions::default(), None)
4126                .await
4127                .expect("create_object failed"),
4128        );
4129        transaction.commit().await.expect("commit failed");
4130        for _ in 0..100 {
4131            let cloned_object = object.clone();
4132            let writer = fasync::Task::spawn(async move {
4133                let mut buf = cloned_object.allocate_buffer(10).await;
4134                buf.fill(123);
4135                cloned_object.write_or_append(Some(0), buf.as_ref()).await.expect("write failed");
4136            });
4137            let cloned_object = object.clone();
4138            let reader = fasync::Task::spawn(async move {
4139                let wait_time = rand::random_range(0..5);
4140                fasync::Timer::new(Duration::from_millis(wait_time)).await;
4141                let mut buf =
4142                    cloned_object.allocate_buffer(cloned_object.block_size().get() as usize).await;
4143                buf.fill(23);
4144                let amount =
4145                    cloned_object.read_aligned(0, buf.as_mut()).await.expect("read failed");
4146                // If we succeed in reading data, it must include the write; i.e. if we see the size
4147                // change, we should see the data too.  For this to succeed it requires locking on
4148                // the read size to ensure that when we read the size, we get the extents changed in
4149                // that same transaction.
4150                if amount != 0 {
4151                    assert_eq!(amount, 10);
4152                    assert_eq!(&buf.as_ptr_slice().subslice(0..10).to_vec()[..], &[123; 10]);
4153                }
4154            });
4155            writer.await;
4156            reader.await;
4157            object.truncate(0).await.expect("truncate failed");
4158        }
4159        fs.close().await.expect("Close failed");
4160    }
4161
4162    #[fuchsia::test]
4163    async fn test_allocated_size() {
4164        let (fs, object) = test_filesystem_and_object_with_key(None, true).await;
4165
4166        let before = object.get_properties().await.expect("get_properties failed").allocated_size;
4167        let mut buf = object.allocate_buffer(5).await;
4168        buf.copy_from_slice(b"hello");
4169        object.write_or_append(Some(0), buf.as_ref()).await.expect("write failed");
4170        let after = object.get_properties().await.expect("get_properties failed").allocated_size;
4171        assert_eq!(after, before + fs.block_size());
4172
4173        // Do the same write again and there should be no change.
4174        object.write_or_append(Some(0), buf.as_ref()).await.expect("write failed");
4175        assert_eq!(
4176            object.get_properties().await.expect("get_properties failed").allocated_size,
4177            after
4178        );
4179
4180        // extend...
4181        let mut transaction = object.new_transaction().await.expect("new_transaction failed");
4182        let offset = 1000 * fs.block_size();
4183        let before = after;
4184        object
4185            .extend(&mut transaction, offset..offset + fs.block_size())
4186            .await
4187            .expect("extend failed");
4188        transaction.commit().await.expect("commit failed");
4189        let after = object.get_properties().await.expect("get_properties failed").allocated_size;
4190        assert_eq!(after, before + fs.block_size());
4191
4192        // truncate...
4193        let before = after;
4194        let size = object.get_size();
4195        object.truncate(size - fs.block_size()).await.expect("extend failed");
4196        let after = object.get_properties().await.expect("get_properties failed").allocated_size;
4197        assert_eq!(after, before - fs.block_size());
4198
4199        // preallocate_range...
4200        let mut transaction = object.new_transaction().await.expect("new_transaction failed");
4201        let before = after;
4202        let mut file_range = offset..offset + fs.block_size();
4203        object.preallocate_range(&mut transaction, &mut file_range).await.expect("extend failed");
4204        transaction.commit().await.expect("commit failed");
4205        let after = object.get_properties().await.expect("get_properties failed").allocated_size;
4206        assert_eq!(after, before + fs.block_size());
4207        fs.close().await.expect("Close failed");
4208    }
4209
4210    #[fuchsia::test(threads = 10)]
4211    async fn test_zero() {
4212        let (fs, object) = test_filesystem_and_object().await;
4213        let expected_size = object.get_size();
4214        let mut transaction = object.new_transaction().await.expect("new_transaction failed");
4215        object.zero(&mut transaction, 0..fs.block_size() * 10).await.expect("zero failed");
4216        transaction.commit().await.expect("commit failed");
4217        assert_eq!(object.get_size(), expected_size);
4218        let mut buf = object.allocate_buffer((fs.block_size() * 10) as usize).await;
4219        assert_eq!(
4220            object.read_aligned(0, buf.as_mut()).await.expect("read failed") as u64,
4221            expected_size
4222        );
4223        assert_eq!(
4224            &buf.as_ptr_slice().subslice(0..expected_size as usize).to_vec()[..],
4225            vec![0u8; expected_size as usize].as_slice()
4226        );
4227        fs.close().await.expect("Close failed");
4228    }
4229
4230    #[fuchsia::test]
4231    async fn test_properties() {
4232        let (fs, object) = test_filesystem_and_object().await;
4233        const CRTIME: Timestamp = Timestamp::from_nanos(1234);
4234        const MTIME: Timestamp = Timestamp::from_nanos(5678);
4235        const CTIME: Timestamp = Timestamp::from_nanos(8765);
4236
4237        // ObjectProperties can be updated through `update_attributes`.
4238        // `get_properties` should reflect the latest changes.
4239        let mut transaction = object.new_transaction().await.expect("new_transaction failed");
4240        object
4241            .update_attributes(
4242                &mut transaction,
4243                Some(&fio::MutableNodeAttributes {
4244                    creation_time: Some(CRTIME.as_nanos()),
4245                    modification_time: Some(MTIME.as_nanos()),
4246                    mode: Some(111),
4247                    gid: Some(222),
4248                    ..Default::default()
4249                }),
4250                None,
4251            )
4252            .await
4253            .expect("update_attributes failed");
4254        const MTIME_NEW: Timestamp = Timestamp::from_nanos(12345678);
4255        object
4256            .update_attributes(
4257                &mut transaction,
4258                Some(&fio::MutableNodeAttributes {
4259                    modification_time: Some(MTIME_NEW.as_nanos()),
4260                    gid: Some(333),
4261                    rdev: Some(444),
4262                    ..Default::default()
4263                }),
4264                Some(CTIME),
4265            )
4266            .await
4267            .expect("update_timestamps failed");
4268        transaction.commit().await.expect("commit failed");
4269
4270        let properties = object.get_properties().await.expect("get_properties failed");
4271        assert_matches!(
4272            properties,
4273            ObjectProperties {
4274                refs: 1u64,
4275                allocated_size: TEST_OBJECT_ALLOCATED_SIZE,
4276                data_attribute_size: TEST_OBJECT_SIZE,
4277                creation_time: CRTIME,
4278                modification_time: MTIME_NEW,
4279                posix_attributes: Some(PosixAttributes { mode: 111, gid: 333, rdev: 444, .. }),
4280                change_time: CTIME,
4281                ..
4282            }
4283        );
4284        fs.close().await.expect("Close failed");
4285    }
4286
4287    #[fuchsia::test]
4288    async fn test_is_allocated() {
4289        let (fs, object) = test_filesystem_and_object().await;
4290
4291        // `test_filesystem_and_object()` wrote the buffer `TEST_DATA` to the device at offset
4292        // `TEST_DATA_OFFSET` where the length and offset are aligned to the block size.
4293        let aligned_offset = fs.block_size().align_down(TEST_DATA_OFFSET);
4294        let aligned_length = fs.block_size().align_up(TEST_DATA.len() as u64).unwrap();
4295
4296        // Check for the case where where we have the following extent layout
4297        //       [ unallocated ][ `TEST_DATA` ]
4298        // The extents before `aligned_offset` should not be allocated
4299        let (allocated, count) = object.is_allocated(0).await.expect("is_allocated failed");
4300        assert_eq!(count, aligned_offset);
4301        assert_eq!(allocated, false);
4302
4303        let (allocated, count) =
4304            object.is_allocated(aligned_offset).await.expect("is_allocated failed");
4305        assert_eq!(count, aligned_length);
4306        assert_eq!(allocated, true);
4307
4308        // Check for the case where where we query out of range
4309        let end = aligned_offset + aligned_length;
4310        object
4311            .is_allocated(end)
4312            .await
4313            .expect_err("is_allocated should have returned ERR_OUT_OF_RANGE");
4314
4315        // Check for the case where where we start querying for allocation starting from
4316        // an allocated range to the end of the device
4317        let size = 50 * fs.block_size();
4318        object.truncate(size).await.expect("extend failed");
4319
4320        let (allocated, count) = object.is_allocated(end).await.expect("is_allocated failed");
4321        assert_eq!(count, size - end);
4322        assert_eq!(allocated, false);
4323
4324        // Check for the case where where we have the following extent layout
4325        //      [ unallocated ][ `buf` ][ `buf` ]
4326        let buf_length = 5 * fs.block_size();
4327        let mut buf = object.allocate_buffer(buf_length as usize).await;
4328        buf.fill(123);
4329        let new_offset = end + 20 * fs.block_size();
4330        object.write_or_append(Some(new_offset), buf.as_ref()).await.expect("write failed");
4331        object
4332            .write_or_append(Some(new_offset + buf_length), buf.as_ref())
4333            .await
4334            .expect("write failed");
4335
4336        let (allocated, count) = object.is_allocated(end).await.expect("is_allocated failed");
4337        assert_eq!(count, new_offset - end);
4338        assert_eq!(allocated, false);
4339
4340        let (allocated, count) =
4341            object.is_allocated(new_offset).await.expect("is_allocated failed");
4342        assert_eq!(count, 2 * buf_length);
4343        assert_eq!(allocated, true);
4344
4345        // Check the case where we query from the middle of an extent
4346        let (allocated, count) = object
4347            .is_allocated(new_offset + 4 * fs.block_size())
4348            .await
4349            .expect("is_allocated failed");
4350        assert_eq!(count, 2 * buf_length - 4 * fs.block_size());
4351        assert_eq!(allocated, true);
4352
4353        // Now, write buffer to a location already written to.
4354        // Check for the case when we the following extent layout
4355        //      [ unallocated ][ `other_buf` ][ (part of) `buf` ][ `buf` ]
4356        let other_buf_length = 3 * fs.block_size();
4357        let mut other_buf = object.allocate_buffer(other_buf_length as usize).await;
4358        other_buf.fill(231);
4359        object.write_or_append(Some(new_offset), other_buf.as_ref()).await.expect("write failed");
4360
4361        // We still expect that `is_allocated(..)` will return that  there are 2*`buf_length bytes`
4362        // allocated from `new_offset`
4363        let (allocated, count) =
4364            object.is_allocated(new_offset).await.expect("is_allocated failed");
4365        assert_eq!(count, 2 * buf_length);
4366        assert_eq!(allocated, true);
4367
4368        // Check for the case when we the following extent layout
4369        //   [ unallocated ][ deleted ][ unallocated ][ deleted ][ allocated ]
4370        // Mark TEST_DATA as deleted
4371        let mut transaction = object.new_transaction().await.expect("new_transaction failed");
4372        object
4373            .zero(&mut transaction, aligned_offset..aligned_offset + aligned_length)
4374            .await
4375            .expect("zero failed");
4376        // Mark `other_buf` as deleted
4377        object
4378            .zero(&mut transaction, new_offset..new_offset + buf_length)
4379            .await
4380            .expect("zero failed");
4381        transaction.commit().await.expect("commit transaction failed");
4382
4383        let (allocated, count) = object.is_allocated(0).await.expect("is_allocated failed");
4384        assert_eq!(count, new_offset + buf_length);
4385        assert_eq!(allocated, false);
4386
4387        let (allocated, count) =
4388            object.is_allocated(new_offset + buf_length).await.expect("is_allocated failed");
4389        assert_eq!(count, buf_length);
4390        assert_eq!(allocated, true);
4391
4392        let new_end = new_offset + buf_length + count;
4393
4394        // Check for the case where there are objects with different keys.
4395        // Case that we're checking for:
4396        //      [ unallocated ][ extent (object with different key) ][ unallocated ]
4397        let store = object.owner();
4398        let mut transaction = fs
4399            .root_store()
4400            .new_transaction(lock_keys![], Options::default())
4401            .await
4402            .expect("new_transaction failed");
4403        let object2 =
4404            ObjectStore::create_object(&store, &mut transaction, HandleOptions::default(), None)
4405                .await
4406                .expect("create_object failed");
4407        transaction.commit().await.expect("commit failed");
4408
4409        object2
4410            .write_or_append(Some(new_end + fs.block_size()), buf.as_ref())
4411            .await
4412            .expect("write failed");
4413
4414        // Expecting that the extent with a different key is treated like unallocated extent
4415        let (allocated, count) = object.is_allocated(new_end).await.expect("is_allocated failed");
4416        assert_eq!(count, size - new_end);
4417        assert_eq!(allocated, false);
4418
4419        fs.close().await.expect("close failed");
4420    }
4421
4422    #[fuchsia::test(threads = 10)]
4423    async fn test_read_write_attr() {
4424        let (_fs, object) = test_filesystem_and_object().await;
4425        let data = [0xffu8; 16_384];
4426        object.write_attr(AttributeId(20), &data).await.expect("write_attr failed");
4427        let rdata = object
4428            .read_attr(AttributeId(20))
4429            .await
4430            .expect("read_attr failed")
4431            .expect("no attribute data found");
4432        assert_eq!(&data[..], &rdata[..]);
4433
4434        assert_eq!(object.read_attr(AttributeId(21)).await.expect("read_attr failed"), None);
4435    }
4436
4437    #[fuchsia::test(threads = 10)]
4438    async fn test_read_write_attr_unaligned() {
4439        let (_fs, object) = test_filesystem_and_object().await;
4440        let data_unaligned = [0x55u8; 5000];
4441        object.write_attr(AttributeId(22), &data_unaligned).await.expect("write_attr failed");
4442        let rdata = object
4443            .read_attr(AttributeId(22))
4444            .await
4445            .expect("read_attr failed")
4446            .expect("no attribute data found");
4447        assert_eq!(&data_unaligned[..], &rdata[..]);
4448    }
4449
4450    #[fuchsia::test(threads = 10)]
4451    async fn test_read_write_attr_empty() {
4452        let (_fs, object) = test_filesystem_and_object().await;
4453        object.write_attr(AttributeId(23), &[]).await.expect("write_attr failed");
4454        let rdata = object
4455            .read_attr(AttributeId(23))
4456            .await
4457            .expect("read_attr failed")
4458            .expect("no attribute data found");
4459        assert!(rdata.is_empty());
4460    }
4461
4462    #[fuchsia::test(threads = 10)]
4463    async fn test_allocate_basic() {
4464        let (fs, object) = test_filesystem_and_empty_object().await;
4465        let block_size = fs.block_size();
4466        let file_size = block_size * 10;
4467        object.truncate(file_size).await.unwrap();
4468
4469        let small_buf_size = block_size.get() as usize;
4470        let medium_buf_size = (block_size * 2) as usize;
4471        let large_buf_size = (block_size * 3) as usize;
4472
4473        let mut small_buf = object.allocate_buffer(small_buf_size).await;
4474        let mut medium_buf = object.allocate_buffer(medium_buf_size).await;
4475        let mut large_buf = object.allocate_buffer(large_buf_size).await;
4476
4477        assert_eq!(object.read_aligned(0, small_buf.as_mut()).await.unwrap(), small_buf_size);
4478        assert_eq!(small_buf.to_vec(), vec![0; small_buf_size]);
4479        assert_eq!(object.read_aligned(0, large_buf.as_mut()).await.unwrap(), large_buf_size);
4480        assert_eq!(large_buf.to_vec(), vec![0; large_buf_size]);
4481        assert_eq!(object.read_aligned(0, medium_buf.as_mut()).await.unwrap(), medium_buf_size);
4482        assert_eq!(medium_buf.to_vec(), vec![0; medium_buf_size]);
4483
4484        // Allocation succeeds, and without any writes to the location it shows up as zero.
4485        object.allocate(block_size.get()..block_size * 3).await.unwrap();
4486
4487        // Test starting before, inside, and after the allocated section with every sized buffer.
4488        for (buf_index, buf) in [small_buf, large_buf, medium_buf].iter_mut().enumerate() {
4489            for offset in 0..4 {
4490                assert_eq!(
4491                    object.read_aligned(block_size * offset, buf.as_mut()).await.unwrap(),
4492                    buf.len(),
4493                    "buf_index: {}, read offset: {}",
4494                    buf_index,
4495                    offset,
4496                );
4497                assert_eq!(
4498                    &buf.to_vec(),
4499                    &vec![0; buf.len()],
4500                    "buf_index: {}, read offset: {}",
4501                    buf_index,
4502                    offset,
4503                );
4504            }
4505        }
4506
4507        fs.close().await.expect("close failed");
4508    }
4509
4510    #[fuchsia::test(threads = 10)]
4511    async fn test_allocate_extends_file() {
4512        let (fs, object) = test_filesystem_and_empty_object().await;
4513        let block_size = fs.block_size();
4514        let buf_size = block_size.get() as usize;
4515        let mut buf = object.allocate_buffer(buf_size).await;
4516
4517        assert_eq!(object.read_aligned(0, buf.as_mut()).await.unwrap(), buf.len());
4518        assert_eq!(buf.to_vec(), vec![0; buf_size]);
4519
4520        assert!(TEST_OBJECT_SIZE < block_size * 4);
4521        // Allocation succeeds, and without any writes to the location it shows up as zero.
4522        object.allocate(0..block_size * 4).await.unwrap();
4523        assert_eq!(object.read_aligned(0, buf.as_mut()).await.unwrap(), buf.len());
4524        assert_eq!(buf.to_vec(), vec![0; buf_size]);
4525        assert_eq!(object.read_aligned(block_size.get(), buf.as_mut()).await.unwrap(), buf.len());
4526        assert_eq!(buf.to_vec(), vec![0; buf_size]);
4527        assert_eq!(object.read_aligned(block_size * 3, buf.as_mut()).await.unwrap(), buf.len());
4528        assert_eq!(buf.to_vec(), vec![0; buf_size]);
4529
4530        fs.close().await.expect("close failed");
4531    }
4532
4533    #[fuchsia::test(threads = 10)]
4534    async fn test_allocate_past_end() {
4535        let (fs, object) = test_filesystem_and_empty_object().await;
4536        let block_size = fs.block_size();
4537        let buf_size = block_size.get() as usize;
4538        let mut buf = object.allocate_buffer(buf_size).await;
4539
4540        assert_eq!(object.read_aligned(0, buf.as_mut()).await.unwrap(), buf.len());
4541        assert_eq!(buf.to_vec(), vec![0; buf_size]);
4542
4543        assert!(TEST_OBJECT_SIZE < block_size * 4);
4544        // Allocation succeeds, and without any writes to the location it shows up as zero.
4545        object.allocate(block_size * 4..block_size * 6).await.unwrap();
4546        assert_eq!(object.read_aligned(0, buf.as_mut()).await.unwrap(), buf.len());
4547        assert_eq!(buf.to_vec(), vec![0; buf_size]);
4548        assert_eq!(object.read_aligned(block_size * 4, buf.as_mut()).await.unwrap(), buf.len());
4549        assert_eq!(buf.to_vec(), vec![0; buf_size]);
4550        assert_eq!(object.read_aligned(block_size * 5, buf.as_mut()).await.unwrap(), buf.len());
4551        assert_eq!(buf.to_vec(), vec![0; buf_size]);
4552
4553        fs.close().await.expect("close failed");
4554    }
4555
4556    #[fuchsia::test(threads = 10)]
4557    async fn test_allocate_read_attr() {
4558        let (fs, object) = test_filesystem_and_empty_object().await;
4559        let block_size = fs.block_size();
4560        let file_size = block_size * 4;
4561        object.truncate(file_size).await.unwrap();
4562
4563        let content = object
4564            .read_attr(object.attribute_id())
4565            .await
4566            .expect("failed to read attr")
4567            .expect("attr returned none");
4568        assert_eq!(content.as_ref(), &vec![0; file_size as usize]);
4569
4570        object.allocate(block_size.get()..block_size * 3).await.unwrap();
4571
4572        let content = object
4573            .read_attr(object.attribute_id())
4574            .await
4575            .expect("failed to read attr")
4576            .expect("attr returned none");
4577        assert_eq!(content.as_ref(), &vec![0; file_size as usize]);
4578
4579        fs.close().await.expect("close failed");
4580    }
4581
4582    #[fuchsia::test(threads = 10)]
4583    async fn test_allocate_existing_data() {
4584        struct Case {
4585            written_ranges: Vec<Range<usize>>,
4586            allocate_range: Range<u64>,
4587        }
4588        let cases = [
4589            Case { written_ranges: vec![4..7], allocate_range: 4..7 },
4590            Case { written_ranges: vec![4..7], allocate_range: 3..8 },
4591            Case { written_ranges: vec![4..7], allocate_range: 5..6 },
4592            Case { written_ranges: vec![4..7], allocate_range: 5..8 },
4593            Case { written_ranges: vec![4..7], allocate_range: 3..5 },
4594            Case { written_ranges: vec![0..1, 2..3, 4..5, 6..7, 8..9], allocate_range: 0..10 },
4595            Case { written_ranges: vec![0..2, 4..6, 7..10], allocate_range: 1..8 },
4596        ];
4597
4598        for case in cases {
4599            let (fs, object) = test_filesystem_and_empty_object().await;
4600            let block_size = fs.block_size();
4601            let file_size = block_size * 10;
4602            object.truncate(file_size).await.unwrap();
4603
4604            for write in &case.written_ranges {
4605                let write_len = (write.end - write.start) * block_size.get() as usize;
4606                let mut write_buf = object.allocate_buffer(write_len).await;
4607                write_buf.fill(0xff);
4608                assert_eq!(
4609                    object
4610                        .write_or_append(Some(block_size * write.start as u64), write_buf.as_ref())
4611                        .await
4612                        .unwrap(),
4613                    file_size
4614                );
4615            }
4616
4617            let mut expected_buf = object.allocate_buffer(file_size as usize).await;
4618            assert_eq!(
4619                object.read_aligned(0, expected_buf.as_mut()).await.unwrap(),
4620                expected_buf.len()
4621            );
4622
4623            object
4624                .allocate(
4625                    case.allocate_range.start * block_size..case.allocate_range.end * block_size,
4626                )
4627                .await
4628                .unwrap();
4629
4630            let mut read_buf = object.allocate_buffer(file_size as usize).await;
4631            assert_eq!(object.read_aligned(0, read_buf.as_mut()).await.unwrap(), read_buf.len());
4632            assert_eq!(read_buf.to_vec(), expected_buf.to_vec());
4633
4634            fs.close().await.expect("close failed");
4635        }
4636    }
4637
4638    async fn get_modes(
4639        obj: &DataObjectHandle<ObjectStore>,
4640        mut search_range: Range<u64>,
4641    ) -> Vec<(Range<u64>, ExtentMode)> {
4642        let mut modes = Vec::new();
4643        let store = obj.store();
4644        let tree = store.tree();
4645        let layer_set = tree.layer_set();
4646        let mut merger = layer_set.merger();
4647        let mut iter = merger
4648            .query(Query::FullRange(&ObjectKey::attribute(
4649                obj.object_id(),
4650                AttributeId::DATA,
4651                AttributeKey::Extent(Extent::search_key_from_offset(search_range.start)),
4652            )))
4653            .await
4654            .unwrap();
4655        loop {
4656            match iter.get() {
4657                Some(ItemRef {
4658                    key:
4659                        ObjectKey {
4660                            object_id,
4661                            data:
4662                                ObjectKeyData::Attribute(
4663                                    AttributeId::DATA,
4664                                    AttributeKey::Extent(extent),
4665                                ),
4666                        },
4667                    value: ObjectValue::Extent(ExtentValue::Some { mode, .. }),
4668                    ..
4669                }) if *object_id == obj.object_id() => {
4670                    if search_range.end <= extent.start {
4671                        break;
4672                    }
4673                    let found_range = std::cmp::max(search_range.start, extent.start)
4674                        ..std::cmp::min(search_range.end, extent.end);
4675                    search_range.start = found_range.end;
4676                    modes.push((found_range, mode.clone()));
4677                    if search_range.start == search_range.end {
4678                        break;
4679                    }
4680                    iter.advance().await.unwrap();
4681                }
4682                x => panic!("looking for extent record, found this {:?}", x),
4683            }
4684        }
4685        modes
4686    }
4687
4688    async fn assert_all_overwrite(
4689        obj: &DataObjectHandle<ObjectStore>,
4690        mut search_range: Range<u64>,
4691    ) {
4692        let modes = get_modes(obj, search_range.clone()).await;
4693        for mode in modes {
4694            assert_eq!(
4695                mode.0.start, search_range.start,
4696                "missing mode in range {}..{}",
4697                search_range.start, mode.0.start
4698            );
4699            match mode.1 {
4700                ExtentMode::Overwrite | ExtentMode::OverwritePartial(_) => (),
4701                m => panic!("mode at range {:?} was not overwrite, instead found {:?}", mode.0, m),
4702            }
4703            assert!(
4704                mode.0.end <= search_range.end,
4705                "mode ends beyond search range (bug in test) - search_range: {:?}, mode: {:?}",
4706                search_range,
4707                mode,
4708            );
4709            search_range.start = mode.0.end;
4710        }
4711        assert_eq!(
4712            search_range.start, search_range.end,
4713            "missing mode in range {:?}",
4714            search_range
4715        );
4716    }
4717
4718    #[fuchsia::test(threads = 10)]
4719    async fn test_multi_overwrite() {
4720        #[derive(Debug)]
4721        struct Case {
4722            pre_writes: Vec<Range<usize>>,
4723            allocate_ranges: Vec<Range<u64>>,
4724            overwrites: Vec<Vec<Range<u64>>>,
4725        }
4726        let cases = [
4727            Case {
4728                pre_writes: Vec::new(),
4729                allocate_ranges: vec![1..3],
4730                overwrites: vec![vec![1..3]],
4731            },
4732            Case {
4733                pre_writes: Vec::new(),
4734                allocate_ranges: vec![0..1, 1..2, 2..3, 3..4],
4735                overwrites: vec![vec![0..4]],
4736            },
4737            Case {
4738                pre_writes: Vec::new(),
4739                allocate_ranges: vec![0..4],
4740                overwrites: vec![vec![0..1], vec![1..2], vec![3..4]],
4741            },
4742            Case {
4743                pre_writes: Vec::new(),
4744                allocate_ranges: vec![0..4],
4745                overwrites: vec![vec![3..4]],
4746            },
4747            Case {
4748                pre_writes: Vec::new(),
4749                allocate_ranges: vec![0..4],
4750                overwrites: vec![vec![3..4], vec![2..3], vec![1..2]],
4751            },
4752            Case {
4753                pre_writes: Vec::new(),
4754                allocate_ranges: vec![1..2, 5..6, 7..8],
4755                overwrites: vec![vec![5..6]],
4756            },
4757            Case {
4758                pre_writes: Vec::new(),
4759                allocate_ranges: vec![1..3],
4760                overwrites: vec![
4761                    vec![1..3],
4762                    vec![1..3],
4763                    vec![1..3],
4764                    vec![1..3],
4765                    vec![1..3],
4766                    vec![1..3],
4767                    vec![1..3],
4768                    vec![1..3],
4769                ],
4770            },
4771            Case {
4772                pre_writes: Vec::new(),
4773                allocate_ranges: vec![0..5],
4774                overwrites: vec![
4775                    vec![1..3],
4776                    vec![1..3],
4777                    vec![1..3],
4778                    vec![1..3],
4779                    vec![1..3],
4780                    vec![1..3],
4781                    vec![1..3],
4782                    vec![1..3],
4783                ],
4784            },
4785            Case {
4786                pre_writes: Vec::new(),
4787                allocate_ranges: vec![0..5],
4788                overwrites: vec![vec![0..2, 2..4, 4..5]],
4789            },
4790            Case {
4791                pre_writes: Vec::new(),
4792                allocate_ranges: vec![0..5, 5..10],
4793                overwrites: vec![vec![1..2, 2..3, 4..7, 7..8]],
4794            },
4795            Case {
4796                pre_writes: Vec::new(),
4797                allocate_ranges: vec![0..4, 6..10],
4798                overwrites: vec![vec![2..3, 7..9]],
4799            },
4800            Case {
4801                pre_writes: Vec::new(),
4802                allocate_ranges: vec![0..10],
4803                overwrites: vec![vec![1..2, 5..10], vec![0..1, 5..10], vec![0..5, 5..10]],
4804            },
4805            Case {
4806                pre_writes: Vec::new(),
4807                allocate_ranges: vec![0..10],
4808                overwrites: vec![vec![0..2, 2..4, 4..6, 6..8, 8..10], vec![0..5, 5..10]],
4809            },
4810            Case {
4811                pre_writes: vec![1..3],
4812                allocate_ranges: vec![1..3],
4813                overwrites: vec![vec![1..3]],
4814            },
4815            Case {
4816                pre_writes: vec![1..3],
4817                allocate_ranges: vec![4..6],
4818                overwrites: vec![vec![5..6]],
4819            },
4820            Case {
4821                pre_writes: vec![1..3],
4822                allocate_ranges: vec![0..4],
4823                overwrites: vec![vec![0..4]],
4824            },
4825            Case {
4826                pre_writes: vec![1..3],
4827                allocate_ranges: vec![2..4],
4828                overwrites: vec![vec![2..4]],
4829            },
4830            Case {
4831                pre_writes: vec![3..5],
4832                allocate_ranges: vec![1..3, 6..7],
4833                overwrites: vec![vec![1..3, 6..7]],
4834            },
4835            Case {
4836                pre_writes: vec![1..3, 5..7, 8..9],
4837                allocate_ranges: vec![0..5],
4838                overwrites: vec![vec![0..2, 2..5], vec![0..5]],
4839            },
4840            Case {
4841                pre_writes: Vec::new(),
4842                allocate_ranges: vec![0..10, 4..6],
4843                overwrites: Vec::new(),
4844            },
4845            Case {
4846                pre_writes: Vec::new(),
4847                allocate_ranges: vec![3..8, 5..10],
4848                overwrites: Vec::new(),
4849            },
4850            Case {
4851                pre_writes: Vec::new(),
4852                allocate_ranges: vec![5..10, 3..8],
4853                overwrites: Vec::new(),
4854            },
4855        ];
4856
4857        for (i, case) in cases.into_iter().enumerate() {
4858            log::info!("running case {} - {:?}", i, case);
4859            let (fs, object) = test_filesystem_and_empty_object().await;
4860            let block_size = fs.block_size();
4861            let file_size = block_size * 10;
4862            object.truncate(file_size).await.unwrap();
4863
4864            for write in case.pre_writes {
4865                let write_len = (write.end - write.start) * block_size.get() as usize;
4866                let mut write_buf = object.allocate_buffer(write_len).await;
4867                write_buf.fill(0xff);
4868                assert_eq!(
4869                    object
4870                        .write_or_append(Some(block_size * write.start as u64), write_buf.as_ref())
4871                        .await
4872                        .unwrap(),
4873                    file_size
4874                );
4875            }
4876
4877            for allocate_range in &case.allocate_ranges {
4878                object
4879                    .allocate(allocate_range.start * block_size..allocate_range.end * block_size)
4880                    .await
4881                    .unwrap();
4882            }
4883
4884            for allocate_range in case.allocate_ranges {
4885                assert_all_overwrite(
4886                    &object,
4887                    allocate_range.start * block_size..allocate_range.end * block_size,
4888                )
4889                .await;
4890            }
4891
4892            for overwrite in case.overwrites {
4893                let mut write_len = 0;
4894                let overwrite = overwrite
4895                    .into_iter()
4896                    .map(|r| {
4897                        write_len += (r.end - r.start) * block_size;
4898                        r.start * block_size..r.end * block_size
4899                    })
4900                    .collect::<Vec<_>>();
4901                let mut write_buf = object.allocate_buffer(write_len as usize).await;
4902                let data = (0..20).cycle().take(write_len as usize).collect::<Vec<_>>();
4903                write_buf.copy_from_slice(&data);
4904
4905                let mut expected_buf = object.allocate_buffer(file_size as usize).await;
4906                assert_eq!(
4907                    object.read_aligned(0, expected_buf.as_mut()).await.unwrap(),
4908                    expected_buf.len()
4909                );
4910                let mut expected_buf_slice = expected_buf.as_mut_ptr_slice();
4911                let mut data_slice = data.as_slice();
4912                for r in &overwrite {
4913                    let len = r.length().unwrap() as usize;
4914                    let (copy_from, rest) = data_slice.split_at(len);
4915                    expected_buf_slice
4916                        .subslice_mut(r.start as usize..r.end as usize)
4917                        .copy_from_slice(&copy_from);
4918                    data_slice = rest;
4919                }
4920
4921                let mut transaction = object.new_transaction().await.unwrap();
4922                object
4923                    .multi_overwrite(
4924                        &mut transaction,
4925                        AttributeId::DATA,
4926                        &overwrite,
4927                        write_buf.as_mut(),
4928                    )
4929                    .await
4930                    .unwrap_or_else(|_| panic!("multi_overwrite error on case {}", i));
4931                // Double check the emitted checksums. We should have one u64 checksum for every
4932                // block we wrote to disk.
4933                let mut checksummed_range_length = 0;
4934                let mut num_checksums = 0;
4935                for (device_range, checksums, _) in transaction.checksums() {
4936                    let range_len = device_range.end - device_range.start;
4937                    let checksums_len = checksums.len() as u64;
4938                    assert_eq!(range_len / checksums_len, block_size);
4939                    checksummed_range_length += range_len;
4940                    num_checksums += checksums_len;
4941                }
4942                assert_eq!(checksummed_range_length, write_len);
4943                assert_eq!(num_checksums, write_len / block_size);
4944                transaction.commit().await.unwrap();
4945
4946                let mut buf = object.allocate_buffer(file_size as usize).await;
4947                assert_eq!(
4948                    object.read_aligned(0, buf.as_mut()).await.unwrap(),
4949                    buf.len(),
4950                    "failed length check on case {}",
4951                    i,
4952                );
4953                assert_eq!(buf.to_vec(), expected_buf.to_vec(), "failed on case {}", i);
4954            }
4955
4956            fsck_volume(&fs, object.store().store_object_id(), None).await.expect("fsck failed");
4957            fs.close().await.expect("close failed");
4958        }
4959    }
4960
4961    #[fuchsia::test(threads = 10)]
4962    async fn test_multi_overwrite_mode_updates() {
4963        let (fs, object) = test_filesystem_and_empty_object().await;
4964        let block_size = fs.block_size();
4965        let file_size = block_size * 10;
4966        object.truncate(file_size).await.unwrap();
4967
4968        let mut expected_bitmap = BitVec::from_elem(10, false);
4969
4970        object.allocate(0..10 * block_size).await.unwrap();
4971        assert_eq!(
4972            get_modes(&object, 0..10 * block_size).await,
4973            vec![(0..10 * block_size, ExtentMode::OverwritePartial(expected_bitmap.clone()))]
4974        );
4975
4976        let mut write_buf = object.allocate_buffer((2 * block_size) as usize).await;
4977        let data = (0..20).cycle().take(write_buf.len()).collect::<Vec<_>>();
4978        write_buf.copy_from_slice(&data);
4979        let mut transaction = object.new_transaction().await.unwrap();
4980        object
4981            .multi_overwrite(
4982                &mut transaction,
4983                AttributeId::DATA,
4984                &[2 * block_size..4 * block_size],
4985                write_buf.as_mut(),
4986            )
4987            .await
4988            .unwrap();
4989        transaction.commit().await.unwrap();
4990
4991        expected_bitmap.set(2, true);
4992        expected_bitmap.set(3, true);
4993        assert_eq!(
4994            get_modes(&object, 0..10 * block_size).await,
4995            vec![(0..10 * block_size, ExtentMode::OverwritePartial(expected_bitmap.clone()))]
4996        );
4997
4998        let mut write_buf = object.allocate_buffer((3 * block_size) as usize).await;
4999        let data = (0..20).cycle().take(write_buf.len()).collect::<Vec<_>>();
5000        write_buf.copy_from_slice(&data);
5001        let mut transaction = object.new_transaction().await.unwrap();
5002        object
5003            .multi_overwrite(
5004                &mut transaction,
5005                AttributeId::DATA,
5006                &[3 * block_size..5 * block_size, 6 * block_size..7 * block_size],
5007                write_buf.as_mut(),
5008            )
5009            .await
5010            .unwrap();
5011        transaction.commit().await.unwrap();
5012
5013        expected_bitmap.set(4, true);
5014        expected_bitmap.set(6, true);
5015        assert_eq!(
5016            get_modes(&object, 0..10 * block_size).await,
5017            vec![(0..10 * block_size, ExtentMode::OverwritePartial(expected_bitmap.clone()))]
5018        );
5019
5020        let mut write_buf = object.allocate_buffer((6 * block_size) as usize).await;
5021        let data = (0..20).cycle().take(write_buf.len()).collect::<Vec<_>>();
5022        write_buf.copy_from_slice(&data);
5023        let mut transaction = object.new_transaction().await.unwrap();
5024        object
5025            .multi_overwrite(
5026                &mut transaction,
5027                AttributeId::DATA,
5028                &[
5029                    0..2 * block_size,
5030                    5 * block_size..6 * block_size,
5031                    7 * block_size..10 * block_size,
5032                ],
5033                write_buf.as_mut(),
5034            )
5035            .await
5036            .unwrap();
5037        transaction.commit().await.unwrap();
5038
5039        assert_eq!(
5040            get_modes(&object, 0..10 * block_size).await,
5041            vec![(0..10 * block_size, ExtentMode::Overwrite)]
5042        );
5043
5044        fs.close().await.expect("close failed");
5045    }
5046
5047    #[fuchsia::test(threads = 10)]
5048    async fn test_check_unwritten_zero() {
5049        let device = DeviceHolder::new(FakeDevice::new(256 * 1024, TEST_DEVICE_BLOCK_SIZE));
5050        let fs = FxFilesystem::new_empty(device).await.expect("new_empty failed");
5051        let object = create_object_with_key(fs.clone(), Some(&new_insecure_crypt()), false).await;
5052        let block_size = fs.block_size();
5053
5054        // Set up a file with eight blocks to look like this:
5055        // | None | COW | COW | None | Overwrite(unwritten) | Overwrite(written) | None |
5056        let file_size = block_size * 7;
5057        object.truncate(file_size).await.unwrap();
5058        assert!(object.check_unwritten_zero(0..file_size).await.unwrap());
5059
5060        let mut buffer = object.allocate_buffer(block_size.get() as usize).await;
5061        buffer.fill(1);
5062        object
5063            .write_or_append(Some(block_size.get()), buffer.as_ref())
5064            .await
5065            .expect("write failed");
5066        object.write_or_append(Some(block_size * 2), buffer.as_ref()).await.expect("write failed");
5067
5068        object.allocate((block_size * 4)..(block_size * 6)).await.expect("Allocate failed");
5069        let mut transaction = fs
5070            .root_store()
5071            .new_transaction(
5072                lock_keys![LockKey::object(object.store().store_object_id(), object.object_id(),)],
5073                Options::default(),
5074            )
5075            .await
5076            .expect("new_transaction failed");
5077        object
5078            .multi_overwrite(
5079                &mut transaction,
5080                AttributeId::DATA,
5081                &vec![(block_size * 5)..(block_size * 6)],
5082                buffer.as_mut(),
5083            )
5084            .await
5085            .expect("Multi overwrite");
5086        transaction.commit().await.expect("Committing overwrite");
5087
5088        // Anything touching the COW ranges should fail.
5089        assert!(!object.check_unwritten_zero(0..(block_size * 2)).await.unwrap());
5090        assert!(!object.check_unwritten_zero(block_size.get()..(block_size * 3)).await.unwrap());
5091        assert!(!object.check_unwritten_zero((block_size * 2)..(block_size * 4)).await.unwrap());
5092
5093        // This should be fine, as the OverwritePartial should only touch the unwritten block.
5094        assert!(object.check_unwritten_zero((block_size * 3)..(block_size * 5)).await.unwrap());
5095
5096        // These should touch the written overwrite block and fail.
5097        assert!(!object.check_unwritten_zero((block_size * 4)..(block_size * 6)).await.unwrap());
5098        assert!(!object.check_unwritten_zero((block_size * 5)..(block_size * 7)).await.unwrap());
5099
5100        fs.close().await.expect("close failed");
5101    }
5102
5103    #[fuchsia::test]
5104    async fn test_allocate_large_file() {
5105        let device = DeviceHolder::new(FakeDevice::new(8192, 4096));
5106        let fs = FxFilesystem::new_empty(device).await.expect("new_empty failed");
5107        let object = create_object_with_key(fs.clone(), Some(&new_insecure_crypt()), false).await;
5108        let block_size = fs.block_size().get();
5109
5110        for i in (0..1600).step_by(2) {
5111            object.allocate(i * block_size..(i + 1) * block_size).await.expect("allocate failed");
5112        }
5113        object.allocate(0..1600 * block_size).await.expect("allocate failed");
5114
5115        fs.close().await.expect("close failed");
5116    }
5117}