Skip to main content

fxfs/object_store/
graveyard.rs

1// Copyright 2021 The Fuchsia Authors. All rights reserved.
2// Use of this source code is governed by a BSD-style license that can be
3// found in the LICENSE file.
4
5use crate::errors::FxfsError;
6use crate::filesystem::FxFilesystem;
7use crate::log::*;
8use crate::lsm_tree::Query;
9use crate::lsm_tree::merge::{Merger, MergerIterator};
10use crate::lsm_tree::types::{ItemRef, LayerIterator};
11use crate::object_store::object_record::{
12    ObjectAttributes, ObjectKey, ObjectKeyData, ObjectKind, ObjectValue, Timestamp,
13};
14use crate::object_store::transaction::{Mutation, Transaction};
15use crate::object_store::{AttributeId, ObjectStore};
16use anyhow::{Context, Error, anyhow, bail};
17use fuchsia_async::{self as fasync};
18use fuchsia_sync::Mutex;
19use futures::StreamExt;
20use futures::channel::mpsc::{UnboundedReceiver, UnboundedSender, unbounded};
21use futures::channel::oneshot;
22use fxfs_trace::{TraceFutureExt, trace_future_args};
23use std::collections::BTreeSet;
24use std::sync::atomic::Ordering;
25use std::sync::{Arc, Weak};
26
27enum ReaperTask {
28    None,
29    Pending(UnboundedReceiver<Message>),
30    Running(fasync::Task<()>),
31}
32
33/// A graveyard exists as a place to park objects that should be deleted when they are no longer in
34/// use.  How objects enter and leave the graveyard is up to the caller to decide.  The intention is
35/// that at mount time, any objects in the graveyard will get removed.  Each object store has a
36/// directory like object that contains a list of the objects within that store that are part of the
37/// graveyard.  A single instance of this Graveyard struct manages *all* stores.
38///
39/// Tombstoning (purging) an object or attribute in the graveyard deallocates its extents, removes
40/// its graveyard entry, and inserts an LSM-tree tombstone record (`ObjectValue::None`) that deletes
41/// all remaining records for the object or attribute during compaction.
42pub struct Graveyard {
43    filesystem: Weak<FxFilesystem>,
44    reaper_task: Mutex<ReaperTask>,
45    channel: UnboundedSender<Message>,
46}
47
48enum Message {
49    // Tombstone the object identified by <store-id>, <object-id>, Option<attribute-id>. If
50    // <attribute-id> is Some, tombstone just the attribute instead of the entire object.
51    Tombstone(u64, u64, Option<AttributeId>),
52
53    // Trims the identified object.
54    Trim(u64, u64),
55
56    // When the flush message is processed, notifies sender.  This allows the receiver to know
57    // that all preceding tombstone messages have been processed.
58    Flush(oneshot::Sender<()>),
59}
60
61#[fxfs_trace::trace]
62impl Graveyard {
63    /// Creates a new instance of the graveyard manager.
64    pub fn new(filesystem: Weak<FxFilesystem>) -> Arc<Self> {
65        let (sender, receiver) = unbounded();
66        Arc::new(Graveyard {
67            filesystem,
68            reaper_task: Mutex::new(ReaperTask::Pending(receiver)),
69            channel: sender,
70        })
71    }
72
73    /// Creates a graveyard object in `store`.  Returns the object ID for the graveyard object.
74    pub async fn create(
75        transaction: &mut Transaction<'_>,
76        store: &ObjectStore,
77    ) -> Result<u64, Error> {
78        let reserved_object_id = store.get_next_object_id(transaction).await?;
79        let object_id = reserved_object_id.get();
80        let now = Timestamp::now();
81        transaction.add(
82            store.store_object_id,
83            Mutation::insert_object(
84                ObjectKey::object(reserved_object_id.release().get()),
85                ObjectValue::Object {
86                    kind: ObjectKind::Graveyard,
87                    attributes: ObjectAttributes {
88                        creation_time: now.clone(),
89                        modification_time: now,
90                        ..Default::default()
91                    },
92                },
93            ),
94        );
95        Ok(object_id)
96    }
97
98    /// Starts an asynchronous task to reap the graveyard for all entries older than
99    /// |journal_offset| (exclusive).
100    /// If a task is already started, this has no effect, even if that task was targeting an older
101    /// |journal_offset|.
102    pub fn reap_async(self: Arc<Self>) {
103        let mut reaper_task = self.reaper_task.lock();
104        if let ReaperTask::Pending(_) = &*reaper_task {
105            if let ReaperTask::Pending(receiver) =
106                std::mem::replace(&mut *reaper_task, ReaperTask::None)
107            {
108                *reaper_task = ReaperTask::Running(fasync::Task::spawn(
109                    self.clone()
110                        .reap_task(receiver)
111                        .trace(trace_future_args!("Graveyard::reap_task")),
112                ));
113            } else {
114                unreachable!();
115            }
116        }
117    }
118
119    /// Returns a future which completes when the ongoing reap task (if it exists) completes.
120    pub async fn wait_for_reap(&self) {
121        self.channel.close_channel();
122        let task = std::mem::replace(&mut *self.reaper_task.lock(), ReaperTask::None);
123        if let ReaperTask::Running(task) = task {
124            task.await;
125        }
126    }
127
128    async fn reap_task(self: Arc<Self>, mut receiver: UnboundedReceiver<Message>) {
129        // Wait and process reap requests.
130        while let Some(message) = receiver.next().await {
131            match message {
132                Message::Tombstone(store_id, object_id, attribute_id) => {
133                    let Some(fs) = self.filesystem.upgrade() else { return };
134                    let res = if let Some(attribute_id) = attribute_id {
135                        fs.tombstone_attribute(store_id, object_id, attribute_id).await
136                    } else {
137                        fs.tombstone_object(store_id, object_id, None).await
138                    };
139                    if let Err(e) = res {
140                        error!(
141                            error:? = e,
142                            store_id,
143                            oid = object_id,
144                            attribute_id;
145                            "Tombstone error"
146                        );
147                    }
148                }
149                Message::Trim(store_id, object_id) => {
150                    let Some(fs) = self.filesystem.upgrade() else { return };
151                    if let Err(e) = Self::trim(&fs, store_id, object_id).await {
152                        error!(error:? = e, store_id, oid = object_id; "Tombstone error");
153                    }
154                }
155                Message::Flush(sender) => {
156                    let _ = sender.send(());
157                }
158            }
159        }
160    }
161
162    /// Performs the initial mount-time reap for the given store.  This will queue all items in the
163    /// graveyard.  Concurrently adding more entries to the graveyard will lead to undefined
164    /// behaviour: the entries might or might not be immediately tombstoned, so callers should wait
165    /// for this to return before changing to a state where more entries can be added.  Once this
166    /// has returned, entries will be tombstoned in the background.
167    #[trace]
168    pub async fn initial_reap(self: &Arc<Self>, store: &ObjectStore) -> Result<usize, Error> {
169        if store.filesystem().options().skip_initial_reap {
170            return Ok(0);
171        }
172        let mut count = 0;
173        let layer_set = store.tree().layer_set();
174        let mut merger = layer_set.merger();
175        let graveyard_object_id = store.graveyard_directory_object_id();
176        let mut iter = Self::iter(graveyard_object_id, &mut merger).await?;
177        let store_id = store.store_object_id();
178        let mut queued_objects = BTreeSet::new();
179        while let Some(GraveyardEntryInfo { object_id, attribute_id, value }) = iter.get() {
180            store.graveyard_entries.fetch_add(1, Ordering::Relaxed);
181            match value {
182                ObjectValue::Some => {
183                    if let Some(attribute_id) = attribute_id {
184                        // If the object is already queued for tombstone, don't queue any attributes
185                        // under it as well. The object tombstone will clean up any attributes as
186                        // well as their graveyard entries.
187                        if !queued_objects.contains(&(store_id, object_id)) {
188                            self.queue_tombstone_attribute(store_id, object_id, attribute_id)
189                        }
190                    } else {
191                        queued_objects.insert((store_id, object_id));
192                        self.queue_tombstone_object(store_id, object_id)
193                    }
194                }
195                ObjectValue::Trim => {
196                    if attribute_id.is_some() {
197                        return Err(anyhow!(
198                            "Trim is not currently supported for a single attribute"
199                        ));
200                    }
201                    self.queue_trim(store_id, object_id)
202                }
203                _ => bail!(anyhow!(FxfsError::Inconsistent).context("Bad graveyard value")),
204            }
205            count += 1;
206            iter.advance().await?;
207        }
208        Ok(count)
209    }
210    /// Queues an object in the graveyard for asynchronous tombstoning (see
211    /// [`FxFilesystem::tombstone_object`]).
212    pub fn queue_tombstone_object(&self, store_id: u64, object_id: u64) {
213        let _ = self.channel.unbounded_send(Message::Tombstone(store_id, object_id, None));
214    }
215
216    /// Queues an object's attribute in the graveyard for asynchronous tombstoning (see
217    /// [`FxFilesystem::tombstone_attribute`]).
218    pub fn queue_tombstone_attribute(
219        &self,
220        store_id: u64,
221        object_id: u64,
222        attribute_id: AttributeId,
223    ) {
224        let _ = self.channel.unbounded_send(Message::Tombstone(
225            store_id,
226            object_id,
227            Some(attribute_id),
228        ));
229    }
230
231    fn queue_trim(&self, store_id: u64, object_id: u64) {
232        let _ = self.channel.unbounded_send(Message::Trim(store_id, object_id));
233    }
234
235    /// Waits for all preceding queued tombstones to finish.
236    pub async fn flush(&self) {
237        let (sender, receiver) = oneshot::channel::<()>();
238        self.channel.unbounded_send(Message::Flush(sender)).unwrap();
239        receiver.await.unwrap();
240    }
241
242    async fn trim(fs: &FxFilesystem, store_id: u64, object_id: u64) -> Result<(), Error> {
243        let store = fs
244            .object_manager()
245            .store(store_id)
246            .with_context(|| format!("Failed to get store {}", store_id))?;
247        let truncate_guard = fs.truncate_guard(store_id, object_id).await;
248        store.trim(object_id, &truncate_guard).await.context("Failed to trim object")
249    }
250
251    /// Returns an iterator that will return graveyard entries skipping deleted ones.  Example
252    /// usage:
253    ///
254    ///   let layer_set = graveyard.store().tree().layer_set();
255    ///   let mut merger = layer_set.merger();
256    ///   let mut iter = graveyard.iter(&mut merger).await?;
257    ///
258    pub async fn iter<'a, 'b>(
259        graveyard_object_id: u64,
260        merger: &'a mut Merger<'b, ObjectKey, ObjectValue>,
261    ) -> Result<GraveyardIterator<'a, 'b>, Error> {
262        Self::iter_from(merger, graveyard_object_id, 0).await
263    }
264
265    /// Like "iter", but seeks from a specific (store-id, object-id) tuple.  Example usage:
266    ///
267    ///   let layer_set = graveyard.store().tree().layer_set();
268    ///   let mut merger = layer_set.merger();
269    ///   let mut iter = graveyard.iter_from(&mut merger, (2, 3)).await?;
270    ///
271    async fn iter_from<'a, 'b>(
272        merger: &'a mut Merger<'b, ObjectKey, ObjectValue>,
273        graveyard_object_id: u64,
274        from: u64,
275    ) -> Result<GraveyardIterator<'a, 'b>, Error> {
276        GraveyardIterator::new(
277            graveyard_object_id,
278            merger
279                .query(Query::FullRange(&ObjectKey::graveyard_entry(graveyard_object_id, from)))
280                .await?,
281        )
282        .await
283    }
284}
285
286pub struct GraveyardIterator<'a, 'b> {
287    object_id: u64,
288    iter: MergerIterator<'a, 'b, ObjectKey, ObjectValue>,
289}
290
291/// Contains information about a graveyard entry associated with a particular object or
292/// attribute.
293#[derive(Debug, PartialEq)]
294pub struct GraveyardEntryInfo {
295    object_id: u64,
296    attribute_id: Option<AttributeId>,
297    value: ObjectValue,
298}
299
300impl GraveyardEntryInfo {
301    pub fn object_id(&self) -> u64 {
302        self.object_id
303    }
304
305    pub fn attribute_id(&self) -> Option<AttributeId> {
306        self.attribute_id
307    }
308
309    pub fn value(&self) -> &ObjectValue {
310        &self.value
311    }
312}
313
314impl<'a, 'b> GraveyardIterator<'a, 'b> {
315    async fn new(
316        object_id: u64,
317        iter: MergerIterator<'a, 'b, ObjectKey, ObjectValue>,
318    ) -> Result<GraveyardIterator<'a, 'b>, Error> {
319        let mut iter = GraveyardIterator { object_id, iter };
320        iter.skip_deleted_entries().await?;
321        Ok(iter)
322    }
323
324    async fn skip_deleted_entries(&mut self) -> Result<(), Error> {
325        loop {
326            match self.iter.get() {
327                Some(ItemRef {
328                    key: ObjectKey { object_id, .. },
329                    value: ObjectValue::None,
330                    ..
331                }) if *object_id == self.object_id => {}
332                _ => return Ok(()),
333            }
334            self.iter.advance().await?;
335        }
336    }
337
338    pub fn get(&self) -> Option<GraveyardEntryInfo> {
339        match self.iter.get() {
340            Some(ItemRef {
341                key: ObjectKey { object_id: oid, data: ObjectKeyData::GraveyardEntry { object_id } },
342                value,
343                ..
344            }) if *oid == self.object_id => Some(GraveyardEntryInfo {
345                object_id: *object_id,
346                attribute_id: None,
347                value: value.clone(),
348            }),
349            Some(ItemRef {
350                key:
351                    ObjectKey {
352                        object_id: oid,
353                        data: ObjectKeyData::GraveyardAttributeEntry { object_id, attribute_id },
354                    },
355                value,
356                ..
357            }) if *oid == self.object_id => Some(GraveyardEntryInfo {
358                object_id: *object_id,
359                attribute_id: Some(*attribute_id),
360                value: value.clone(),
361            }),
362            _ => None,
363        }
364    }
365
366    pub async fn advance(&mut self) -> Result<(), Error> {
367        self.iter.advance().await?;
368        self.skip_deleted_entries().await
369    }
370}
371
372#[cfg(test)]
373mod tests {
374    use super::{Graveyard, GraveyardEntryInfo, ObjectStore};
375    use crate::errors::FxfsError;
376    use crate::filesystem::{FxFilesystem, FxFilesystemBuilder};
377    use crate::fsck::fsck;
378    use crate::hooks::Hooks;
379    use crate::object_handle::ObjectHandle;
380    use crate::object_store::data_object_handle::WRITE_ATTR_BATCH_SIZE;
381    use crate::object_store::directory::Directory;
382    use crate::object_store::journal::JournalOptions;
383    use crate::object_store::object_record::{AttributeKey, ObjectValue};
384    use crate::object_store::transaction::{Options, lock_keys};
385    use crate::object_store::volume::root_volume;
386    use crate::object_store::{
387        AttributeId, HandleOptions, LockKey, Mutation, NewChildStoreOptions, ObjectKey, Timestamp,
388    };
389    use assert_matches::assert_matches;
390    use storage_device::DeviceHolder;
391    use storage_device::fake_device::FakeDevice;
392
393    const TEST_DEVICE_BLOCK_SIZE: u32 = 512;
394
395    #[fuchsia::test]
396    async fn test_graveyard() {
397        let device = DeviceHolder::new(FakeDevice::new(8192, TEST_DEVICE_BLOCK_SIZE));
398        let fs = FxFilesystem::new_empty(device).await.expect("new_empty failed");
399        let root_store = fs.root_store();
400
401        assert_eq!(root_store.graveyard_count(), 0);
402
403        let mut transaction = fs
404            .root_store()
405            .new_transaction(lock_keys![], Options::default())
406            .await
407            .expect("new_transaction failed");
408        let handle1 = ObjectStore::create_object(
409            &root_store,
410            &mut transaction,
411            HandleOptions::default(),
412            None,
413        )
414        .await
415        .expect("create_object failed");
416        let handle2 = ObjectStore::create_object(
417            &root_store,
418            &mut transaction,
419            HandleOptions::default(),
420            None,
421        )
422        .await
423        .expect("create_object failed");
424        transaction.commit().await.expect("commit failed");
425        let id1 = handle1.object_id();
426        let id2 = handle2.object_id();
427
428        // Create and add two objects to the graveyard.
429        let mut transaction = fs
430            .root_store()
431            .new_transaction(lock_keys![], Options::default())
432            .await
433            .expect("new_transaction failed");
434
435        root_store.add_to_graveyard(&mut transaction, id1);
436        root_store.add_to_graveyard(&mut transaction, id2);
437        transaction.commit().await.expect("commit failed");
438
439        assert_eq!(root_store.graveyard_count(), 2);
440
441        // Check that we see the objects we added.
442        {
443            let layer_set = root_store.tree().layer_set();
444            let mut merger = layer_set.merger();
445            let mut iter = Graveyard::iter(root_store.graveyard_directory_object_id(), &mut merger)
446                .await
447                .expect("iter failed");
448            assert_matches!(
449                iter.get().expect("missing entry"),
450                GraveyardEntryInfo { object_id, attribute_id: None, value: ObjectValue::Some }
451                if object_id == id1
452            );
453            iter.advance().await.expect("advance failed");
454            assert_matches!(
455                iter.get().expect("missing entry"),
456                GraveyardEntryInfo { object_id, attribute_id: None, value: ObjectValue::Some }
457                if object_id == id2
458            );
459            iter.advance().await.expect("advance failed");
460            assert_eq!(iter.get(), None);
461        }
462
463        // Remove one of the objects.
464        let mut transaction = fs
465            .root_store()
466            .new_transaction(lock_keys![], Options::default())
467            .await
468            .expect("new_transaction failed");
469        root_store.remove_from_graveyard(&mut transaction, id2);
470        transaction.commit().await.expect("commit failed");
471
472        assert_eq!(root_store.graveyard_count(), 1);
473
474        // Check that the graveyard has been updated as expected.
475        let layer_set = root_store.tree().layer_set();
476        let mut merger = layer_set.merger();
477        let mut iter = Graveyard::iter(root_store.graveyard_directory_object_id(), &mut merger)
478            .await
479            .expect("iter failed");
480        assert_matches!(
481            iter.get().expect("missing entry"),
482            GraveyardEntryInfo { object_id, attribute_id: None, value: ObjectValue::Some }
483            if object_id == id1
484        );
485        iter.advance().await.expect("advance failed");
486        assert_eq!(iter.get(), None);
487    }
488
489    #[fuchsia::test]
490    async fn test_graveyard_count_replay() {
491        let device = DeviceHolder::new(FakeDevice::new(8192, TEST_DEVICE_BLOCK_SIZE));
492        let (device, _object_ids) = {
493            let fs = FxFilesystemBuilder::new()
494                .skip_initial_reap(true)
495                .format(true)
496                .open(device)
497                .await
498                .expect("open failed");
499            let root_store = fs.root_store();
500
501            let mut object_ids = Vec::new();
502            let mut transaction = fs
503                .root_store()
504                .new_transaction(lock_keys![], Options::default())
505                .await
506                .expect("new_transaction failed");
507            let handle1 = ObjectStore::create_object(
508                &root_store,
509                &mut transaction,
510                HandleOptions::default(),
511                None,
512            )
513            .await
514            .expect("create_object failed");
515            let handle2 = ObjectStore::create_object(
516                &root_store,
517                &mut transaction,
518                HandleOptions::default(),
519                None,
520            )
521            .await
522            .expect("create_object failed");
523            transaction.commit().await.expect("commit failed");
524            object_ids.push(handle1.object_id());
525            object_ids.push(handle2.object_id());
526
527            // Create and add two objects to the graveyard.
528            let mut transaction = fs
529                .root_store()
530                .new_transaction(lock_keys![], Options::default())
531                .await
532                .expect("new_transaction failed");
533
534            root_store.add_to_graveyard(&mut transaction, object_ids[0]);
535            root_store.add_to_graveyard(&mut transaction, object_ids[1]);
536            transaction.commit().await.expect("commit failed");
537
538            assert_eq!(root_store.graveyard_count(), 2);
539            fs.close().await.expect("close failed");
540            (fs.take_device().await, object_ids)
541        };
542        device.reopen(false);
543        let device = {
544            let fs =
545                FxFilesystemBuilder::new().read_only(true).open(device).await.expect("open failed");
546            let root_store = fs.root_store();
547            // Counter is 0 because initial_reap is not called for read-only mounts.
548            assert_eq!(root_store.graveyard_count(), 0);
549
550            // Now manually run it. This will count and queue (but the reaper isn't running).
551            let count =
552                fs.graveyard().initial_reap(&root_store).await.expect("initial_reap failed");
553            let actual_count = root_store.graveyard_count();
554            assert_eq!(count, 2, "initial_reap found wrong number of items (count={})", count);
555            assert_eq!(
556                actual_count, 2,
557                "graveyard_count returned {} but initial_reap found {}",
558                actual_count, count
559            );
560
561            fs.close().await.expect("close failed");
562            fs.take_device().await
563        };
564        device.reopen(false);
565        {
566            // Now test the full flow where they are automatically reaped.
567            let fs = FxFilesystem::open(device).await.expect("open failed");
568            let root_store = fs.root_store();
569
570            // They might or might not have been reaped yet.
571            // Wait for the reaper to finish.
572            fs.graveyard().wait_for_reap().await;
573
574            // Now the count MUST be 0.
575            assert_eq!(root_store.graveyard_count(), 0);
576            fs.close().await.expect("close failed");
577        }
578    }
579
580    #[fuchsia::test]
581    async fn test_tombstone_attribute() {
582        let device = DeviceHolder::new(FakeDevice::new(8192, TEST_DEVICE_BLOCK_SIZE));
583        let fs = FxFilesystem::new_empty(device).await.expect("new_empty failed");
584        let root_store = fs.root_store();
585        let mut transaction = fs
586            .root_store()
587            .new_transaction(lock_keys![], Options::default())
588            .await
589            .expect("new_transaction failed");
590
591        let handle = ObjectStore::create_object(
592            &root_store,
593            &mut transaction,
594            HandleOptions::default(),
595            None,
596        )
597        .await
598        .expect("failed to create object");
599        transaction.commit().await.expect("commit failed");
600
601        handle
602            .write_attr(AttributeId::TEST_ID, &[0; 8192])
603            .await
604            .expect("failed to write attribute");
605        let object_id = handle.object_id();
606        let mut transaction = handle.new_transaction().await.expect("new_transaction failed");
607        transaction.add(
608            root_store.store_object_id(),
609            Mutation::replace_or_insert_object(
610                ObjectKey::graveyard_attribute_entry(
611                    root_store.graveyard_directory_object_id(),
612                    object_id,
613                    AttributeId::TEST_ID,
614                ),
615                ObjectValue::Some,
616            ),
617        );
618
619        transaction.commit().await.expect("commit failed");
620
621        fs.close().await.expect("failed to close filesystem");
622        let device = fs.take_device().await;
623        device.reopen(false);
624
625        let fs =
626            FxFilesystemBuilder::new().read_only(true).open(device).await.expect("open failed");
627        fsck(fs.clone()).await.expect("fsck failed");
628        fs.close().await.expect("failed to close filesystem");
629        let device = fs.take_device().await;
630        device.reopen(false);
631
632        // On open, the filesystem will call initial_reap which will call queue_tombstone().
633        let fs = FxFilesystem::open(device).await.expect("open failed");
634        // `wait_for_reap` ensures that the Message::Tombstone is actually processed.
635        fs.graveyard().wait_for_reap().await;
636        let root_store = fs.root_store();
637
638        let handle =
639            ObjectStore::open_object(&root_store, object_id, HandleOptions::default(), None)
640                .await
641                .expect("failed to open object");
642
643        assert_eq!(handle.read_attr(AttributeId::TEST_ID).await.expect("read_attr failed"), None);
644        fsck(fs.clone()).await.expect("fsck failed");
645    }
646
647    #[fuchsia::test]
648    async fn test_tombstone_attribute_and_object() {
649        let device = DeviceHolder::new(FakeDevice::new(8192, TEST_DEVICE_BLOCK_SIZE));
650        let fs = FxFilesystem::new_empty(device).await.expect("new_empty failed");
651        let root_store = fs.root_store();
652        let mut transaction = fs
653            .root_store()
654            .new_transaction(lock_keys![], Options::default())
655            .await
656            .expect("new_transaction failed");
657
658        let handle = ObjectStore::create_object(
659            &root_store,
660            &mut transaction,
661            HandleOptions::default(),
662            None,
663        )
664        .await
665        .expect("failed to create object");
666        transaction.commit().await.expect("commit failed");
667
668        const ATTR_1: AttributeId = AttributeId::TEST_ID;
669        const ATTR_2: AttributeId = AttributeId::TEST_ID.next();
670        // With both of these it will test that both their graveyard entries got cleaned up in
671        // trim_or_tombstone() via two different paths.
672        handle.write_attr(ATTR_1, &[0; 8192]).await.expect("failed to write attribute");
673        handle.write_attr(ATTR_2, &[0; 8192]).await.expect("failed to write attribute");
674        let object_id = handle.object_id();
675        let mut transaction = handle.new_transaction().await.expect("new_transaction failed");
676        transaction.add(
677            root_store.store_object_id(),
678            Mutation::replace_or_insert_object(
679                ObjectKey::graveyard_attribute_entry(
680                    root_store.graveyard_directory_object_id(),
681                    object_id,
682                    ATTR_1,
683                ),
684                ObjectValue::Some,
685            ),
686        );
687        transaction.add(
688            root_store.store_object_id(),
689            Mutation::replace_or_insert_object(
690                ObjectKey::graveyard_attribute_entry(
691                    root_store.graveyard_directory_object_id(),
692                    object_id,
693                    ATTR_2,
694                ),
695                ObjectValue::Some,
696            ),
697        );
698        transaction.commit().await.expect("commit failed");
699        let mut transaction = handle.new_transaction().await.expect("new_transaction failed");
700        transaction.add(
701            root_store.store_object_id(),
702            Mutation::replace_or_insert_object(
703                ObjectKey::graveyard_entry(root_store.graveyard_directory_object_id(), object_id),
704                ObjectValue::Some,
705            ),
706        );
707        transaction.commit().await.expect("commit failed");
708
709        fs.close().await.expect("failed to close filesystem");
710        let device = fs.take_device().await;
711        device.reopen(false);
712
713        let fs =
714            FxFilesystemBuilder::new().read_only(true).open(device).await.expect("open failed");
715        fsck(fs.clone()).await.expect("fsck failed");
716        fs.close().await.expect("failed to close filesystem");
717        let device = fs.take_device().await;
718        device.reopen(false);
719
720        // On open, the filesystem will call initial_reap which will call queue_tombstone().
721        let fs = FxFilesystem::open(device).await.expect("open failed");
722        // `wait_for_reap` ensures that the two tombstone messages are processed.
723        fs.graveyard().wait_for_reap().await;
724
725        let root_store = fs.root_store();
726        if let Err(e) =
727            ObjectStore::open_object(&root_store, object_id, HandleOptions::default(), None).await
728        {
729            assert!(FxfsError::NotFound.matches(&e));
730        } else {
731            panic!("open_object succeeded");
732        };
733        fsck(fs.clone()).await.expect("fsck failed");
734    }
735
736    #[fuchsia::test]
737    async fn test_tombstone_large_attribute() {
738        let device = DeviceHolder::new(FakeDevice::new(8192, TEST_DEVICE_BLOCK_SIZE));
739        let fs = FxFilesystem::new_empty(device).await.expect("new_empty failed");
740        let root_store = fs.root_store();
741        let mut transaction = fs
742            .root_store()
743            .new_transaction(lock_keys![], Options::default())
744            .await
745            .expect("new_transaction failed");
746
747        let handle = ObjectStore::create_object(
748            &root_store,
749            &mut transaction,
750            HandleOptions::default(),
751            None,
752        )
753        .await
754        .expect("failed to create object");
755        transaction.commit().await.expect("commit failed");
756
757        let object_id = {
758            let mut transaction = handle.new_transaction().await.expect("new_transaction failed");
759            transaction.add(
760                root_store.store_object_id(),
761                Mutation::replace_or_insert_object(
762                    ObjectKey::graveyard_attribute_entry(
763                        root_store.graveyard_directory_object_id(),
764                        handle.object_id(),
765                        AttributeId::TEST_ID,
766                    ),
767                    ObjectValue::Some,
768                ),
769            );
770
771            // This write should span three transactions. This test mimics the behavior when the
772            // last transaction gets interrupted by a filesystem.close().
773            handle
774                .write_new_attr_in_batches(
775                    &mut transaction,
776                    AttributeId::TEST_ID,
777                    &vec![0; 3 * WRITE_ATTR_BATCH_SIZE],
778                    WRITE_ATTR_BATCH_SIZE,
779                )
780                .await
781                .expect("failed to write attribute");
782
783            handle.object_id()
784            // Drop the transaction to simulate interrupting the attribute creation as well as to
785            // release the transaction locks.
786        };
787
788        fs.close().await.expect("failed to close filesystem");
789        let device = fs.take_device().await;
790        device.reopen(false);
791
792        let fs =
793            FxFilesystemBuilder::new().read_only(true).open(device).await.expect("open failed");
794        fsck(fs.clone()).await.expect("fsck failed");
795        fs.close().await.expect("failed to close filesystem");
796        let device = fs.take_device().await;
797        device.reopen(false);
798
799        // On open, the filesystem will call initial_reap which will call queue_tombstone().
800        let fs = FxFilesystem::open(device).await.expect("open failed");
801        // `wait_for_reap` ensures that the two tombstone messages are processed.
802        fs.graveyard().wait_for_reap().await;
803
804        let root_store = fs.root_store();
805
806        let handle =
807            ObjectStore::open_object(&root_store, object_id, HandleOptions::default(), None)
808                .await
809                .expect("failed to open object");
810
811        assert_eq!(handle.read_attr(AttributeId::TEST_ID).await.expect("read_attr failed"), None);
812        fsck(fs.clone()).await.expect("fsck failed");
813    }
814
815    async fn populate_store(store: &ObjectStore, graveyard: bool, object_count: usize) {
816        let now = Timestamp::now();
817        let graveyard_id = store.graveyard_directory_object_id();
818        let batch_size = 100;
819        for batch_start in (0..object_count).step_by(batch_size) {
820            let mut transaction = store
821                .new_transaction(lock_keys![], Options::default())
822                .await
823                .expect("new_transaction failed");
824            let count = std::cmp::min(batch_size, object_count - batch_start);
825            for i in 0..count {
826                let oid = 1000 + (batch_start + i) as u64;
827                if graveyard {
828                    transaction.add(
829                        store.store_object_id(),
830                        Mutation::insert_object(
831                            ObjectKey::graveyard_entry(graveyard_id, oid),
832                            ObjectValue::Some,
833                        ),
834                    );
835                }
836                transaction.add(
837                    store.store_object_id(),
838                    Mutation::insert_object(
839                        ObjectKey::object(oid),
840                        ObjectValue::file(1, 0, now, now, now, now, None, None),
841                    ),
842                );
843                transaction.add(
844                    store.store_object_id(),
845                    Mutation::insert_object(
846                        ObjectKey::attribute(oid, AttributeId::DATA, AttributeKey::Attribute),
847                        ObjectValue::attribute(0, false),
848                    ),
849                );
850            }
851            transaction.commit().await.expect("commit failed");
852        }
853        store.flush().await.expect("flush failed");
854    }
855
856    // This test verifies that graveyard operations check for journal space before committing
857    // transactions (skip_journal_checks: false). If compactions are paused, graveyard operations
858    // will throttle when journal space runs low and wait for compactions to free space (invoking
859    // the check space callback), rather than borrowing unbounded metadata space.
860    #[fuchsia::test]
861    async fn test_graveyard_borrowed_metadata_space_fails_compaction() {
862        let reclaim_size = 65536;
863        // 200,000 blocks of 512 bytes = 100 MiB device.
864        let device = DeviceHolder::new(FakeDevice::new(200000, TEST_DEVICE_BLOCK_SIZE));
865        let (mut hooks, fs_hooks) = Hooks::new();
866        let fs = FxFilesystemBuilder::new()
867            .hooks(fs_hooks)
868            .journal_options(JournalOptions { reclaim_size, ..Default::default() })
869            .format(true)
870            .open(device)
871            .await
872            .expect("open failed");
873
874        let root_volume = root_volume(fs.clone()).await.expect("root_volume failed");
875        let num_stores = 9;
876        let objects_per_store = 2000;
877        let mut stores = Vec::with_capacity(num_stores);
878
879        for s in 0..num_stores {
880            let store = root_volume
881                .new_volume(&format!("test_{s}"), NewChildStoreOptions::default())
882                .await
883                .expect("new_volume failed");
884            stores.push(store);
885        }
886
887        // Fill stores 0..n-1 with objects and graveyard markers for them.
888        for store in &stores[0..num_stores - 1] {
889            populate_store(store, true, objects_per_store).await;
890        }
891
892        // Last store is populated with surviving objects.
893        let store_last = &stores[num_stores - 1];
894        populate_store(store_last, false, objects_per_store).await;
895
896        // Force a compaction so the journal is empty before compactions are paused.
897        fs.journal().force_compact().await.expect("force_compact failed");
898
899        // Pause compactions so the graveyard would accumulate journal usage past the reclaim limit
900        // if compactions remain paused.
901        fs.journal().pause_compactions().await;
902
903        // Set a callback which will automatically unblock the journal and resume compactions when
904        // waiting for journal space. The idea is that we don't want background compactions to run
905        // while we are flushing the graveyard, but if the graveyard would block on the journal, we
906        // want to unblock it.
907        let journal_clone = fs.journal().clone();
908        hooks.set_waiting_for_journal_space(move || {
909            journal_clone.resume_compactions();
910        });
911
912        for store in &stores[0..num_stores - 1] {
913            let store_id = store.store_object_id();
914            for oid in 1000..1000 + objects_per_store as u64 {
915                fs.graveyard().queue_tombstone_object(store_id, oid);
916            }
917        }
918        fs.graveyard().flush().await;
919
920        // Perform a transaction to store_last to trigger compaction.  If the graveyard properly
921        // blocked on journal space, that should have triggered a blocking compaction and there
922        // should be enough space.  If the graveyard did not ever block on journal space, this will
923        // fail because there will be too much borrowed metadata space.
924        let root_dir_last = Directory::open(store_last, store_last.root_directory_object_id())
925            .await
926            .expect("open root_dir_last failed");
927        let mut transaction = store_last
928            .new_transaction(
929                lock_keys![LockKey::object(
930                    store_last.store_object_id(),
931                    root_dir_last.object_id()
932                )],
933                Options::default(),
934            )
935            .await
936            .expect("new_transaction failed");
937        root_dir_last.create_child_file(&mut transaction, "trigger").await.expect("create failed");
938        transaction.commit().await.expect("commit failed");
939
940        fs.close().await.expect("close failed");
941    }
942}