From e7c1d57fd52eb164a6acb10e73d1ae879e334098 Mon Sep 17 00:00:00 2001 From: leipeng Date: Sun, 13 Sep 2026 00:03:28 +0800 Subject: [PATCH 01/58] Disable trial object downloads in debug builds without private sources --- Makefile | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/Makefile b/Makefile index 20649fb0f..c90ef355a 100644 --- a/Makefile +++ b/Makefile @@ -451,6 +451,9 @@ ifndef WITH_TOPLING_ROCKS # default 1 WITH_TOPLING_ROCKS := 1 endif +ifeq ($(filter 0,${DEBUG_LEVEL})$(wildcard sideplugin/topling-rocks/src/table/top_patent_algo.cc),) + override WITH_TOPLING_ROCKS := 0 +endif ifeq (${WITH_TOPLING_ROCKS},1) ifneq (,$(wildcard sideplugin/topling-rocks)) @@ -493,6 +496,9 @@ endif # allow override by env or cmd line WITH_CSPP_MEMTABLE ?= 1 +ifeq ($(filter 0,${DEBUG_LEVEL})$(wildcard sideplugin/cspp-memtable/cspp_memtable.cc),) + override WITH_CSPP_MEMTABLE := 0 +endif ifeq (${WITH_CSPP_MEMTABLE}${WITH_TOPLING_ROCKS},10) $(error "When WITH_CSPP_MEMTABLE is 1, WITH_TOPLING_ROCKS must be 1 also") From c06699c1ca3c8705cfba4ff633939df283fd526b Mon Sep 17 00:00:00 2001 From: leipeng Date: Sun, 13 Sep 2026 20:55:34 +0800 Subject: [PATCH 02/58] Route Topling non-concurrent inserts through caller hint. Borrow NeedsUserKeyCompareInGet so Topling pins the writer token on the caller's hint and skips per-prefix insert_hints_. --- db/memtable.cc | 10 +++++++++- 1 file changed, 9 insertions(+), 1 deletion(-) diff --git a/db/memtable.cc b/db/memtable.cc index 6c8bbfab6..8753d014d 100644 --- a/db/memtable.cc +++ b/db/memtable.cc @@ -784,8 +784,8 @@ Status MemTable::Add(SequenceNumber s, ValueType type, size_t encoded_len = MemTableRep::EncodeKeyValueSize(key, real_value); if (!allow_concurrent) { - // Extract prefix for insert with hint. if (insert_with_hint_prefix_extractor_ != nullptr && + needs_user_key_cmp_in_get_ && // disable per-prefix hint for Topling insert_with_hint_prefix_extractor_->InDomain(key)) { Slice prefix = insert_with_hint_prefix_extractor_->Transform(key); hint = &insert_hints_[prefix]; // overwrite hint? @@ -793,6 +793,14 @@ Status MemTable::Add(SequenceNumber s, ValueType type, if (UNLIKELY(!res)) { return Status::TryAgain("key+seq exists"); } + } else if (hint && !needs_user_key_cmp_in_get_) { + // Borrow NeedsUserKeyCompareInGet: false on Topling (hint pins the + // writer token; pair with FinishHint), true on upstream RocksDB + // (ignore caller hint except the prefix-extractor path above). + bool res = table->InsertKeyValueWithHint(tag, key, value, hint); + if (UNLIKELY(!res)) { + return Status::TryAgain("key+seq exists"); + } } else { bool res = table->InsertKeyValue(tag, key, value); if (UNLIKELY(!res)) { From 060a317e3da2d1c8535feba740e361378e5ea832 Mon Sep 17 00:00:00 2001 From: leipeng Date: Sun, 13 Sep 2026 21:09:09 +0800 Subject: [PATCH 03/58] Add hint and concurrent-insert switches to memtablerep_bench. fillseq pins one hint through FinishHint for Topling tokens and concurrent heap splices; SkipList sequential splices are arena-owned. --use_concurrent_insert selects the Concurrent insert APIs independently of hint. --- memtable/memtablerep_bench.cc | 42 +++++++++++++++++++++++++++++++++-- 1 file changed, 40 insertions(+), 2 deletions(-) diff --git a/memtable/memtablerep_bench.cc b/memtable/memtablerep_bench.cc index 4f0c797b4..72680b2a9 100644 --- a/memtable/memtablerep_bench.cc +++ b/memtable/memtablerep_bench.cc @@ -94,6 +94,11 @@ DEFINE_bool(if_log_bucket_dist_when_flash, true, DEFINE_bool(enable_zero_copy, false, "enable zero copy"); +bool g_use_hint = false; +DEFINE_bool(use_concurrent_insert, false, + "Use Insert*Concurrently / Insert*WithHintConcurrently " + "(independent of whether hint is used)"); + DEFINE_bool(reverse, false, "readseq/scan in reverse order"); DEFINE_int32( @@ -252,7 +257,17 @@ class FillBenchmarkThread : public BenchmarkThread { uint64_t tag = ++(*sequence_); Slice ukey(key_buf, sizeof(key_buf)); Slice value = generator_.Generate(FLAGS_item_size); - table_->InsertKeyValueConcurrently(tag, ukey, value); + if (g_use_hint) { + if (FLAGS_use_concurrent_insert) + table_->InsertKeyValueWithHintConcurrently(tag, ukey, value, &hint_); + else + table_->InsertKeyValueWithHint(tag, ukey, value, &hint_); + } else { + if (FLAGS_use_concurrent_insert) + table_->InsertKeyValueConcurrently(tag, ukey, value); + else + table_->InsertKeyValue(tag, ukey, value); + } *bytes_written_ += internal_key_size + FLAGS_item_size + 1; } else { @@ -276,7 +291,17 @@ class FillBenchmarkThread : public BenchmarkThread { memcpy(p, bytes.data(), FLAGS_item_size); p += FLAGS_item_size; assert(p == buf + encoded_len); - table_->Insert(handle); + if (g_use_hint) { + if (FLAGS_use_concurrent_insert) + table_->InsertWithHintConcurrently(handle, &hint_); + else + table_->InsertWithHint(handle, &hint_); + } else { + if (FLAGS_use_concurrent_insert) + table_->InsertConcurrently(handle); + else + table_->Insert(handle); + } *bytes_written_ += encoded_len; } @@ -284,7 +309,18 @@ class FillBenchmarkThread : public BenchmarkThread { for (unsigned int i = 0; i < num_ops_; ++i) { FillOne(); } + // SkipList sequential InsertWithHint stores an arena splice; the + // default FinishHint is delete[] and is only valid for heap splices + // (concurrent InsertWithHint) or Topling token parking. + if (hint_ != nullptr && + (g_is_topling_memtab || FLAGS_use_concurrent_insert)) { + table_->FinishHint(hint_); + } + hint_ = nullptr; } + + protected: + void* hint_ = nullptr; }; class ConcurrentFillBenchmarkThread : public FillBenchmarkThread { @@ -702,6 +738,7 @@ int main(int argc, char** argv) { name = ROCKSDB_NAMESPACE::Slice(benchmarks, sep - benchmarks); benchmarks = sep + 1; } + g_use_hint = false; std::unique_ptr benchmark; if (name == ROCKSDB_NAMESPACE::Slice("fillseq")) { memtablerep.reset(createMemtableRep()); @@ -709,6 +746,7 @@ int main(int argc, char** argv) { &rng, ROCKSDB_NAMESPACE::SEQUENTIAL, FLAGS_num_operations)); benchmark.reset(new ROCKSDB_NAMESPACE::FillBenchmark( memtablerep.get(), key_gen.get(), &sequence)); + g_use_hint = true; } else if (name == ROCKSDB_NAMESPACE::Slice("fillrandom")) { memtablerep.reset(createMemtableRep()); key_gen.reset(new ROCKSDB_NAMESPACE::KeyGenerator( From bee2a6537dbf02592f58d117d63252faa4576a8e Mon Sep 17 00:00:00 2001 From: leipeng Date: Sun, 13 Sep 2026 23:36:58 +0800 Subject: [PATCH 04/58] Add a CSPP twin of the ToplingDB fillseq CI suite The OffsetSkipList fillseq pass stays. CSPP gets the same workload, sampling, and pages sections so the two memtables can be compared on equal footing. --- .github/scripts/bench_logs_to_pages.py | 114 +++++++++++++++++++++- .github/scripts/bench_pages_common.py | 18 ++++ .github/scripts/test_bench_rss_series.py | 97 ++++++++++++++++++ .github/workflows/db_bench-avx512-run.yml | 97 +++++++++--------- .github/workflows/db_bench-run.yml | 106 +++++++++++--------- 5 files changed, 337 insertions(+), 95 deletions(-) diff --git a/.github/scripts/bench_logs_to_pages.py b/.github/scripts/bench_logs_to_pages.py index 235294797..9b9aa9618 100644 --- a/.github/scripts/bench_logs_to_pages.py +++ b/.github/scripts/bench_logs_to_pages.py @@ -60,6 +60,7 @@ YAML_USED_NAMES = ( "db_bench-fillrandom.yaml", "db_bench-fillseq.yaml", + "db_bench-fillseq-cspp.yaml", ) ENGINE_LABELS = { "zipkeyonly": "ToplingDB zipkeyonly", @@ -137,7 +138,7 @@ def format_iec(num_bytes: int) -> str: return f"{n:.1f}{units[idx]}" -SHM_WORKLOADS = ("fillrandom", "fillseq") +SHM_WORKLOADS = ("fillrandom", "fillseq", "fillseq-cspp") SHM_WORKLOAD_LABELS = SHM_SUITE_LABELS @@ -160,7 +161,14 @@ def load_shm_usages(eng_dir: Path) -> Dict[str, Optional[Dict[str, int]]]: return out -RSS_WORKLOADS = ("fillrandom", "fillseq", "fillrandom-omit", "fillseq-omit") +RSS_WORKLOADS = ( + "fillrandom", + "fillseq", + "fillrandom-omit", + "fillseq-omit", + "fillseq-cspp", + "fillseq-cspp-omit", +) def parse_rss_usage(text: str) -> Optional[int]: @@ -240,6 +248,10 @@ def _bytes(eng: str, wl: str, key: str) -> Optional[int]: rows_html = [] for wl in SHM_WORKLOADS: + if wl == "fillseq-cspp" and all( + _bytes(e, wl, "allocated_bytes") is None for e in ENGINES + ): + continue cells = [f"{html.escape(SHM_WORKLOAD_LABELS.get(wl, wl))}"] for e in ENGINES: b = _bytes(e, wl, "allocated_bytes") @@ -633,6 +645,26 @@ def build_db_bench_compare( LAZY_ENGINES = ("zipkeyonly", "zipkeyvalue", "rocksdb-v8.10") +def _has_topling_fillseq_cspp(engines: Dict[str, Any]) -> bool: + return any( + bool((engines.get(e) or {}).get("db_bench_fillseq_cspp")) + for e in TOPLING_ENGINES + ) + + +def _db_bench_by_engine( + engines: Dict[str, Any], + *, + topling_key: str, + rocks_key: str = "db_bench", +) -> Dict[str, List[Dict[str, str]]]: + out: Dict[str, List[Dict[str, str]]] = {} + for e in ENGINES: + key = topling_key if e in TOPLING_ENGINES else rocks_key + out[e] = (engines.get(e) or {}).get(key) or [] + return out + + def _hl(text: str, kind: str) -> str: """Color a short phrase: kind is 'faster' (green) or 'slower' (red).""" return f'{html.escape(text)}' @@ -860,13 +892,22 @@ def _load_engine_logs(log_root: Path) -> Dict[str, Dict[str, Any]]: ) omit_fr_rows: List[Dict[str, str]] = [] omit_fs_rows: List[Dict[str, str]] = [] + omit_fs_cspp_rows: List[Dict[str, str]] = [] + fs_cspp_path = eng_dir / "db_bench-fillseq-cspp.log" + fs_cspp_rows: List[Dict[str, str]] = [] + if fs_cspp_path.is_file(): + fs_cspp_rows = parse_db_bench( + fs_cspp_path.read_text(encoding="utf-8", errors="replace") + ) if eng == "rocksdb-v8.10": # Reuse readseq×3 from the main fill* suites (no separate omit/scan pass). omit_fr_rows = _readseq_rows(fr_rows) omit_fs_rows = _readseq_rows(db_rows) + omit_fs_cspp_rows = omit_fs_rows else: omit_fr = eng_dir / "db_bench-fillrandom-omit.log" omit_fs = eng_dir / "db_bench-fillseq-omit.log" + omit_fs_cspp = eng_dir / "db_bench-fillseq-cspp-omit.log" if omit_fr.is_file(): omit_fr_rows = parse_db_bench( omit_fr.read_text(encoding="utf-8", errors="replace") @@ -875,6 +916,10 @@ def _load_engine_logs(log_root: Path) -> Dict[str, Dict[str, Any]]: omit_fs_rows = parse_db_bench( omit_fs.read_text(encoding="utf-8", errors="replace") ) + if omit_fs_cspp.is_file(): + omit_fs_cspp_rows = parse_db_bench( + omit_fs_cspp.read_text(encoding="utf-8", errors="replace") + ) skiplist_rows: List[Dict[str, str]] = [] cspp_rows: List[Dict[str, str]] = [] offset_skiplist_rows: List[Dict[str, str]] = [] @@ -893,8 +938,10 @@ def _load_engine_logs(log_root: Path) -> Dict[str, Dict[str, Any]]: result[eng] = { "db_bench": db_rows, "db_bench_fillrandom": fr_rows, + "db_bench_fillseq_cspp": fs_cspp_rows, "db_bench_omit_fillrandom": omit_fr_rows, "db_bench_omit_fillseq": omit_fs_rows, + "db_bench_omit_fillseq_cspp": omit_fs_cspp_rows, "memtablerep_skiplist": skiplist_rows, "memtablerep_cspp": cspp_rows, "memtablerep_OffsetSkipList": offset_skiplist_rows, @@ -1117,6 +1164,15 @@ def _build_per_engine_details(engines_data: Dict[str, Any]) -> str: detail_parts.append( _table(db_bench_detail_keys, data["db_bench"], db_bench_detail_keys) ) + if data.get("db_bench_fillseq_cspp"): + detail_parts.append("

db_bench (fillseq suite, CSPP)

") + detail_parts.append( + _table( + db_bench_detail_keys, + data["db_bench_fillseq_cspp"], + db_bench_detail_keys, + ) + ) if eng in TOPLING_ENGINES: if data.get("db_bench_omit_fillrandom"): detail_parts.append( @@ -1140,6 +1196,17 @@ def _build_per_engine_details(engines_data: Dict[str, Any]) -> str: db_bench_detail_keys, ) ) + if data.get("db_bench_omit_fillseq_cspp"): + detail_parts.append( + "

db_bench omit lazy-load (fillseq CSPP DB)

" + ) + detail_parts.append( + _table( + db_bench_detail_keys, + data["db_bench_omit_fillseq_cspp"], + db_bench_detail_keys, + ) + ) if data.get("memtablerep_skiplist"): detail_parts.append("

memtablerep_bench (skiplist)

") detail_parts.append( @@ -1208,22 +1275,30 @@ def emit(args: argparse.Namespace) -> None: "db_bench-fillrandom.log", "db_bench-fillrandom-omit.log", "db_bench-fillseq-omit.log", + "db_bench-fillseq-cspp.log", + "db_bench-fillseq-cspp-omit.log", "memtablerep_bench-skiplist.log", "memtablerep_bench-cspp.log", "memtablerep_bench-OffsetSkipList.log", "shm_usage.txt", "shm_usage-fillrandom.txt", "shm_usage-fillseq.txt", + "shm_usage-fillseq-cspp.txt", "rss_usage-fillrandom.txt", "rss_usage-fillseq.txt", "rss_usage-fillrandom-omit.txt", "rss_usage-fillseq-omit.txt", + "rss_usage-fillseq-cspp.txt", + "rss_usage-fillseq-cspp-omit.txt", "statm_series-fillrandom.txt", "statm_series-fillseq.txt", + "statm_series-fillseq-cspp.txt", "time-fillrandom.txt", "time-fillseq.txt", "time-fillrandom-omit.txt", "time-fillseq-omit.txt", + "time-fillseq-cspp.txt", + "time-fillseq-cspp-omit.txt", "bench_settings.txt", *YAML_USED_NAMES, "engine-meta.json", @@ -1360,12 +1435,18 @@ def emit(args: argparse.Namespace) -> None: "db_bench_fillrandom": engines_data.get(eng, {}).get( "db_bench_fillrandom", [] ), + "db_bench_fillseq_cspp": engines_data.get(eng, {}).get( + "db_bench_fillseq_cspp", [] + ), "db_bench_omit_fillrandom": engines_data.get(eng, {}).get( "db_bench_omit_fillrandom", [] ), "db_bench_omit_fillseq": engines_data.get(eng, {}).get( "db_bench_omit_fillseq", [] ), + "db_bench_omit_fillseq_cspp": engines_data.get(eng, {}).get( + "db_bench_omit_fillseq_cspp", [] + ), "shm_usage": engines_data.get(eng, {}).get("shm_usage") or {wl: None for wl in SHM_WORKLOADS}, "rss_usage": rss_by_eng.get(eng) @@ -1457,12 +1538,17 @@ def _render_latest_section( eng_rss_raw = engines.get(e, {}).get("rss_usage") or {} rss_data[e] = {wl: v for wl, v in eng_rss_raw.items()} if e in ROCKSDB_ENGINES: - for src, dst in (("fillrandom", "fillrandom-omit"), ("fillseq", "fillseq-omit")): + for src, dst in ( + ("fillrandom", "fillrandom-omit"), + ("fillseq", "fillseq-omit"), + ("fillseq-cspp", "fillseq-cspp-omit"), + ): if rss_data[e].get(dst) is None and rss_data[e].get(src) is not None: rss_data[e][dst] = rss_data[e][src] if ( rss_data[e].get("fillrandom-omit") is not None or rss_data[e].get("fillseq-omit") is not None + or rss_data[e].get("fillseq-cspp-omit") is not None ): rss_derived_engines.add(e) if pages_root is not None: @@ -1484,6 +1570,26 @@ def _render_latest_section( omit_fs_table = build_lazy_load_compare( {e: engines.get(e, {}).get("db_bench_omit_fillseq") or [] for e in LAZY_ENGINES} ) + fs_cspp_compare = "" + omit_fs_cspp_block = "" + if _has_topling_fillseq_cspp(engines): + fs_cspp_compare = ( + "

Comparison: db_bench fillseq suite (CSPP) (perf)

\n" + '

Same as fillrandom, except ToplingDB fillseq uses CSPP. ' + "RocksDB fillseq benefits from shortcuts: trivial_move on " + "non-overlapping SSTs; refit level skips zstd on L6: faster, " + "larger size. Seqno-zeroing compact still runs.

\n" + f"{build_db_bench_compare(_db_bench_by_engine(engines, topling_key='db_bench_fillseq_cspp'))}" + ) + omit_fs_cspp_block = ( + "

scan-omit-value on data from fillseq (CSPP)

\n" + + build_lazy_load_compare( + { + e: engines.get(e, {}).get("db_bench_omit_fillseq_cspp") or [] + for e in LAZY_ENGINES + } + ) + ) t_eng = engines.get("zipkeyonly") or {} r_eng = engines.get("rocksdb-v8.10") or {} @@ -1533,12 +1639,14 @@ def _render_latest_section(

Comparison: db_bench fillseq suite (perf)

Same as fillrandom, except ToplingDB fillseq uses OffsetSkipList (fillrandom still uses CSPP). RocksDB fillseq benefits from shortcuts: trivial_move on non-overlapping SSTs; refit level skips zstd on L6: faster, larger size. Seqno-zeroing compact still runs.

{db_compare_fs} + {fs_cspp_compare}

Lazy load demo (scan; RocksDB v8.10 baseline)

zipkey* needs an extra omit pass: scan_omit_key/value enables lazy value load (no real value load). RocksDB has no lazy load, so the baseline is readseq×3 already present in the main fill* suite (no extra pass). RocksDB nextwithkey cells are =readseq. master omitted here (v8.10 is the stronger RocksDB baseline). {_color_sign()}.

scan-omit-value on data from fillrandom

{omit_fr_table}

scan-omit-value on data from fillseq

{omit_fs_table} + {omit_fs_cspp_block}

memtablerep_bench: OffsetSkipList and CSPP vs skiplist

Focus: {_hl('OffsetSkipList / CSPP (ToplingDB)', 'faster')} vs skiplist. Baseline = RocksDB v8.10 skiplist. {_color_sign()}.

{memtablerep_compare} diff --git a/.github/scripts/bench_pages_common.py b/.github/scripts/bench_pages_common.py index 0504eec03..6d7d74ca8 100644 --- a/.github/scripts/bench_pages_common.py +++ b/.github/scripts/bench_pages_common.py @@ -76,6 +76,7 @@ def stage_window_rss_bytes( SUITE_READRANDOM = ( ("fillrandom", "db_bench_fillrandom", "fillrandom-readrandom"), ("fillseq", "db_bench", "fillseq-readrandom"), + ("fillseq-cspp", "db_bench_fillseq_cspp", "fillseq-cspp-readrandom"), ) @@ -104,6 +105,7 @@ def attach_suite_readrandom_rss( SHM_SUITE_LABELS = { "fillrandom": "fillrandom suite", "fillseq": "fillseq suite", + "fillseq-cspp": "fillseq suite (CSPP)", } RSS_WORKLOAD_ORDER = ( "fillrandom", @@ -112,6 +114,9 @@ def attach_suite_readrandom_rss( "fillseq", "fillseq-readrandom", "fillseq-omit", + "fillseq-cspp", + "fillseq-cspp-readrandom", + "fillseq-cspp-omit", ) RSS_WORKLOAD_LABELS = { "fillrandom": "fillrandom suite peak", @@ -120,6 +125,9 @@ def attach_suite_readrandom_rss( "fillseq-readrandom": "fillseq suite readrandom", "fillrandom-omit": "fillrandom scan-omit-value", "fillseq-omit": "fillseq scan-omit-value", + "fillseq-cspp": "fillseq suite (CSPP) peak", + "fillseq-cspp-readrandom": "fillseq suite (CSPP) readrandom", + "fillseq-cspp-omit": "fillseq scan-omit-value (CSPP)", } RSS_WORKLOAD_TIPS = { "fillrandom-readrandom": ( @@ -128,6 +136,9 @@ def attach_suite_readrandom_rss( "fillseq-readrandom": ( "peak RSS during the readrandom stage of the fillseq suite" ), + "fillseq-cspp-readrandom": ( + "peak RSS during the readrandom stage of the fillseq suite (CSPP)" + ), "fillrandom-omit": ( "restart process with reuse db data of fillrandom, " "scan without access value, benefited by lazy load value (ToplingDB feature)" @@ -136,6 +147,10 @@ def attach_suite_readrandom_rss( "restart process with reuse db data of fillseq, " "scan without access value, benefited by lazy load value (ToplingDB feature)" ), + "fillseq-cspp-omit": ( + "restart process with reuse db data of fillseq (CSPP), " + "scan without access value, benefited by lazy load value (ToplingDB feature)" + ), } @@ -445,6 +460,8 @@ def combine_db_bench_logs(engine_raw: Path) -> None: "db_bench-fillrandom-omit.log", "db_bench.log", "db_bench-fillseq-omit.log", + "db_bench-fillseq-cspp.log", + "db_bench-fillseq-cspp-omit.log", ) chunks = [ (engine_raw / name).read_bytes().rstrip(b"\n") @@ -504,6 +521,7 @@ def build_rss_svg_section( for suite, bench_key in [ ("fillrandom", "db_bench_fillrandom"), ("fillseq", "db_bench"), + ("fillseq-cspp", "db_bench_fillseq_cspp"), ]: series_path = eng_dir / f"statm_series-{suite}.txt" if not series_path.is_file(): diff --git a/.github/scripts/test_bench_rss_series.py b/.github/scripts/test_bench_rss_series.py index 88b3275ac..c8188705e 100755 --- a/.github/scripts/test_bench_rss_series.py +++ b/.github/scripts/test_bench_rss_series.py @@ -301,6 +301,97 @@ def check_pages_contract(mod, variant: str) -> None: assert "dcompact bench →" in home assert "offloads most CPU and memory cost" in home assert "not RocksDB CompactionService" not in home + assert "Comparison: db_bench fillseq suite (CSPP)" not in home + assert "fillseq suite (CSPP)" not in home + assert "db_bench (fillseq suite, CSPP)" not in result_html + + +def check_fillseq_cspp_pages(mod) -> None: + """ToplingDB CSPP fillseq is a full twin of the OffsetSkipList fillseq suite.""" + with tempfile.TemporaryDirectory() as tmp: + tmp_path = Path(tmp) + log_root = tmp_path / "logs" + emit_out = tmp_path / "emit" + site = tmp_path / "site" + _write_min_logs(log_root) + bench_body = ( + _DB_BENCH_LINE + + "readrandom : 1.0 micros/op 1000 ops/sec 1.0 seconds " + "1000 operations; x\n" + ) + for eng in ("zipkeyonly", "zipkeyvalue"): + eng_dir = log_root / eng + (eng_dir / "db_bench-fillseq-omit.log").write_text( + "$ fillseq-omit\n" + "nextwithkey : 1.0 micros/op 1000 ops/sec 1.0 seconds " + "1000 operations; x\n", + encoding="utf-8", + ) + (eng_dir / "db_bench-fillseq-cspp.log").write_text( + "$ fillseq-cspp\n" + bench_body, encoding="utf-8" + ) + (eng_dir / "db_bench-fillseq-cspp-omit.log").write_text( + "$ fillseq-cspp-omit\n" + "nextwithkey : 1.0 micros/op 1000 ops/sec 1.0 seconds " + "1000 operations; x\n", + encoding="utf-8", + ) + (eng_dir / "statm_series-fillseq-cspp.txt").write_text( + _STATM_SERIES, encoding="utf-8" + ) + (eng_dir / "shm_usage-fillseq-cspp.txt").write_text( + "apparent_bytes=1000\nallocated_bytes=2000\n", encoding="utf-8" + ) + (eng_dir / "rss_usage-fillseq-cspp.txt").write_text( + "max_rss_bytes=4096\n", encoding="utf-8" + ) + (eng_dir / "rss_usage-fillseq-cspp-omit.txt").write_text( + "max_rss_bytes=2048\n", encoding="utf-8" + ) + emit_args = argparse.Namespace( + variant="plain", + run_id="cspp-fillseq", + log_root=str(log_root), + engine_meta_root=None, + actions_run_url="", + out=str(emit_out), + ) + mod.emit(emit_args) + run_dirs = list((emit_out / "runs").iterdir()) + assert len(run_dirs) == 1, run_dirs + result_html = (run_dirs[0] / "index.html").read_text(encoding="utf-8") + assert "db_bench (fillseq suite)" in result_html + assert "db_bench (fillseq suite, CSPP)" in result_html + assert "db_bench omit lazy-load (fillseq DB)" in result_html + assert "db_bench omit lazy-load (fillseq CSPP DB)" in result_html + combined = ( + run_dirs[0] / "raw" / "zipkeyonly" / "db_bench-all.log" + ).read_text(encoding="utf-8") + assert combined.index("$ fillseq-omit\n") < combined.index("$ fillseq-cspp\n") + assert combined.index("$ fillseq-cspp\n") < combined.index( + "$ fillseq-cspp-omit\n" + ) + meta = json.loads((emit_out / "run-meta.json").read_text(encoding="utf-8")) + zko_rss = meta["engines"]["zipkeyonly"]["rss_usage"] + assert zko_rss.get("fillseq-readrandom") is not None + assert zko_rss.get("fillseq-cspp-readrandom") is not None + assert zko_rss.get("fillseq-cspp") == 4096 + assert zko_rss.get("fillseq-cspp-omit") == 2048 + mod.merge( + argparse.Namespace( + merge_into=str(site), + from_dir=str(emit_out), + variant="plain", + ) + ) + home = (site / "index.html").read_text(encoding="utf-8") + assert "Comparison: db_bench fillseq suite (perf)" in home + assert "Comparison: db_bench fillseq suite (CSPP) (perf)" in home + assert "scan-omit-value on data from fillseq" in home + assert "scan-omit-value on data from fillseq (CSPP)" in home + assert "fillseq suite (CSPP) peak" in home + assert "fillseq suite (CSPP) readrandom" in home + assert "fillseq scan-omit-value (CSPP)" in home def check_dcompact_home_nav(mod) -> None: @@ -596,6 +687,11 @@ def main() -> int: ("db_bench-fillrandom-omit.log", "$ fillrandom-omit\nomit output\n"), ("db_bench.log", "$ fillseq\nfillseq output\n"), ("db_bench-fillseq-omit.log", "$ fillseq-omit\nomit output\n"), + ("db_bench-fillseq-cspp.log", "$ fillseq-cspp\ncspp output\n"), + ( + "db_bench-fillseq-cspp-omit.log", + "$ fillseq-cspp-omit\ncspp omit output\n", + ), ) for name, content in source_logs: (eng_raw / name).write_text(content, encoding="utf-8") @@ -692,6 +788,7 @@ def main() -> int: assert "TestOS" in html check_readrandom_highlight(mod) check_suite_readrandom_peak(mod) + check_fillseq_cspp_pages(mod) if name == "bench_dcompact_pages": check_dcompact_home_nav(mod) check_dcompact_rss_row_tips(mod) diff --git a/.github/workflows/db_bench-avx512-run.yml b/.github/workflows/db_bench-avx512-run.yml index a11469afd..9bc697ca6 100644 --- a/.github/workflows/db_bench-avx512-run.yml +++ b/.github/workflows/db_bench-avx512-run.yml @@ -206,50 +206,59 @@ jobs: record_shm fillrandom rm -rf "$DB_PATH" - # Pass 2: fillseq — OffsetSkipList; prefix 6 zipkeyonly (keep L6). - prepare_db - yaml_fs="${logdir}/db_bench-fillseq.yaml" - python3 .github/scripts/graft_bench_yaml.py \ - --prefix-level-writers 6 zipkeyonly \ - --target-file-size-base 128M \ - --target-file-size-multiplier 1 \ - --memtable-factory '"${offset_skiplist}"' \ - --out "$yaml_fs" \ - "$yaml" - args=( - -json "$yaml_fs" - -num=100000000 - -key_size=8 - -value_size="${VALUE_SIZE}" - -batch_size=1000 - -benchmarks=fillseq,flush,compact,readseq,readseq,readseq,readrandom - -enable_zero_copy - -progress_reports=false - -compact_target_level=6 - ) - echo '$' "$TOPLING/bin/db_bench" "${args[@]}" >"${logdir}/db_bench.log" - /usr/bin/time -f 'max_rss_kb=%M' -o "${logdir}/time-fillseq.txt" -- \ - "$TOPLING/bin/db_bench" "${args[@]}" >>"${logdir}/db_bench.log" 2>&1 - cat "${logdir}/db_bench.log" - args_omit_fs=( - -json "$yaml_fs" - -num=100000000 - -key_size=8 - -value_size="${VALUE_SIZE}" - -batch_size=1000 - -benchmarks=nextwithkey,nextwithkey,nextwithkey,readseq,readseq,readseq - -scan_omit_key -scan_omit_value - -use_existing_db=1 - -enable_zero_copy - -progress_reports=false - ) - echo '$' "$TOPLING/bin/db_bench" "${args_omit_fs[@]}" >"${logdir}/db_bench-fillseq-omit.log" - /usr/bin/time -f 'max_rss_kb=%M' -o "${logdir}/time-fillseq-omit.txt" -- \ - "$TOPLING/bin/db_bench" "${args_omit_fs[@]}" >>"${logdir}/db_bench-fillseq-omit.log" 2>&1 - cat "${logdir}/db_bench-fillseq-omit.log" - record_rss fillseq - record_rss fillseq-omit - record_shm fillseq + # Pass 2: fillseq — OffsetSkipList then CSPP; prefix 6 zipkeyonly (keep L6). + run_topling_fillseq() { + local factory="$1" + local tag="$2" + local yaml_fs="${logdir}/db_bench-${tag}.yaml" + local main_log="${logdir}/db_bench.log" + [ "$tag" = "fillseq" ] || main_log="${logdir}/db_bench-${tag}.log" + local omit_log="${logdir}/db_bench-${tag}-omit.log" + prepare_db + python3 .github/scripts/graft_bench_yaml.py \ + --prefix-level-writers 6 zipkeyonly \ + --target-file-size-base 128M \ + --target-file-size-multiplier 1 \ + --memtable-factory '"${'"${factory}"'}"' \ + --out "$yaml_fs" \ + "$yaml" + args=( + -json "$yaml_fs" + -num=100000000 + -key_size=8 + -value_size="${VALUE_SIZE}" + -batch_size=1000 + -benchmarks=fillseq,flush,compact,readseq,readseq,readseq,readrandom + -enable_zero_copy + -progress_reports=false + -compact_target_level=6 + ) + echo '$' "$TOPLING/bin/db_bench" "${args[@]}" >"$main_log" + /usr/bin/time -f 'max_rss_kb=%M' -o "${logdir}/time-${tag}.txt" -- \ + "$TOPLING/bin/db_bench" "${args[@]}" >>"$main_log" 2>&1 + cat "$main_log" + args_omit_fs=( + -json "$yaml_fs" + -num=100000000 + -key_size=8 + -value_size="${VALUE_SIZE}" + -batch_size=1000 + -benchmarks=nextwithkey,nextwithkey,nextwithkey,readseq,readseq,readseq + -scan_omit_key -scan_omit_value + -use_existing_db=1 + -enable_zero_copy + -progress_reports=false + ) + echo '$' "$TOPLING/bin/db_bench" "${args_omit_fs[@]}" >"$omit_log" + /usr/bin/time -f 'max_rss_kb=%M' -o "${logdir}/time-${tag}-omit.txt" -- \ + "$TOPLING/bin/db_bench" "${args_omit_fs[@]}" >>"$omit_log" 2>&1 + cat "$omit_log" + record_rss "$tag" + record_rss "${tag}-omit" + record_shm "$tag" + } + run_topling_fillseq offset_skiplist fillseq + run_topling_fillseq cspp fillseq-cspp if [ "$run_memtable" = "1" ]; then mt=( -benchmarks=fillrandom,readrandom diff --git a/.github/workflows/db_bench-run.yml b/.github/workflows/db_bench-run.yml index ccc59c043..fdc6c1cd4 100644 --- a/.github/workflows/db_bench-run.yml +++ b/.github/workflows/db_bench-run.yml @@ -255,54 +255,64 @@ jobs: record_shm fillrandom rm -rf "$DB_PATH" - # Pass 2: fillseq — OffsetSkipList; prefix 6 zipkeyonly (keep L6). - prepare_db - yaml_fs="${logdir}/db_bench-fillseq.yaml" - python3 .github/scripts/graft_bench_yaml.py \ - --prefix-level-writers 6 zipkeyonly \ - --target-file-size-base 128M \ - --target-file-size-multiplier 1 \ - --memtable-factory '"${offset_skiplist}"' \ - --out "$yaml_fs" \ - "$yaml" - args=( - -json "$yaml_fs" - -num="${NUM}" - -key_size=8 - -value_size="${VALUE_SIZE}" - -batch_size=1000 - -benchmarks=fillseq,flush,compact,readseq,readseq,readseq,readrandom - -enable_zero_copy - -progress_reports=false - -report_bench_start_time - -compact_target_level=6 - ) - echo '$' "$TOPLING/bin/db_bench" "${args[@]}" >"${logdir}/db_bench.log" - .github/scripts/run_sample_statm_fdcache.sh "${logdir}/statm_series-fillseq.txt" "${logdir}/time-fillseq.txt" \ - "$TOPLING/bin/db_bench" "${args[@]}" \ - >>"${logdir}/db_bench.log" 2>&1 - cat "${logdir}/db_bench.log" - save_db_log fillseq - args_omit_fs=( - -json "$yaml_fs" - -num="${NUM}" - -key_size=8 - -value_size="${VALUE_SIZE}" - -batch_size=1000 - -benchmarks=nextwithkey,nextwithkey,nextwithkey,readseq,readseq,readseq - -scan_omit_key -scan_omit_value - -use_existing_db=1 - -enable_zero_copy - -progress_reports=false - ) - echo '$' "$TOPLING/bin/db_bench" "${args_omit_fs[@]}" >"${logdir}/db_bench-fillseq-omit.log" - /usr/bin/time -f 'max_rss_kb=%M' -o "${logdir}/time-fillseq-omit.txt" -- \ - "$TOPLING/bin/db_bench" "${args_omit_fs[@]}" >>"${logdir}/db_bench-fillseq-omit.log" 2>&1 - cat "${logdir}/db_bench-fillseq-omit.log" - save_db_log fillseq-omit - record_rss fillseq - record_rss fillseq-omit - record_shm fillseq + # Pass 2: fillseq — OffsetSkipList then CSPP; prefix 6 zipkeyonly (keep L6). + run_topling_fillseq() { + local factory="$1" + local tag="$2" + local yaml_fs="${logdir}/db_bench-${tag}.yaml" + local main_log="${logdir}/db_bench.log" + [ "$tag" = "fillseq" ] || main_log="${logdir}/db_bench-${tag}.log" + local omit_log="${logdir}/db_bench-${tag}-omit.log" + prepare_db + python3 .github/scripts/graft_bench_yaml.py \ + --prefix-level-writers 6 zipkeyonly \ + --target-file-size-base 128M \ + --target-file-size-multiplier 1 \ + --memtable-factory '"${'"${factory}"'}"' \ + --out "$yaml_fs" \ + "$yaml" + args=( + -json "$yaml_fs" + -num="${NUM}" + -key_size=8 + -value_size="${VALUE_SIZE}" + -batch_size=1000 + -benchmarks=fillseq,flush,compact,readseq,readseq,readseq,readrandom + -enable_zero_copy + -progress_reports=false + -report_bench_start_time + -compact_target_level=6 + ) + echo '$' "$TOPLING/bin/db_bench" "${args[@]}" >"$main_log" + .github/scripts/run_sample_statm_fdcache.sh \ + "${logdir}/statm_series-${tag}.txt" "${logdir}/time-${tag}.txt" \ + "$TOPLING/bin/db_bench" "${args[@]}" \ + >>"$main_log" 2>&1 + cat "$main_log" + save_db_log "$tag" + args_omit_fs=( + -json "$yaml_fs" + -num="${NUM}" + -key_size=8 + -value_size="${VALUE_SIZE}" + -batch_size=1000 + -benchmarks=nextwithkey,nextwithkey,nextwithkey,readseq,readseq,readseq + -scan_omit_key -scan_omit_value + -use_existing_db=1 + -enable_zero_copy + -progress_reports=false + ) + echo '$' "$TOPLING/bin/db_bench" "${args_omit_fs[@]}" >"$omit_log" + /usr/bin/time -f 'max_rss_kb=%M' -o "${logdir}/time-${tag}-omit.txt" -- \ + "$TOPLING/bin/db_bench" "${args_omit_fs[@]}" >>"$omit_log" 2>&1 + cat "$omit_log" + save_db_log "${tag}-omit" + record_rss "$tag" + record_rss "${tag}-omit" + record_shm "$tag" + } + run_topling_fillseq offset_skiplist fillseq + run_topling_fillseq cspp fillseq-cspp if [ "$run_memtable" = "1" ]; then mt=( -benchmarks=fillrandom,readrandom From ee4642acaa856133fc6c8ffa4b8dc2f204d77341 Mon Sep 17 00:00:00 2001 From: leipeng Date: Sat, 19 Sep 2026 09:49:35 +0800 Subject: [PATCH 05/58] Skip empty range-del table in ShouldFlushNow Range-del memory is zero when the table is empty; do not touch it on the flush check. Cover the no-range-del path with a unit test. --- db/db_memtable_test.cc | 24 ++++++++++++++++++++++++ db/memtable.cc | 4 +++- 2 files changed, 27 insertions(+), 1 deletion(-) diff --git a/db/db_memtable_test.cc b/db/db_memtable_test.cc index bd17b7c04..833c8156a 100644 --- a/db/db_memtable_test.cc +++ b/db/db_memtable_test.cc @@ -361,6 +361,30 @@ TEST_F(DBMemTableTest, InsertWithHint) { ASSERT_EQ("vvv", Get("NotInPrefixDomain")); } +TEST_F(DBMemTableTest, ShouldFlushNowWithoutRangeDel) { + Options options; + options.memtable_as_log_index = false; + options.write_buffer_size = 64 * 1024; + InternalKeyComparator cmp(BytewiseComparator()); + auto factory = std::make_shared(); + options.memtable_factory = factory; + ImmutableOptions ioptions(options); + WriteBufferManager wb(options.db_write_buffer_size); + MemTable* mem = new MemTable(cmp, ioptions, MutableCFOptions(options), &wb, + kMaxSequenceNumber, 0 /* column_family_id */); + std::string value(128, 'v'); + int i = 0; + for (; i < 2000; ++i) { + ASSERT_OK(mem->Add(i + 1, kTypeValue, "k" + std::to_string(i), value, + nullptr /* kv_prot_info */)); + if (mem->ShouldFlushNow()) { + break; + } + } + ASSERT_LT(i, 2000); + delete mem; +} + TEST_F(DBMemTableTest, ColumnFamilyId) { // Verifies MemTableRepFactory is told the right column family id. Options options; diff --git a/db/memtable.cc b/db/memtable.cc index 8753d014d..a87c97bef 100644 --- a/db/memtable.cc +++ b/db/memtable.cc @@ -214,8 +214,10 @@ bool MemTable::ShouldFlushNow() { // If arena still have room for new block allocation, we can safely say it // shouldn't flush. auto allocated_memory = table_->ApproximateMemoryUsage() + - range_del_table_->ApproximateMemoryUsage() + arena_.MemoryAllocatedBytes(); + if (!is_range_del_table_empty_.load(std::memory_order_relaxed)) { + allocated_memory += range_del_table_->ApproximateMemoryUsage(); + } approximate_memory_usage_.store(allocated_memory, std::memory_order_relaxed); From 77c82cb3e2b2e84b99293952e65086fc6b5e0d43 Mon Sep 17 00:00:00 2001 From: leipeng Date: Sat, 19 Sep 2026 09:52:43 +0800 Subject: [PATCH 06/58] Defer UpdateFlushState on hinted inserts until FinishHint Non-concurrent hinted Adds skip the per-key flush check; the inserter updates flush state once when the hint map is torn down. --- db/memtable.cc | 4 +++- db/memtable.h | 6 +++--- db/write_batch.cc | 5 ++++- 3 files changed, 10 insertions(+), 5 deletions(-) diff --git a/db/memtable.cc b/db/memtable.cc index a87c97bef..79ae0f04b 100644 --- a/db/memtable.cc +++ b/db/memtable.cc @@ -866,7 +866,9 @@ Status MemTable::Add(SequenceNumber s, ValueType type, // TODO(yuzhangyu): support updating newest UDT for when `allow_concurrent` // is true. MaybeUpdateNewestUDT(key); // user key - UpdateFlushState(); + if (!hint || needs_user_key_cmp_in_get_) { + UpdateFlushState(); + } } else { bool res = (hint == nullptr) diff --git a/db/memtable.h b/db/memtable.h index 04549a318..e1a69ac26 100644 --- a/db/memtable.h +++ b/db/memtable.h @@ -235,6 +235,9 @@ class MemTable : public CacheAlignedNewDelete { std::memory_order_relaxed); } + // Updates flush_state_ using ShouldFlushNow() + void UpdateFlushState(); + // Return an iterator that yields the contents of the memtable. // // The caller must ensure that the underlying MemTable remains live @@ -738,9 +741,6 @@ class MemTable : public CacheAlignedNewDelete { terark::minimal_sso<32> newest_udt_; #endif - // Updates flush_state_ using ShouldFlushNow() - void UpdateFlushState(); - void UpdateOldestKeyTime(); diff --git a/db/write_batch.cc b/db/write_batch.cc index 165571a24..5e0423aba 100644 --- a/db/write_batch.cc +++ b/db/write_batch.cc @@ -2319,10 +2319,13 @@ class MemTableInserter : public WriteBatch::Handler { reinterpret_cast(&duplicate_detector_) ->~DuplicateDetector(); } - GetHintMap().for_each([](auto& iter) { + GetHintMap().for_each([this](auto& iter) { // In base MemTableRep, FinishHint do delete [] (char*)(hint). // In ToplingDB CSPP PatriciaTrie, FinishHint idle/release token. iter.first->FinishHint(iter.second); + if (!concurrent_memtable_writes_) { + iter.first->UpdateFlushState(); + } }); delete rebuilding_trx_; } From efe77d0a471e7794e5da66bb372b356031e70916 Mon Sep 17 00:00:00 2001 From: leipeng Date: Sat, 19 Sep 2026 09:59:01 +0800 Subject: [PATCH 07/58] Mark rare bloom and first-seqno checks UNLIKELY in MemTable::Add The filter is usually absent and first_seqno is set after the first key. --- db/memtable.cc | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/db/memtable.cc b/db/memtable.cc index 79ae0f04b..f8e7545bd 100644 --- a/db/memtable.cc +++ b/db/memtable.cc @@ -836,7 +836,7 @@ Status MemTable::Add(SequenceNumber s, ValueType type, largest_seqno_.store(s, std::memory_order_relaxed); } - if (bloom_filter_) { + if (UNLIKELY(nullptr != bloom_filter_)) { #if defined(TOPLINGDB_WITH_TIMESTAMP) size_t ts_sz = GetInternalKeyComparator().user_comparator()->timestamp_size(); Slice key_without_ts = StripTimestampFromUserKey(key, ts_sz); @@ -853,7 +853,7 @@ Status MemTable::Add(SequenceNumber s, ValueType type, // The first sequence number inserted into the memtable assert(first_seqno_ == 0 || s >= first_seqno_); - if (first_seqno_ == 0) { + if (UNLIKELY(first_seqno_.load(std::memory_order_relaxed) == 0)) { first_seqno_.store(s, std::memory_order_relaxed); if (earliest_seqno_ == kMaxSequenceNumber) { @@ -894,7 +894,7 @@ Status MemTable::Add(SequenceNumber s, ValueType type, post_process_info->largest_seqno = s; } - if (bloom_filter_) { + if (UNLIKELY(nullptr != bloom_filter_)) { #if defined(TOPLINGDB_WITH_TIMESTAMP) size_t ts_sz = GetInternalKeyComparator().user_comparator()->timestamp_size(); Slice key_without_ts = StripTimestampFromUserKey(key, ts_sz); From 659c283517b11da3c1c13c84e8d5c22773de6113 Mon Sep 17 00:00:00 2001 From: leipeng Date: Sat, 19 Sep 2026 10:05:31 +0800 Subject: [PATCH 08/58] ci: fold memtablerep write/read us/op into Elapsed time Keep the pages compact by dropping standalone us/op rows and appending those numbers to the Elapsed time cell. --- .github/scripts/bench_logs_to_pages.py | 55 +++++++++++++++++++++++--- 1 file changed, 50 insertions(+), 5 deletions(-) diff --git a/.github/scripts/bench_logs_to_pages.py b/.github/scripts/bench_logs_to_pages.py index 9b9aa9618..e890d1d1c 100644 --- a/.github/scripts/bench_logs_to_pages.py +++ b/.github/scripts/bench_logs_to_pages.py @@ -794,11 +794,51 @@ def _cost_ratio_cell(baseline: Optional[float], subject: Optional[float]) -> str ) _CSPP_METRICS_LOW = ( "Elapsed time", - "write us/op", - "read us/op", ) +def _memtablerep_elapsed_display( + mmap: Dict[str, str], bench: str, elapsed_raw: str +) -> str: + """Append write/read us/op onto the Elapsed time cell.""" + write_us = mmap.get(f"{bench}|write us/op") + read_us = mmap.get(f"{bench}|read us/op") + extras: List[str] = [] + if write_us and read_us: + extras.append(f"write {write_us} us/op") + extras.append(f"read {read_us} us/op") + elif write_us: + extras.append(f"{write_us} us/op") + elif read_us: + extras.append(f"{read_us} us/op") + if not extras: + return elapsed_raw or "—" + base = elapsed_raw or "—" + return f"{base} ({', '.join(extras)})" + + +def _fold_memtablerep_usop(rows: List[Dict[str, str]]) -> List[Dict[str, str]]: + """Drop standalone us/op rows; fold them into Elapsed time.""" + mmap = _metric_map(rows) + out: List[Dict[str, str]] = [] + for row in rows: + metric = row["metric"] + if metric in ("write us/op", "read us/op"): + continue + if metric == "Elapsed time": + out.append( + { + **row, + "value": _memtablerep_elapsed_display( + mmap, row["benchmark"], row["value"] + ), + } + ) + else: + out.append(row) + return out + + def build_memtablerep_compare( cspp_rows: List[Dict[str, str]], skiplist_topling: List[Dict[str, str]], @@ -838,6 +878,11 @@ def build_memtablerep_compare( _metric_number(c_raw), _metric_number(r_raw), ) + if metric == "Elapsed time": + o_raw = _memtablerep_elapsed_display(offset_skiplist, bench, o_raw) + c_raw = _memtablerep_elapsed_display(cspp, bench, c_raw) + t_raw = _memtablerep_elapsed_display(skip_t, bench, t_raw) + r_raw = _memtablerep_elapsed_display(skip_r, bench, r_raw) if metric in _CSPP_METRICS_HIGH: offset_skiplist_ratio = _throughput_ratio_cell(r_n, o_n) cspp_ratio = _throughput_ratio_cell(r_n, c_n) @@ -1212,7 +1257,7 @@ def _build_per_engine_details(engines_data: Dict[str, Any]) -> str: detail_parts.append( _table( ["benchmark", "metric", "value"], - data["memtablerep_skiplist"], + _fold_memtablerep_usop(data["memtablerep_skiplist"]), ["benchmark", "metric", "value"], ) ) @@ -1221,7 +1266,7 @@ def _build_per_engine_details(engines_data: Dict[str, Any]) -> str: detail_parts.append( _table( ["benchmark", "metric", "value"], - data["memtablerep_cspp"], + _fold_memtablerep_usop(data["memtablerep_cspp"]), ["benchmark", "metric", "value"], ) ) @@ -1232,7 +1277,7 @@ def _build_per_engine_details(engines_data: Dict[str, Any]) -> str: detail_parts.append( _table( ["benchmark", "metric", "value"], - data["memtablerep_OffsetSkipList"], + _fold_memtablerep_usop(data["memtablerep_OffsetSkipList"]), ["benchmark", "metric", "value"], ) ) From b425de1a6039350fccf484433a5bb11a789c6728 Mon Sep 17 00:00:00 2001 From: leipeng Date: Sat, 26 Sep 2026 10:42:23 +0800 Subject: [PATCH 09/58] Use ConvertToSST for eligible MemTables on DB close Use ToplingDB's cheap, millisecond-level ConvertToSST on close when all MemTables support it, avoiding WAL replay on the next open. Upstream RocksDB skips flushing WAL-backed MemTables on close because BuildTable is expensive and can cause a long wait. ToplingDB's MemTables support much cheaper, millisecond-level ConvertToSST, but the existing close path does not take advantage of this capability. Without this change, convertible MemTables are still left for WAL replay on the next open, adding avoidable recovery work and startup latency despite the availability of cheap ConvertToSST. --- Makefile | 3 + db/db_impl/db_impl.cc | 3 +- db/db_impl/db_impl.h | 2 + db/db_impl/db_impl_compaction_flush.cc | 32 +++- db/db_memtable_convert_test.cc | 230 +++++++++++++++++++++++++ db/flush_job.cc | 15 +- db/memtable_list.cc | 9 + db/memtable_list.h | 2 + include/rocksdb/options.h | 11 +- src.mk | 1 + 10 files changed, 299 insertions(+), 9 deletions(-) create mode 100644 db/db_memtable_convert_test.cc diff --git a/Makefile b/Makefile index c90ef355a..7be0bc134 100644 --- a/Makefile +++ b/Makefile @@ -2074,6 +2074,9 @@ db_dynamic_level_test: $(OBJ_DIR)/db/db_dynamic_level_test.o $(TEST_LIBRARY) $(L db_flush_test: $(OBJ_DIR)/db/db_flush_test.o $(TEST_LIBRARY) $(LIBRARY) $(AM_LINK) +db_memtable_convert_test: $(OBJ_DIR)/db/db_memtable_convert_test.o $(TEST_LIBRARY) $(LIBRARY) + $(AM_LINK) + db_inplace_update_test: $(OBJ_DIR)/db/db_inplace_update_test.o $(TEST_LIBRARY) $(LIBRARY) $(AM_LINK) diff --git a/db/db_impl/db_impl.cc b/db/db_impl/db_impl.cc index 776518476..a6efbc900 100644 --- a/db/db_impl/db_impl.cc +++ b/db/db_impl/db_impl.cc @@ -619,7 +619,8 @@ void DBImpl::CancelAllBackgroundWork(bool wait) { InstrumentedMutexLock l(&mutex_); if (!shutting_down_.load(std::memory_order_acquire) && - has_unpersisted_data_.load(std::memory_order_relaxed) && + (has_unpersisted_data_.load(std::memory_order_relaxed) || + AllMemtablesSupportConvertToSST()) && !mutable_db_options_.avoid_flush_during_shutdown) { s = DBImpl::FlushAllColumnFamilies(FlushOptions(), FlushReason::kShutDown); s.PermitUncheckedError(); //**TODO: What to do on error? diff --git a/db/db_impl/db_impl.h b/db/db_impl/db_impl.h index 759fa9b07..ad9fece98 100644 --- a/db/db_impl/db_impl.h +++ b/db/db_impl/db_impl.h @@ -1466,6 +1466,8 @@ class DBImpl : public DB { void NotifyOnExternalFileIngested( ColumnFamilyData* cfd, const ExternalSstFileIngestionJob& ingestion_job); + bool AllMemtablesSupportConvertToSST() const; + Status FlushAllColumnFamilies(const FlushOptions& flush_options, FlushReason flush_reason); diff --git a/db/db_impl/db_impl_compaction_flush.cc b/db/db_impl/db_impl_compaction_flush.cc index 2dc47c904..a6e012bce 100644 --- a/db/db_impl/db_impl_compaction_flush.cc +++ b/db/db_impl/db_impl_compaction_flush.cc @@ -1982,6 +1982,18 @@ int DBImpl::Level0StopWriteTrigger(ColumnFamilyHandle* column_family) { ->mutable_cf_options.level0_stop_writes_trigger; } +bool DBImpl::AllMemtablesSupportConvertToSST() const { + mutex_.AssertHeld(); + for (auto* cfd : *versions_->GetColumnFamilySet()) { + if (!cfd->IsDropped() && + (cfd->mem() == nullptr || !cfd->mem()->SupportConvertToSST() || + !cfd->imm()->UnflushedMemtablesSupportConvertToSST())) { + return false; + } + } + return true; +} + Status DBImpl::FlushAllColumnFamilies(const FlushOptions& flush_options, FlushReason flush_reason) { mutex_.AssertHeld(); @@ -3310,6 +3322,23 @@ Status DBImpl::BackgroundFlush(bool* made_progress, JobContext* job_context, } #endif /* !NDEBUG */ *reason = bg_flush_args[0].flush_reason_; + if (status.ok() && + (shutdown_initiated_ || *reason == FlushReason::kShutDown)) { + // Conversion picks one memtable at a time; finish the shutdown flush. + FlushRequest remaining{*reason, {}}; + for (const auto& arg : bg_flush_args) { + auto* cfd = arg.cfd_; + if (!cfd->IsDropped() && cfd->imm()->NumNotFlushed() != 0 && + cfd->imm()->GetEarliestMemTableID() <= arg.max_memtable_id_) { + cfd->imm()->FlushRequested(); + if (cfd->imm()->IsFlushPending()) { + remaining.cfd_to_max_mem_id_to_persist.emplace( + cfd, arg.max_memtable_id_); + } + } + } + SchedulePendingFlush(remaining); + } for (auto& arg : bg_flush_args) { ColumnFamilyData* cfd = arg.cfd_; if (cfd->UnrefAndTryDelete()) { @@ -4370,7 +4399,8 @@ Status DBImpl::WaitForCompact( return s; } } else if (wait_for_compact_options.close_db && - has_unpersisted_data_.load(std::memory_order_relaxed) && + (has_unpersisted_data_.load(std::memory_order_relaxed) || + AllMemtablesSupportConvertToSST()) && !mutable_db_options_.avoid_flush_during_shutdown) { Status s = DBImpl::FlushAllColumnFamilies(FlushOptions(), FlushReason::kShutDown); diff --git a/db/db_memtable_convert_test.cc b/db/db_memtable_convert_test.cc new file mode 100644 index 000000000..d3d20c436 --- /dev/null +++ b/db/db_memtable_convert_test.cc @@ -0,0 +1,230 @@ +// Copyright (c) 2026-present, Topling Inc. +// Flush/Close conversion, independently of crash-safe recovery. + +#include +#include +#include + +#include +#include + +#include "db/db_impl/db_impl.h" +#include "db/db_test_util.h" +#include "port/stack_trace.h" +#include "test_util/sync_point.h" + +namespace ROCKSDB_NAMESPACE { + +class DBMemtableConvertTest + : public DBTestBase, + public ::testing::WithParamInterface> { + public: + DBMemtableConvertTest() + : DBTestBase("db_memtable_convert_test", /*env_do_fsync=*/false) {} + ~DBMemtableConvertTest() override { + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + } + + protected: + Options ConvertOptions() { + Options options = CurrentOptions(); + options.disable_auto_compactions = true; + options.avoid_flush_during_recovery = true; + options.write_buffer_size = 64 << 20; + options.max_write_buffer_number = 8; + options.min_write_buffer_number_to_merge = 4; + options.atomic_flush = std::get<2>(GetParam()); + const bool osl = std::get<0>(GetParam()); + const json params = {{"mem_cap", 16777216}, + {"convert_to_sst", std::get<1>(GetParam())}}; + auto& repo = repo_; + options.memtable_factory = PluginFactorySP::AcquirePlugin( + osl ? "OffsetSkipList" : "CSPPMemTab", params, repo); + repo.Put("default", options.table_factory); + repo.Put("converted", PluginFactorySP::AcquirePlugin( + osl ? "OffsetSkipListTable" : "CSPPMemTabTable", params, repo)); + options.table_factory = PluginFactorySP::AcquirePlugin( + "Dispatch", {{"default", "$default"}}, repo); + DispatcherTableBackPatch(options.table_factory.get(), repo); + return options; + } + + void ObserveConversion(bool fail = false) { + const auto caller = std::this_thread::get_id(); + SyncPoint::GetInstance()->SetCallBack( + "FlushJob::ConvertToSST:Status", [this, caller, fail](void* arg) { + EXPECT_NE(std::this_thread::get_id(), caller); + EXPECT_TRUE(static_cast(arg)->ok()); + ++converts_; + if (fail) { + *static_cast(arg) = Status::IOError("injected convert"); + } + }); + SyncPoint::GetInstance()->SetCallBack( + "FlushJob::WriteLevel0Table:s", [this](void*) { ++builds_; }); + SyncPoint::GetInstance()->EnableProcessing(); + } + + void CheckClose(bool wait_for_compact) { + Options options = ConvertOptions(); + DestroyAndReopen(options); + ASSERT_OK(Put("a", "1")); + ASSERT_OK(dbfull()->TEST_SwitchMemtable()); + ASSERT_OK(Put("b", "2")); + ASSERT_OK(dbfull()->TEST_SwitchMemtable()); + ASSERT_OK(Put("c", "3")); + ObserveConversion(); + if (wait_for_compact) { + WaitForCompactOptions wait; + wait.close_db = true; + ASSERT_OK(db_->WaitForCompact(wait)); + } else { + ASSERT_OK(db_->Close()); + } + Close(); + ASSERT_EQ(converts_.load(), 3); + ASSERT_EQ(builds_.load(), 0); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("a"), "1"); + ASSERT_EQ(Get("b"), "2"); + ASSERT_EQ(Get("c"), "3"); + Close(); + } + + std::atomic converts_{0}; + std::atomic builds_{0}; + SidePluginRepo repo_; +}; + +TEST_P(DBMemtableConvertTest, ManualFlushConverts) { + Options options = ConvertOptions(); + DestroyAndReopen(options); + ASSERT_OK(Put("k", "v")); + ObserveConversion(); + ASSERT_OK(db_->Flush(FlushOptions())); + ASSERT_EQ(converts_.load(), 1); + ASSERT_EQ(builds_.load(), 0); + ASSERT_EQ(Get("k"), "v"); + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("k"), "v"); + Close(); +} + +TEST_P(DBMemtableConvertTest, CloseConvertsAllMemtables) { + CheckClose(false); +} + +TEST_P(DBMemtableConvertTest, WaitForCompactConvertsBeforeClose) { + CheckClose(true); +} + +TEST_P(DBMemtableConvertTest, AvoidCloseFlush) { + Options options = ConvertOptions(); + options.avoid_flush_during_shutdown = true; + DestroyAndReopen(options); + ASSERT_OK(Put("k", "v")); + ObserveConversion(); + ASSERT_OK(db_->Close()); + Close(); + ASSERT_EQ(converts_.load(), 0); + ASSERT_EQ(builds_.load(), 0); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("k"), "v"); + Close(); +} + +TEST_P(DBMemtableConvertTest, MixedColumnFamilies) { + Options options = ConvertOptions(); + Options plain = CurrentOptions(); + const std::vector cf_options = {options, plain}; + DestroyAndReopen(options); + CreateColumnFamilies({"plain"}, plain); + ASSERT_OK(TryReopenWithColumnFamilies({"default", "plain"}, cf_options)); + ASSERT_OK(Put(0, "convert", "v1")); + ASSERT_OK(Put(1, "plain", "v2")); + ObserveConversion(); + Close(); + ASSERT_EQ(converts_.load(), 0); + ASSERT_EQ(builds_.load(), 0); + ASSERT_OK(TryReopenWithColumnFamilies({"default", "plain"}, cf_options)); + ASSERT_EQ(Get(0, "convert"), "v1"); + ASSERT_EQ(Get(1, "plain"), "v2"); + Close(); +} + +TEST_P(DBMemtableConvertTest, UnpersistedDataStillFlushes) { + Options options = ConvertOptions(); + options.memtable_factory = CurrentOptions().memtable_factory; + DestroyAndReopen(options); + WriteOptions write; + write.disableWAL = true; + ASSERT_OK(db_->Put(write, "k", "v")); + ObserveConversion(); + ASSERT_OK(db_->Close()); + Close(); + ASSERT_EQ(converts_.load(), 0); + ASSERT_EQ(builds_.load(), 1); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("k"), "v"); + Close(); +} + +TEST_P(DBMemtableConvertTest, CloseConvertFailureDoesNotBuildTable) { + Options options = ConvertOptions(); + DestroyAndReopen(options); + ASSERT_OK(Put("k", "v")); + ObserveConversion(true); + Close(); + ASSERT_EQ(converts_.load(), 1); + ASSERT_EQ(builds_.load(), 0); + SyncPoint::GetInstance()->DisableProcessing(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("k"), "v"); + Close(); +} + +TEST_P(DBMemtableConvertTest, InFlightFlushThenClose) { + Options options = ConvertOptions(); + options.max_background_flushes = 1; + DestroyAndReopen(options); + ASSERT_OK(Put("head", "1")); + ASSERT_OK(dbfull()->TEST_SwitchMemtable()); + ASSERT_OK(Put("tail", "2")); + test::SleepingBackgroundTask sleeping; + env_->Schedule(&test::SleepingBackgroundTask::DoSleepTask, &sleeping, + Env::Priority::HIGH); + sleeping.WaitUntilSleeping(); + ObserveConversion(); + FlushOptions flush; + flush.wait = false; + ASSERT_OK(db_->Flush(flush)); + SyncPoint::GetInstance()->SetCallBack( + options.atomic_flush ? "DBImpl::AtomicFlushMemTables:AfterScheduleFlush" + : "DBImpl::FlushMemTable:AfterScheduleFlush", + [&](void*) { sleeping.WakeUp(); }); + std::thread closer([&] { Close(); }); + sleeping.WaitUntilDone(); + closer.join(); + ASSERT_EQ(converts_.load(), 2); + ASSERT_EQ(builds_.load(), 0); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("head"), "1"); + ASSERT_EQ(Get("tail"), "2"); + Close(); +} + +INSTANTIATE_TEST_CASE_P( + Formats, DBMemtableConvertTest, + ::testing::Combine(::testing::Bool(), + ::testing::Values("kDumpMem", "kFileMmap"), + ::testing::Bool())); + +} // namespace ROCKSDB_NAMESPACE + +int main(int argc, char** argv) { + ROCKSDB_NAMESPACE::port::InstallStackTraceHandler(); + ::testing::InitGoogleTest(&argc, argv); + return RUN_ALL_TESTS(); +} diff --git a/db/flush_job.cc b/db/flush_job.cc index 1adb729d2..9370b1a32 100644 --- a/db/flush_job.cc +++ b/db/flush_job.cc @@ -980,6 +980,7 @@ Status FlushJob::WriteLevel0Table() { blob_file_additions.push_back(std::move(b)); }; s = memtable->ConvertToSST(&meta_, tboptions); + TEST_SYNC_POINT_CALLBACK("FlushJob::ConvertToSST:Status", &s); if (!s.ok()) { ROCKS_LOG_BUFFER(log_buffer_, "[%s] [JOB %d] Level-0 ConvertToSST #%" PRIu64 ": ApproximateMemoryUsage %" PRIu64 @@ -987,12 +988,16 @@ Status FlushJob::WriteLevel0Table() { cfd_->GetName().c_str(), job_context_->job_id, meta_.fd.GetNumber(), memtable->ApproximateMemoryUsage(), s.ToString().c_str()); - goto UseBuildTable; + // Do not turn a failed close conversion into a full table rebuild. + if (flush_reason_ != FlushReason::kShutDown) { + goto UseBuildTable; + } + } else { + meta_.fd.smallest_seqno = std::min(memtable->GetEarliestSequenceNumber(), + memtable->GetFirstSequenceNumber()); + meta_.fd.largest_seqno = memtable->largest_seqno(); + meta_.marked_for_compaction = true; } - meta_.fd.smallest_seqno = std::min(memtable->GetEarliestSequenceNumber(), - memtable->GetFirstSequenceNumber()); - meta_.fd.largest_seqno = memtable->largest_seqno(); - meta_.marked_for_compaction = true; for (auto* p_iter : memtables) { // memtables is vec of memtab iters std::destroy_at(p_iter); // Attention!!! must! } diff --git a/db/memtable_list.cc b/db/memtable_list.cc index 9eacca84e..c02bc5f9e 100644 --- a/db/memtable_list.cc +++ b/db/memtable_list.cc @@ -688,6 +688,15 @@ size_t MemTableList::ApproximateUnflushedMemTablesMemoryUsage() { return total_size; } +bool MemTableList::UnflushedMemtablesSupportConvertToSST() const { + for (MemTable* m : current_->memlist_) { + if (!m->SupportConvertToSST()) { + return false; + } + } + return true; +} + size_t MemTableList::ApproximateMemoryUsage() { return current_memory_usage_; } size_t MemTableList::MemoryAllocatedBytesExcludingLast() const { diff --git a/db/memtable_list.h b/db/memtable_list.h index 6630046b6..598694b31 100644 --- a/db/memtable_list.h +++ b/db/memtable_list.h @@ -332,6 +332,8 @@ class MemTableList { // the unflushed mem-tables. size_t ApproximateUnflushedMemTablesMemoryUsage(); + bool UnflushedMemtablesSupportConvertToSST() const; + // Returns an estimate of the timestamp of the earliest key. uint64_t ApproximateOldestKeyTime() const; diff --git a/include/rocksdb/options.h b/include/rocksdb/options.h index eb3f17909..1803664ce 100644 --- a/include/rocksdb/options.h +++ b/include/rocksdb/options.h @@ -1254,8 +1254,15 @@ struct DBOptions { bool avoid_flush_during_recovery = false; // By default RocksDB will flush all memtables on DB close if there are - // unpersisted data (i.e. with WAL disabled) The flush can be skip to speedup - // DB close. Unpersisted data WILL BE LOST. + // unpersisted data (i.e. with WAL disabled) or all memtables support ConvertToSST. + // Upstream RocksDB skips flushing WAL-backed memtables on close: + // BuildTable is expensive and can cause a long wait. ConvertToSST is much + // cheaper, so when all memtables support it, we also flush (convert) them on close + // without that long wait and avoid WAL replay on the next open. + // Setting this option to true unconditionally skips all flushes (including + // conversion) during DB close. Data is lost only if there is unpersisted data + // written with WAL disabled; WAL-backed data is recovered by replaying WAL + // on the next open. Skipping conversion does not add a data-loss risk. // // DEFAULT: false // diff --git a/src.mk b/src.mk index bf8959d03..25966e97d 100644 --- a/src.mk +++ b/src.mk @@ -486,6 +486,7 @@ TEST_MAIN_SOURCES = \ db/db_dynamic_level_test.cc \ db/db_encryption_test.cc \ db/db_flush_test.cc \ + db/db_memtable_convert_test.cc \ db/db_readonly_with_timestamp_test.cc \ db/db_with_timestamp_basic_test.cc \ db/import_column_family_test.cc \ From a84c9c6db903a4640cdd3d0f5d7810353f09c6e7 Mon Sep 17 00:00:00 2001 From: leipeng Date: Sat, 26 Sep 2026 11:10:11 +0800 Subject: [PATCH 10/58] Honor small max_open_files limits for cheap DB open Honor small max_open_files values, including 0, so callers can request a cheap DB open without unnecessary SST opens. This matters for short-lived operations that do not read table data. The existing behavior silently raises this limit to 20 and still opens SSTs even when opening them for statistics has been disabled. This defeats the caller's explicit request and violates the principle of least surprise. Without this change, callers cannot avoid unnecessary SST opens. Short-lived DB opens therefore incur avoidable I/O, latency and resource usage, particularly when accessing SSTs is expensive. --- db/db_impl/db_impl.cc | 5 +++-- db/db_impl/db_impl_open.cc | 4 +++- db/db_options_test.cc | 33 +++++++++++++++++++++++++++++++++ include/rocksdb/options.h | 8 ++++++++ 4 files changed, 47 insertions(+), 3 deletions(-) diff --git a/db/db_impl/db_impl.cc b/db/db_impl/db_impl.cc index a6efbc900..48bb09ce7 100644 --- a/db/db_impl/db_impl.cc +++ b/db/db_impl/db_impl.cc @@ -381,9 +381,10 @@ DBImpl::DBImpl(const DBOptions& options, const std::string& dbname, // Reserve ten files or so for other uses and give the rest to TableCache. // Give a large number for setting of "infinite" open files. + // Small limits must not turn the 10-file reserve into a huge cache capacity. const int table_cache_size = (mutable_db_options_.max_open_files == -1) ? TableCache::kInfiniteCapacity - : mutable_db_options_.max_open_files - 10; + : std::max(0, mutable_db_options_.max_open_files - 10); LRUCacheOptions co; co.capacity = table_cache_size; co.num_shard_bits = immutable_db_options_.table_cache_numshardbits; @@ -1518,7 +1519,7 @@ Status DBImpl::SetDBOptions( new_options.delayed_write_rate); table_cache_.get()->SetCapacity(new_options.max_open_files == -1 ? TableCache::kInfiniteCapacity - : new_options.max_open_files - 10); + : std::max(0, new_options.max_open_files - 10)); wal_other_option_changed = mutable_db_options_.wal_bytes_per_sync != new_options.wal_bytes_per_sync; wal_size_option_changed = mutable_db_options_.max_total_wal_size != diff --git a/db/db_impl/db_impl_open.cc b/db/db_impl/db_impl_open.cc index 8c62470ac..9eaff4a5f 100644 --- a/db/db_impl/db_impl_open.cc +++ b/db/db_impl/db_impl_open.cc @@ -63,7 +63,9 @@ DBOptions SanitizeOptions(const std::string& dbname, const DBOptions& src, if (max_max_open_files == -1) { max_max_open_files = 0x400000; } - ClipToRange(&result.max_open_files, 20, max_max_open_files); + // Do not raise small limits: turning 0 into 20 would force SST preloading + // despite the caller's request for a cheap open. + terark::minimize(result.max_open_files, max_max_open_files); TEST_SYNC_POINT_CALLBACK("SanitizeOptions::AfterChangeMaxOpenFiles", &result.max_open_files); } diff --git a/db/db_options_test.cc b/db/db_options_test.cc index 8ccf940ee..411b281dd 100644 --- a/db/db_options_test.cc +++ b/db/db_options_test.cc @@ -876,6 +876,39 @@ TEST_F(DBOptionsTest, MaxOpenFilesChange) { Close(); } +TEST_F(DBOptionsTest, SmallMaxOpenFiles) { + Options options = CurrentOptions(); + options.table_cache_numshardbits = 0; + options.statistics = CreateDBStatistics(); + options.disable_auto_compactions = true; + options.skip_stats_update_on_db_open = true; + Reopen(options); + ASSERT_OK(Put("k", "v")); + ASSERT_OK(Flush()); + Close(); + for (int max_open_files : {0, 1, 9, 10, 11}) { + SCOPED_TRACE(max_open_files); + options.max_open_files = max_open_files; + options.statistics->getAndResetTickerCount(NO_FILE_OPENS); + Reopen(options); + ASSERT_EQ(options.statistics->getTickerCount(NO_FILE_OPENS), 0); + Cache* tc = dbfull()->TEST_table_cache(); + const size_t capacity = std::max(0, max_open_files - 10); + ASSERT_EQ(db_->GetDBOptions().max_open_files, max_open_files); + ASSERT_EQ(tc->GetCapacity(), capacity); + ASSERT_OK(db_->SetDBOptions({{"max_background_jobs", "4"}})); + ASSERT_EQ(tc->GetCapacity(), capacity); + ASSERT_OK(db_->SetDBOptions({{"max_open_files", "1024"}})); + ASSERT_EQ(tc->GetCapacity(), 1014U); + ASSERT_OK(db_->SetDBOptions( + {{"max_open_files", std::to_string(max_open_files)}})); + ASSERT_EQ(tc->GetCapacity(), capacity); + ASSERT_EQ(Get("k"), "v"); + ASSERT_GT(options.statistics->getTickerCount(NO_FILE_OPENS), 0); + Close(); + } +} + TEST_F(DBOptionsTest, SanitizeDelayedWriteRate) { Options options; options.env = CurrentOptions().env; diff --git a/include/rocksdb/options.h b/include/rocksdb/options.h index 1803664ce..24ac6bbb8 100644 --- a/include/rocksdb/options.h +++ b/include/rocksdb/options.h @@ -625,6 +625,14 @@ struct DBOptions { // on target_file_size_base and target_file_size_multiplier for level-based // compaction. For universal-style compaction, you can usually set it to -1. // + // Value 0 disables SST reader caching and eager SST opening during DB::Open(). + // Also set skip_stats_update_on_db_open=true to avoid opening SSTs for stats; + // reads still open SSTs on demand. + // Previously, values below 20 were silently raised to 20, so even 0 caused + // SSTs to be opened during DB::Open(). This defeated the user's intent to + // open without opening any SSTs and violated the principle of least surprise. + // Small nonnegative values are now honored instead of imposing that minimum. + // // A high value or -1 for this option can cause high memory usage. // See BlockBasedTableOptions::cache_usage_options to constrain // memory usage in case of block based table format. From 0ef56e129e822eaeb3195f76f1df1a40c83ecac7 Mon Sep 17 00:00:00 2001 From: leipeng Date: Sat, 26 Sep 2026 11:32:20 +0800 Subject: [PATCH 11/58] Add an opt-in option for MemTable crash-safe recovery Introduce memtable_crash_safe_recover, disabled by default, so users can explicitly choose recovery from crash-safe MemTables. Crash-safe recovery and Flush/Close ConvertToSST serve different purposes. A separate option keeps recovery opt-in without tying it to memtable_as_log_index or changing existing WAL-replay behavior by default. --- include/rocksdb/options.h | 7 +++++++ options/db_options.cc | 7 +++++++ options/db_options.h | 1 + options/options_helper.cc | 1 + options/options_settable_test.cc | 1 + options/options_test.cc | 4 ++++ sideplugin/rockside | 2 +- 7 files changed, 22 insertions(+), 1 deletion(-) diff --git a/include/rocksdb/options.h b/include/rocksdb/options.h index 24ac6bbb8..c27a17b3d 100644 --- a/include/rocksdb/options.h +++ b/include/rocksdb/options.h @@ -933,6 +933,13 @@ struct DBOptions { bool memtable_as_log_index = false; + // Enable crash-safe recovery using mmap MemTables (CSPP / OffsetSkipList) + // and ConvertToSST to reduce WAL replay on the next open. + // Only process crashes (kill -9 / _exit) are covered, not power loss. + // Independent of memtable_as_log_index and Flush/Close ConvertToSST. + // Default: false. Not dynamically changeable through SetDBOptions(). + bool memtable_crash_safe_recover = false; + // If true, each WAL file is probed on DB open to auto-detect its on-disk // format, so recovery works even when memtable_as_log_index was changed // between runs. diff --git a/options/db_options.cc b/options/db_options.cc index dc17320e4..7c14532c8 100644 --- a/options/db_options.cc +++ b/options/db_options.cc @@ -328,6 +328,10 @@ static std::unordered_map {offsetof(struct ImmutableDBOptions, memtable_as_log_index), OptionType::kBoolean, OptionVerificationType::kNormal, OptionTypeFlags::kNone}}, + {"memtable_crash_safe_recover", + {offsetof(struct ImmutableDBOptions, memtable_crash_safe_recover), + OptionType::kBoolean, OptionVerificationType::kNormal, + OptionTypeFlags::kNone}}, {"check_wal_format", {offsetof(struct ImmutableDBOptions, check_wal_format), OptionType::kBoolean, OptionVerificationType::kNormal, @@ -768,6 +772,7 @@ ImmutableDBOptions::ImmutableDBOptions(const DBOptions& options) avoid_unnecessary_blocking_io(options.avoid_unnecessary_blocking_io), persist_stats_to_disk(options.persist_stats_to_disk), memtable_as_log_index(options.memtable_as_log_index), + memtable_crash_safe_recover(options.memtable_crash_safe_recover), check_wal_format(options.check_wal_format), write_dbid_to_manifest(options.write_dbid_to_manifest), log_readahead_size(options.log_readahead_size), @@ -938,6 +943,8 @@ void ImmutableDBOptions::Dump(Logger* log) const { persist_stats_to_disk); ROCKS_LOG_HEADER(log, " Options.memtable_as_log_index: %u", memtable_as_log_index); + ROCKS_LOG_HEADER(log, " Options.memtable_crash_safe_recover: %u", + memtable_crash_safe_recover); ROCKS_LOG_HEADER(log, " Options.check_wal_format: %u", check_wal_format); ROCKS_LOG_HEADER(log, " Options.write_dbid_to_manifest: %d", diff --git a/options/db_options.h b/options/db_options.h index e0c618d63..596d43008 100644 --- a/options/db_options.h +++ b/options/db_options.h @@ -90,6 +90,7 @@ struct ImmutableDBOptions { bool avoid_unnecessary_blocking_io; bool persist_stats_to_disk; bool memtable_as_log_index; + bool memtable_crash_safe_recover; bool check_wal_format; bool write_dbid_to_manifest; size_t log_readahead_size; diff --git a/options/options_helper.cc b/options/options_helper.cc index 61cce16d7..623aca5fc 100644 --- a/options/options_helper.cc +++ b/options/options_helper.cc @@ -118,6 +118,7 @@ DBOptions BuildDBOptions(const ImmutableDBOptions& immutable_db_options, mutable_db_options.stats_persist_period_sec; options.persist_stats_to_disk = immutable_db_options.persist_stats_to_disk; options.memtable_as_log_index = immutable_db_options.memtable_as_log_index; + options.memtable_crash_safe_recover = immutable_db_options.memtable_crash_safe_recover; options.stats_history_buffer_size = mutable_db_options.stats_history_buffer_size; options.advise_random_on_open = immutable_db_options.advise_random_on_open; diff --git a/options/options_settable_test.cc b/options/options_settable_test.cc index 2cd675a16..6bf11d5f9 100644 --- a/options/options_settable_test.cc +++ b/options/options_settable_test.cc @@ -338,6 +338,7 @@ TEST_F(OptionsSettableTest, DBOptionsAllFieldsSettable) { "stats_persist_period_sec=54321;" "persist_stats_to_disk=true;" "memtable_as_log_index=true;" + "memtable_crash_safe_recover=true;" "check_wal_format=true;" "stats_history_buffer_size=14159;" "allow_fallocate=true;" diff --git a/options/options_test.cc b/options/options_test.cc index 450f38c40..35171f73b 100644 --- a/options/options_test.cc +++ b/options/options_test.cc @@ -170,6 +170,7 @@ TEST_F(OptionsTest, GetOptionsFromMapTest) { {"stats_persist_period_sec", "57"}, {"persist_stats_to_disk", "false"}, {"memtable_as_log_index", "false"}, + {"memtable_crash_safe_recover", "false"}, {"check_wal_format", "false"}, {"stats_history_buffer_size", "69"}, {"advise_random_on_open", "true"}, @@ -355,6 +356,7 @@ TEST_F(OptionsTest, GetOptionsFromMapTest) { ASSERT_EQ(new_db_opt.stats_persist_period_sec, 57U); ASSERT_EQ(new_db_opt.persist_stats_to_disk, false); ASSERT_EQ(new_db_opt.memtable_as_log_index, false); + ASSERT_EQ(new_db_opt.memtable_crash_safe_recover, false); ASSERT_EQ(new_db_opt.check_wal_format, false); ASSERT_EQ(new_db_opt.stats_history_buffer_size, 69U); ASSERT_EQ(new_db_opt.advise_random_on_open, true); @@ -2394,6 +2396,7 @@ TEST_F(OptionsOldApiTest, GetOptionsFromMapTest) { {"stats_persist_period_sec", "57"}, {"persist_stats_to_disk", "false"}, {"memtable_as_log_index", "false"}, + {"memtable_crash_safe_recover", "false"}, {"check_wal_format", "false"}, {"stats_history_buffer_size", "69"}, {"advise_random_on_open", "true"}, @@ -2581,6 +2584,7 @@ TEST_F(OptionsOldApiTest, GetOptionsFromMapTest) { ASSERT_EQ(new_db_opt.stats_persist_period_sec, 57U); ASSERT_EQ(new_db_opt.persist_stats_to_disk, false); ASSERT_EQ(new_db_opt.memtable_as_log_index, false); + ASSERT_EQ(new_db_opt.memtable_crash_safe_recover, false); ASSERT_EQ(new_db_opt.check_wal_format, false); ASSERT_EQ(new_db_opt.stats_history_buffer_size, 69U); ASSERT_EQ(new_db_opt.advise_random_on_open, true); diff --git a/sideplugin/rockside b/sideplugin/rockside index c308ced76..d066c938d 160000 --- a/sideplugin/rockside +++ b/sideplugin/rockside @@ -1 +1 @@ -Subproject commit c308ced7614708b9aa2635066d1527346e0e69c8 +Subproject commit d066c938d64b2581407119d74104c10a5d0cc893 From 79af951e0d1e61b87c5025cf745d3ea538028fbe Mon Sep 17 00:00:00 2001 From: leipeng Date: Sat, 26 Sep 2026 11:57:59 +0800 Subject: [PATCH 12/58] Add crash-safe recovery methods to MemTableRepFactory Add methods to MemTableRepFactory for reusing crash-safe file mmap MemTables during recovery. The DB recovery path needs these methods to reuse the on-disk files of MemTable data across different implementations without knowing their file formats. Otherwise it must rely on plugin-specific code or rebuild recoverable data through WAL replay. Factories that do not support crash-safe recovery remain opted out by default. --- include/rocksdb/memtablerep.h | 46 +++++++++++++++++++++++++++++++++++ 1 file changed, 46 insertions(+) diff --git a/include/rocksdb/memtablerep.h b/include/rocksdb/memtablerep.h index 5ddfc6d76..9f2ae0de3 100644 --- a/include/rocksdb/memtablerep.h +++ b/include/rocksdb/memtablerep.h @@ -312,6 +312,24 @@ class MemTableRep : public CacheAlignedNewDelete { virtual Status ConvertToSST(struct FileMetaData*, const struct TableBuilderOptions&); protected: + // Leftover / converted SST visibility: packed tag exists only when + // seq <= min(lookup_seq, max_visible_seq). Used by crash-safe MemTableRep. + static bool VisibleTag(uint64_t tag, uint64_t lookup_seq, + uint64_t max_visible_seq) { + const uint64_t seq = tag >> 8; + const uint64_t effective_max = + lookup_seq < max_visible_seq ? lookup_seq : max_visible_seq; + return seq <= effective_max; + } + + static uint64_t CapFindTag(uint64_t find_tag, uint64_t max_visible_seq) { + const uint64_t seq = find_tag >> 8; + if (seq <= max_visible_seq) { + return find_tag; + } + return (max_visible_seq << 8) | (find_tag & 0xff); + } + // When *key is an internal key concatenated with the value, returns the // user key. virtual Slice UserKey(const char* key) const; @@ -363,6 +381,34 @@ class MemTableRepFactory : public Customizable { // false when if the already exists. // Default: false virtual bool CanHandleDuplicatedKey() const { return false; } + + // Return true if leftover mmap memtables created by this factory can be + // RO-loaded after a crash and ConvertToSST during Recover. + // Default: false + virtual bool SupportCrashSafe() const { return false; } + + // Append leftover crash-safe mmap paths under cf_dir (plus factory chroot). + // Default: no leftovers. + virtual void ListCrashSafeLeftovers(const std::string& /*cf_dir*/, + std::vector* /*leftovers*/) { + } + + // Read-only leftover probe for Recover check. Must not truncate/rename. + // wal_dir is used to verify each WAL fileno can still be opened. + virtual Status ProbeCrashSafeLeftover(const std::string& /*path*/, + const std::string& /*wal_dir*/) const { + return Status::OK(); + } + + // Load leftover, cap visible entries, truncate, ConvertToSST. The caller + // supplies the visibility bound in meta->fd.largest_seqno; preserve it for + // subsequent table readers. + // Default: NotSupported. + virtual Status RecoverCrashSafeMemTableToSST( + const std::string& /*leftover_path*/, struct FileMetaData* /*meta*/, + const struct TableBuilderOptions&) { + return Status::NotSupported("RecoverCrashSafeMemTableToSST"); + } }; // This uses a skip list to store keys. It is the default. From 7ecae6ce7c98b89f52a9bf8edd2b2092fd79fa22 Mon Sep 17 00:00:00 2001 From: leipeng Date: Sat, 26 Sep 2026 13:19:13 +0800 Subject: [PATCH 13/58] Enforce option constraints for MemTable crash-safe recovery Require compatible WAL settings and WAL-backed writes for MemTable crash-safe recovery. Disable this recovery mode for read-only Open and warn that full WAL replay may be very slow. Recovery reuses on-disk MemTable files and relies on WAL data to recover writes beyond the published sequence. Disabling WAL, retaining writes in a process-local WAL buffer, or recycling WAL files would invalidate those assumptions. WAL compression is unsupported both by crash-safe recovery and memtable_as_log_index. Read-only Open cannot run ConvertToSST recovery and install its results. It must replay WAL data instead, which can take much longer with the large MemTables encouraged in crash-safe mode. A warning is necessary even without a configured logger so this potential delay is not silent. --- db/db_impl/db_impl_open.cc | 19 +++++- db/db_impl/db_impl_write.cc | 3 +- db/db_options_test.cc | 114 ++++++++++++++++++++++++++++++++++++ include/rocksdb/options.h | 4 ++ 4 files changed, 138 insertions(+), 2 deletions(-) diff --git a/db/db_impl/db_impl_open.cc b/db/db_impl/db_impl_open.cc index 9eaff4a5f..da378816c 100644 --- a/db/db_impl/db_impl_open.cc +++ b/db/db_impl/db_impl_open.cc @@ -25,6 +25,7 @@ #include "rocksdb/wal_filter.h" #include "test_util/sync_point.h" #include "util/rate_limiter_impl.h" +#include #include "util/string_util.h" #include "util/udt_util.h" @@ -47,9 +48,25 @@ DBOptions SanitizeOptions(const std::string& dbname, const DBOptions& src, result.env = Env::Default(); } - if (result.memtable_as_log_index) { + if (read_only && result.memtable_crash_safe_recover) { + result.memtable_crash_safe_recover = false; + const char* warning = + "memtable_crash_safe_recover is disabled for read-only Open; full WAL " + "replay may be very slow because using much larger MemTables than in " + "typical RocksDB configurations is encouraged in crash-safe mode"; + if (result.info_log) { + ROCKS_LOG_WARN(result.info_log, "%s", warning); + } else { + fprintf(stderr, "%s WARNING: %s\n", terark::StrDateTimeNow(), warning); + } + } + + if (result.memtable_as_log_index || result.memtable_crash_safe_recover) { result.recycle_log_file_num = 0; result.manual_wal_flush = false; + result.wal_compression = kNoCompression; + } + if (result.memtable_as_log_index) { #if !defined(ROCKSDB_UNIT_TEST) // avoid infrequent CFs reference too many WALs when frequent CFs // writing many data diff --git a/db/db_impl/db_impl_write.cc b/db/db_impl/db_impl_write.cc index 4ad662e1f..70e0e5b65 100644 --- a/db/db_impl/db_impl_write.cc +++ b/db/db_impl/db_impl_write.cc @@ -190,7 +190,8 @@ Status DBImpl::WriteImpl(const WriteOptions& write_options, write_options.protection_bytes_per_key == 0 || write_options.protection_bytes_per_key == my_batch->GetProtectionBytesPerKey()); - if (immutable_db_options_.memtable_as_log_index) { + if (immutable_db_options_.memtable_as_log_index || + immutable_db_options_.memtable_crash_safe_recover) { const_cast(write_options.disableWAL) = false; } diff --git a/db/db_options_test.cc b/db/db_options_test.cc index 411b281dd..99fcb5941 100644 --- a/db/db_options_test.cc +++ b/db/db_options_test.cc @@ -80,6 +80,120 @@ class DBOptionsTest : public DBTestBase { } }; +TEST_F(DBOptionsTest, SanitizeManualWalFlushAndAtomicFlush) { + DBOptions src; + src.memtable_crash_safe_recover = true; + src.manual_wal_flush = true; + src.recycle_log_file_num = 2; + const DBOptions out = SanitizeOptions(dbname_, src); + ASSERT_FALSE(out.manual_wal_flush); + ASSERT_EQ(out.recycle_log_file_num, 0U); + ASSERT_FALSE(out.atomic_flush); + src.atomic_flush = true; + const DBOptions keep = SanitizeOptions(dbname_, src); + ASSERT_TRUE(keep.atomic_flush); +} + +TEST_F(DBOptionsTest, CrashSafeOrLogIndexDisablesWalCompression) { + if (!StreamingCompressionTypeSupported(kZSTD)) { + return; + } + for (bool crash_safe : {false, true}) { + for (bool log_index : {false, true}) { + SCOPED_TRACE(crash_safe); + SCOPED_TRACE(log_index); + DBOptions src; + src.memtable_crash_safe_recover = crash_safe; + src.memtable_as_log_index = log_index; + src.wal_compression = kZSTD; + const DBOptions out = SanitizeOptions(dbname_, src); + ASSERT_EQ(out.wal_compression, + crash_safe || log_index ? kNoCompression : kZSTD); + } + } +} + +TEST_F(DBOptionsTest, ReadOnlyDisablesCrashSafeAndWarns) { + for (bool use_logger : {false, true}) { + for (bool crash_safe : {false, true}) { + SCOPED_TRACE(use_logger); + SCOPED_TRACE(crash_safe); + DBOptions src; + src.memtable_crash_safe_recover = crash_safe; + src.manual_wal_flush = true; + src.recycle_log_file_num = 2; + src.wal_recovery_mode = WALRecoveryMode::kSkipAnyCorruptedRecords; + if (StreamingCompressionTypeSupported(kZSTD)) { + src.wal_compression = kZSTD; + } + const std::string log_path = dbname_ + "/readonly-warning.log"; + if (use_logger) { + ASSERT_OK(env_->NewLogger(log_path, &src.info_log)); + src.info_log->SetInfoLogLevel(InfoLogLevel::WARN_LEVEL); + } + testing::internal::CaptureStderr(); + const DBOptions out = SanitizeOptions(dbname_, src, /*read_only=*/true); + const std::string stderr_output = testing::internal::GetCapturedStderr(); + ASSERT_FALSE(out.memtable_crash_safe_recover); + ASSERT_EQ(out.info_log, src.info_log); + ASSERT_TRUE(out.manual_wal_flush); + ASSERT_EQ(out.recycle_log_file_num, 2U); + ASSERT_EQ(out.wal_compression, src.wal_compression); + std::string warning = stderr_output; + if (use_logger) { + ASSERT_TRUE(stderr_output.empty()); + src.info_log->Flush(); + ASSERT_OK(ReadFileToString(env_, log_path, &warning)); + } + if (crash_safe) { + ASSERT_NE(warning.find("WARN"), std::string::npos); + ASSERT_NE(warning.find("disabled for read-only Open"), std::string::npos); + ASSERT_NE(warning.find("replay may be very slow"), std::string::npos); + ASSERT_NE(warning.find("much larger MemTables"), std::string::npos); + } else { + ASSERT_TRUE(warning.empty()); + } + } + } +} + +TEST_F(DBOptionsTest, ReadOnlyReplaysWalWithCrashSafeDisabled) { + Options options = CurrentOptions(); + options.memtable_crash_safe_recover = true; + options.avoid_flush_during_shutdown = true; + DestroyAndReopen(options); + ASSERT_OK(Put("k", "v")); + Close(); + DB* ro = nullptr; + ASSERT_OK(DB::OpenForReadOnly(options, dbname_, &ro)); + std::unique_ptr read_only_db(ro); + ASSERT_FALSE(ro->GetDBOptions().memtable_crash_safe_recover); + std::string value; + ASSERT_OK(ro->Get(ReadOptions(), "k", &value)); + ASSERT_EQ(value, "v"); +} + +TEST_F(DBOptionsTest, CrashSafeForcesWal) { + for (bool crash_safe : {false, true}) { + SCOPED_TRACE(crash_safe); + Options options = CurrentOptions(); + options.memtable_as_log_index = false; + options.memtable_crash_safe_recover = crash_safe; + options.statistics = CreateDBStatistics(); + Reopen(options); + WriteOptions write; + write.disableWAL = true; + const uint64_t before = options.statistics->getTickerCount(WAL_FILE_BYTES); + ASSERT_OK(db_->Put(write, "k", "v")); + const uint64_t after = options.statistics->getTickerCount(WAL_FILE_BYTES); + if (crash_safe) { + ASSERT_GT(after, before); + } else { + ASSERT_EQ(after, before); + } + } +} + TEST_F(DBOptionsTest, ImmutableTrackAndVerifyWalsInManifest) { Options options; options.env = env_; diff --git a/include/rocksdb/options.h b/include/rocksdb/options.h index c27a17b3d..b80d59309 100644 --- a/include/rocksdb/options.h +++ b/include/rocksdb/options.h @@ -937,6 +937,10 @@ struct DBOptions { // and ConvertToSST to reduce WAL replay on the next open. // Only process crashes (kill -9 / _exit) are covered, not power loss. // Independent of memtable_as_log_index and Flush/Close ConvertToSST. + // Requires WAL: disableWAL is ignored; manual_wal_flush and WAL recycling + // are disabled, and wal_compression is forced to kNoCompression. + // Read-only Open disables this option and warns that full WAL replay may be + // very slow with the much larger MemTables encouraged in crash-safe mode. // Default: false. Not dynamically changeable through SetDBOptions(). bool memtable_crash_safe_recover = false; From 09f3a2cfd08accfbe854f913fea9c8e44d38d032 Mon Sep 17 00:00:00 2001 From: leipeng Date: Sat, 26 Sep 2026 13:42:24 +0800 Subject: [PATCH 14/58] Add SeekToFileOffset to the WAL reader Allow a fresh WAL reader to resume at a saved record boundary while preserving absolute record offsets for both classic and log-index WALs. Crash-safe recovery needs to replay the WAL tail without reading the prefix already covered by recovered MemTable data. Skipping bytes only in the underlying file loses block alignment and absolute record offsets, while replaying that prefix defeats the recovery speedup for large MemTables. --- db/log_reader.cc | 51 +++++++++++++++++++++++++++ db/log_reader.h | 10 ++++-- db/log_test.cc | 89 ++++++++++++++++++++++++++++++++++++++++++++++++ 3 files changed, 148 insertions(+), 2 deletions(-) diff --git a/db/log_reader.cc b/db/log_reader.cc index 309720b3e..cbbc39d64 100644 --- a/db/log_reader.cc +++ b/db/log_reader.cc @@ -69,6 +69,57 @@ void Reader::InitSetMemTableAsLogIndex(FileSystem& fs) { backing_store_ = nullptr; } +IOStatus Reader::SeekToFileOffset(uint64_t file_offset) { + IOStatus io_s; + TEST_SYNC_POINT_CALLBACK("CrashSafeRecover::SeekToFileOffset:InjectStatus", + &io_s); + if (!io_s.ok()) { + return io_s; + } + buffer_ = Slice(); + eof_ = false; + read_error_ = false; + last_record_offset_ = 0; + eof_offset_ = 0; + if (memtable_as_log_index_) { + if (file_offset > 0) { + io_s = file_->Skip(file_offset); + if (!io_s.ok()) { + return io_s; + } + } + end_of_buffer_offset_ = file_offset; + return IOStatus::OK(); + } + const uint64_t frame_start = file_offset - (file_offset % kBlockSize); + if (frame_start > 0) { + io_s = file_->Skip(frame_start); + if (!io_s.ok()) { + return io_s; + } + } + io_s = file_->Read(kBlockSize, &buffer_, backing_store_, Env::IO_TOTAL); + if (!io_s.ok()) { + buffer_.clear(); + end_of_buffer_offset_ = frame_start; + return io_s; + } + end_of_buffer_offset_ = frame_start + buffer_.size(); + const size_t skip_in_frame = static_cast(file_offset - frame_start); + if (buffer_.size() < skip_in_frame) { + eof_ = true; + eof_offset_ = buffer_.size(); + buffer_.clear(); + return IOStatus::OK(); + } + buffer_.remove_prefix(skip_in_frame); + if (end_of_buffer_offset_ - frame_start < kBlockSize) { + eof_ = true; + eof_offset_ = static_cast(end_of_buffer_offset_ - frame_start); + } + return IOStatus::OK(); +} + IOStatus Reader::IsMemTableAsLogIndexFile (FileSystem& fs, const std::string& fname, bool* result) { FileOptions fopt; diff --git a/db/log_reader.h b/db/log_reader.h index ac694eaa9..acd7c288b 100644 --- a/db/log_reader.h +++ b/db/log_reader.h @@ -89,9 +89,15 @@ class Reader { // Undefined before the first call to ReadRecord. uint64_t LastRecordOffset(); + // Position a fresh reader whose file is still at offset 0. Only for + // uncompressed, non-recycled WALs with no required metadata in the skipped + // prefix. file_offset must be the physical start of a logical record or EOF. + // Subsequent record offsets remain absolute. Discard the reader on failure. + IOStatus SeekToFileOffset(uint64_t file_offset); + // Returns the first physical offset after the last record returned by - // ReadRecord, or zero before first call to ReadRecord. This can also be - // thought of as the "current" position in processing the file bytes. + // ReadRecord. Before the first ReadRecord, returns the SeekToFileOffset + // position, or zero if not positioned. This is the current processing offset. uint64_t LastRecordEnd(); // returns true if the reader has encountered an eof condition. diff --git a/db/log_test.cc b/db/log_test.cc index 0bf3bf5ae..9b7309c3f 100644 --- a/db/log_test.cc +++ b/db/log_test.cc @@ -278,6 +278,95 @@ class LogTest } }; +class LogSeekTest : public LogTest {}; + +TEST_P(LogSeekTest, SeekToFileOffsetAtStart) { + Write("first"); + ASSERT_OK(reader_->SeekToFileOffset(0)); + ASSERT_EQ("first", Read()); + ASSERT_EQ(reader_->LastRecordOffset(), 0U); + ASSERT_EQ(reader_->LastRecordEnd(), WrittenBytes()); +} + +TEST_P(LogSeekTest, SeekToFileOffsetAfterFragmentedRecord) { + Write(BigString("B", kBlockSize * 2 + 100)); + const uint64_t tail_offset = WrittenBytes(); + const std::string tail = BigString("tail", kBlockSize + 100); + Write(tail); + const uint64_t tail_end = WrittenBytes(); + Write("last"); + ASSERT_OK(reader_->SeekToFileOffset(tail_offset)); + ASSERT_EQ(tail, Read()); + ASSERT_EQ(reader_->LastRecordOffset(), tail_offset); + ASSERT_EQ(reader_->LastRecordEnd(), tail_end); + ASSERT_EQ("last", Read()); + ASSERT_EQ(reader_->LastRecordOffset(), tail_end); + ASSERT_EQ(reader_->LastRecordEnd(), WrittenBytes()); + ASSERT_EQ("EOF", Read()); +} + +TEST_P(LogSeekTest, SeekToFileOffsetAtBlockBoundary) { + Write(BigString("B", kBlockSize - kHeaderSize)); + ASSERT_EQ(WrittenBytes(), kBlockSize); + Write("tail"); + ASSERT_OK(reader_->SeekToFileOffset(kBlockSize)); + ASSERT_EQ("tail", Read()); + ASSERT_EQ(reader_->LastRecordOffset(), kBlockSize); + ASSERT_EQ(reader_->LastRecordEnd(), WrittenBytes()); +} + +TEST_P(LogSeekTest, SeekToFileOffsetAtEOF) { + Write("first"); + ASSERT_OK(reader_->SeekToFileOffset(WrittenBytes())); + ASSERT_EQ("EOF", Read()); + ASSERT_TRUE(IsEOF()); + ASSERT_EQ(reader_->LastRecordEnd(), WrittenBytes()); +} + +TEST_P(LogSeekTest, SeekToFileOffsetReadError) { + Write("first"); + ForceError(3); + ASSERT_TRUE(reader_->SeekToFileOffset(0).IsCorruption()); +} + +TEST_P(LogSeekTest, SeekToFileOffsetSkipError) { + Write("first"); + ASSERT_TRUE(reader_->SeekToFileOffset(kBlockSize).IsNotFound()); +} + +INSTANTIATE_TEST_CASE_P( + Log, LogSeekTest, + ::testing::Combine(::testing::Values(0), ::testing::Bool(), + ::testing::Values(CompressionType::kNoCompression))); + +TEST(LogReaderSeekTest, SeekToFileOffsetLogIndex) { + const auto fs = Env::Default()->GetFileSystem(); + const std::string fname = test::PerThreadDBPath("log_seek_index"); + { + std::unique_ptr dest; + ASSERT_OK(WritableFileWriter::Create(fs, fname, FileOptions(), &dest, nullptr)); + Writer writer(std::move(dest), 1, false); + writer.InitReaderMmap(*fs, kBlockSize * 4); + ASSERT_OK(writer.AddRecord(BigString("B", kBlockSize * 2 + 100))); + const uint64_t tail_offset = writer.get_log_offset(); + ASSERT_OK(writer.AddRecord("tail")); + std::unique_ptr file; + ASSERT_OK(SequentialFileReader::Create(fs, fname, FileOptions(), &file, + nullptr, nullptr)); + Reader reader(nullptr, std::move(file), nullptr, true, 1); + reader.InitSetMemTableAsLogIndex(*fs); + ASSERT_OK(reader.SeekToFileOffset(tail_offset - sizeof(RawRecHeader))); + Slice record; + std::string scratch; + ASSERT_TRUE(reader.ReadRecord(&record, &scratch)); + ASSERT_EQ(record.ToString(), "tail"); + ASSERT_EQ(reader.LastRecordOffset(), tail_offset); + ASSERT_EQ(reader.LastRecordEnd(), writer.file()->GetFileSize()); + ASSERT_FALSE(reader.ReadRecord(&record, &scratch)); + } + ASSERT_OK(fs->DeleteFile(fname, IOOptions(), nullptr)); +} + TEST_P(LogTest, Empty) { ASSERT_EQ("EOF", Read()); } TEST_P(LogTest, ReadWrite) { From a2cc14cb36870103ef9267009a3fa2588e359fab Mon Sep 17 00:00:00 2001 From: leipeng Date: Sun, 27 Sep 2026 22:52:43 +0800 Subject: [PATCH 15/58] DB Recovery: Utilize MemTable's Crash-Safe capability MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Principle ToplingDB's file-mmap memtable survives a process crash. This gives the DB an essential ability to recover. ⇒ 1. But the existing recovery model is inherited from RocksDB, of which the approach is replaying the WAL. ⇒ 2. To utilize the crash-safe ability, there is a gap: a crash does not reveal how far the memtable insert went. The crash-safe boundary the memtable can guarantee is each KV, which does not match the DB's atomic granularity --- a WriteBatch. ⇒ 3. So we need to align the memtable's KV boundary to the WriteBatch, this just needs a SeqNum. ⇒ 4. The SeqNum must represent a safe point, we call it pubseq, the LastPublishedSequence is the ideal candidate for pubseq, our solution is to get pubseq as close as possible to LastPublishedSequence ---- this may need hiding a few KVs already inserted into the memtable as a sacrifice. ⇒ 5. Once we do not truly reach LastPublishedSequence, what is hidden/sacrificed in memtable will be an entire WAL Record (containing at least one WriteBatch), then in recovery, we need to replay the tail records in WAL, --- an extremely small part of the WAL. Support 1. On the default write path, an idle DB replays no WAL records. WALs older than the KV in the WAL that makes pubseq visible are skipped. The WAL of this KV is opened and seeked to the end, and no further record is read. 2. two_write_queues without seq_per_batch publishes after the memtable side finishes. An idle DB also replays no records. 3. unordered_write and pipelined_write keep in memory the latest KV in the WAL that makes pubseq visible. CSPUBSEQ trails one WAL group, so that last group is replayed after a write. 4. allow_2pc still converts leftovers. Records before the KV in the WAL that makes pubseq visible are read and not inserted, so recovered_transactions_ is rebuilt. 2PC can still be optimized further. This commit does not do that. 5. TransactionDB WriteCommitted is supported, including prepares on the second queue. WritePrepared and WriteUnprepared fall back to full WAL replay. KV in the WAL that makes pubseq visible 1. Publish pubseq and the KV in the WAL that makes it visible, in CSPUBSEQ. Convert leftover file-mmap memtables up to pubseq, then replay only the WAL tail. 2. Unusable leftovers or an unusable KV in the WAL that makes pubseq visible fall back to full WAL replay. 3. An Open whose WAL kind differs from the saved kind of the KV in the WAL that makes pubseq visible is not handled here. --- Makefile | 3 + db/db_cspp_crash_safe_test.cc | 3072 +++++++++++++++++++++++++++++++++ db/db_impl/db_impl.cc | 8 + db/db_impl/db_impl.h | 42 + db/db_impl/db_impl_open.cc | 641 ++++++- db/db_impl/db_impl_write.cc | 97 +- db/db_write_test.cc | 14 + db/write_batch.cc | 6 + db/write_batch_internal.h | 1 + db/write_thread.h | 3 + file/filename.cc | 4 + file/filename.h | 3 + include/rocksdb/options.h | 7 +- src.mk | 1 + 14 files changed, 3883 insertions(+), 19 deletions(-) create mode 100644 db/db_cspp_crash_safe_test.cc diff --git a/Makefile b/Makefile index 7be0bc134..462fd9cba 100644 --- a/Makefile +++ b/Makefile @@ -2140,6 +2140,9 @@ db_universal_compaction_test: $(OBJ_DIR)/db/db_universal_compaction_test.o $(TES db_wal_test: $(OBJ_DIR)/db/db_wal_test.o $(TEST_LIBRARY) $(LIBRARY) $(AM_LINK) +db_cspp_crash_safe_test: $(OBJ_DIR)/db/db_cspp_crash_safe_test.o $(TEST_LIBRARY) $(LIBRARY) + $(AM_LINK) + db_io_failure_test: $(OBJ_DIR)/db/db_io_failure_test.o $(TEST_LIBRARY) $(LIBRARY) $(AM_LINK) diff --git a/db/db_cspp_crash_safe_test.cc b/db/db_cspp_crash_safe_test.cc new file mode 100644 index 000000000..d2e901761 --- /dev/null +++ b/db/db_cspp_crash_safe_test.cc @@ -0,0 +1,3072 @@ +// Copyright (c) 2026-present, Topling Inc. +// Crash-safe leftover recover: Convert + WAL tail, sync-point injection. + +#include +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include + +#include +#include +#include + +#include "db/column_family.h" +#include "db/db_impl/db_impl.h" +#include "db/db_test_util.h" +#include "db/log_reader.h" +#include "db/log_writer.h" +#include "db/pre_release_callback.h" +#include "file/filename.h" +#include "file/file_util.h" +#include "file/sequence_file_reader.h" +#include "file/writable_file_writer.h" +#include "port/port.h" +#include "port/stack_trace.h" +#include "rocksdb/io_status.h" +#include "rocksdb/statistics.h" +#include "rocksdb/utilities/transaction_db.h" +#include "rocksdb/wal_filter.h" +#include "table/get_context.h" +#include "table/table_builder.h" +#include "table/table_reader.h" +#include "utilities/merge_operators.h" +#include "utilities/fault_injection_fs.h" +#include "test_util/sync_point.h" +#include "test_util/testutil.h" + +namespace ROCKSDB_NAMESPACE { + +extern MemTableRepFactory* NewCSPPMemTabForPlain(const std::string&); +std::shared_ptr EasyNewMemTableRep(Slice cls, Slice js); + +namespace { + +struct PublishedSeqOnDisk { + uint64_t magic = 0; + uint32_t version = 0; + uint32_t header_size = 0; + uint32_t wal_offset_kind = 0; + uint32_t kind_since_wal = 0; + uint64_t generation = 0; + uint64_t pubseq = 0; + uint64_t wal_number = 0; + uint64_t wal_offset = 0; + uint64_t padding = 0; +}; + +void SetupCspp(Options* options, bool file_mmap) { + const char* js = file_mmap + ? R"({"mem_cap":16777216,"convert_to_sst":"kFileMmap"})" + : R"({"mem_cap":16777216,"convert_to_sst":"kDontConvert"})"; + options->memtable_factory.reset(NewCSPPMemTabForPlain(js)); + const SidePluginRepo repo; + options->table_factory = PluginFactorySP::AcquirePlugin( + "CSPPMemTabTable", json::parse(js), repo); +} + +void SetupOsl(Options* options, bool file_mmap) { + const char* js = file_mmap + ? R"({"mem_cap":16777216,"convert_to_sst":"kFileMmap"})" + : R"({"mem_cap":16777216,"convert_to_sst":"kDontConvert"})"; + options->memtable_factory = EasyNewMemTableRep("OffsetSkipList", js); + const SidePluginRepo repo; + options->table_factory = PluginFactorySP::AcquirePlugin( + "OffsetSkipListTable", json::parse(js), repo); +} + +Options BaseCrashSafeOptions(const std::string& dbname, bool recover, + bool log_index) { + Options options; + options.create_if_missing = true; + options.error_if_exists = false; + options.memtable_crash_safe_recover = recover; + options.memtable_as_log_index = log_index; + options.avoid_flush_during_shutdown = false; + options.avoid_flush_during_recovery = true; + options.disable_auto_compactions = true; + options.write_buffer_size = 64 << 20; + options.max_write_buffer_number = 8; + options.level0_file_num_compaction_trigger = 1 << 20; + options.env = Env::Default(); + options.wal_dir = dbname; + SetupCspp(&options, true); + return options; +} + +std::vector ListLeftovers(const Options& options, + const std::string& dir) { + std::vector leftovers; + if (options.memtable_factory) { + options.memtable_factory->ListCrashSafeLeftovers(dir, &leftovers); + } + return leftovers; +} + +bool ReadPublishedSeqFile(const std::string& dbname, PublishedSeqOnDisk* out) { + const std::string path = CrashSafePubSeqFileName(dbname); + int fd = ::open(path.c_str(), O_RDONLY); + if (fd < 0) { + return false; + } + const ssize_t n = ::pread(fd, out, sizeof(*out), 0); + ::close(fd); + return n == static_cast(sizeof(*out)); +} + +bool SetPublishedSeqGeneration(const std::string& dbname, uint64_t g) { + const std::string path = CrashSafePubSeqFileName(dbname); + int fd = ::open(path.c_str(), O_WRONLY); + if (fd < 0) { + return false; + } + const ssize_t n = + ::pwrite(fd, &g, sizeof(g), offsetof(PublishedSeqOnDisk, generation)); + ::close(fd); + return n == static_cast(sizeof(g)); +} + +bool SetPublishedSeqWalOffsetKind(const std::string& dbname, uint32_t kind) { + const std::string path = CrashSafePubSeqFileName(dbname); + int fd = ::open(path.c_str(), O_WRONLY); + if (fd < 0) { + return false; + } + const ssize_t n = + ::pwrite(fd, &kind, sizeof(kind), + offsetof(PublishedSeqOnDisk, wal_offset_kind)); + ::close(fd); + return n == static_cast(sizeof(kind)); +} + +bool ZeroPublishedSeqFile(const std::string& dbname) { + const std::string path = CrashSafePubSeqFileName(dbname); + const std::string zeros(4096, '\0'); + int fd = ::open(path.c_str(), O_WRONLY); + if (fd < 0) { + return false; + } + const ssize_t n = ::pwrite(fd, zeros.data(), zeros.size(), 0); + ::close(fd); + return n == static_cast(zeros.size()); +} + +int CountL0(DB* db, const std::string& cf_name = "") { + std::vector files; + db->GetLiveFilesMetaData(&files); + int n = 0; + for (const auto& f : files) { + if (f.level == 0 && (cf_name.empty() || f.column_family_name == cf_name)) { + n++; + } + } + return n; +} + +#if !defined(OS_WIN) +// Only exec/_exit run between fork and exec; DB and SyncPoint state are +// initialized in the new process, including Env's background worker threads. +int RunCrashChild(const std::string& dbname, const char* scenario, + const std::string& arg = "") { + char exe[] = "/proc/self/exe"; + char flag[] = "--crash-child"; + char* argv[] = {exe, flag, const_cast(scenario), + const_cast(dbname.c_str()), + const_cast(arg.c_str()), nullptr}; + const pid_t pid = ::fork(); + if (pid == 0) { + ::execv(exe, argv); + ::_exit(127); + } + if (pid < 0) { + return -1; + } + int st = 0; + pid_t waited; + do { + waited = ::waitpid(pid, &st, 0); + } while (waited < 0 && errno == EINTR); + return waited == pid && WIFEXITED(st) ? WEXITSTATUS(st) : -1; +} + +const char* crash_child_db = nullptr; +const char* crash_child_arg = nullptr; + +class CrashChild : public ::testing::Test { + protected: + const std::string dbname_ = crash_child_db ? crash_child_db : ""; + const std::string arg_ = crash_child_arg ? crash_child_arg : ""; +}; + +#endif + +} // namespace + +class DBCsppCrashSafeTest : public DBTestBase { + public: + DBCsppCrashSafeTest() + : DBTestBase("db_cspp_crash_safe_test", /*env_do_fsync=*/false) {} +}; + +TEST_F(DBCsppCrashSafeTest, CrashSafeRequiresFileMmap) { + for (const char* cls : {"CSPPMemTab", "OffsetSkipList"}) { + for (const char* mode : {"kDontConvert", "kDumpMem", "kFileMmap"}) { + SCOPED_TRACE(cls); + SCOPED_TRACE(mode); + const SidePluginRepo repo; + auto factory = PluginFactorySP::AcquirePlugin( + cls, {{"convert_to_sst", mode}}, repo); + ASSERT_EQ(factory->SupportCrashSafe(), std::string(mode) == "kFileMmap"); + } + } +} + +TEST_F(DBCsppCrashSafeTest, DangerousFactoryUpdate) { + Close(); + const SidePluginRepo repo; + const json query = {{"html", false}}; + auto check = [&](const auto& factory, const auto* manip, bool allowed) { + auto state = [&] { + return json::parse(manip->ToString(*factory, query, repo)); + }; + auto update = [&](const json& body) { + try { + manip->HandleUpdate(factory.get(), query, body, repo); + return Status::OK(); + } catch (const Status& s) { + return s; + } + }; + ASSERT_EQ(state()["convert_to_sst"], "kFileMmap"); + ASSERT_EQ(state()["allow_dangerous_update"], allowed); + ASSERT_OK(update({{"token_use_idle", false}})); + ASSERT_EQ(state()["token_use_idle"], false); + ASSERT_OK(update({{"convert_to_sst", "kFileMmap"}})); + + const json before = state(); + Status s = update({{"allow_dangerous_update", !allowed}, + {"token_use_idle", true}, {"populate_read", false}}); + ASSERT_TRUE(s.IsInvalidArgument()); + ASSERT_NE(s.ToString().find("cannot be changed online"), std::string::npos); + ASSERT_EQ(state(), before); + s = update({{"allow_dangerous_update", !allowed}, + {"convert_to_sst", "kDontConvert"}}); + ASSERT_TRUE(s.IsInvalidArgument()); + ASSERT_EQ(state(), before); + s = update({{"convert_to_sst", "invalid"}, {"token_use_idle", true}}); + ASSERT_TRUE(s.IsInvalidArgument()); + ASSERT_EQ(state(), before); + ASSERT_OK(update({{"allow_dangerous_update", allowed}, + {"convert_to_sst", "kFileMmap"}})); + if (!allowed) { + s = update({{"convert_to_sst", "kDontConvert"}, + {"token_use_idle", true}, {"populate_read", false}}); + ASSERT_TRUE(s.IsInvalidArgument()); + ASSERT_NE(s.ToString().find("allow_dangerous_update=true"), std::string::npos); + ASSERT_EQ(state(), before); + return; + } + ASSERT_OK(update({{"convert_to_sst", "kDumpMem"}})); + ASSERT_EQ(state()["convert_to_sst"], "kDumpMem"); + ASSERT_OK(update({{"allow_dangerous_update", true}, + {"convert_to_sst", "kDontConvert"}})); + ASSERT_EQ(state()["convert_to_sst"], "kDontConvert"); + ASSERT_OK(update({{"convert_to_sst", "kFileMmap"}})); + ASSERT_EQ(state()["convert_to_sst"], "kFileMmap"); + ASSERT_EQ(state()["allow_dangerous_update"], true); + }; + for (bool allowed : {false, true}) { + SCOPED_TRACE(allowed); + json params = {{"mem_cap", 16777216}, {"convert_to_sst", "kFileMmap"}}; + if (allowed) params["allow_dangerous_update"] = true; + for (const char* cls : {"CSPPMemTab", "OffsetSkipList"}) { + SCOPED_TRACE(cls); + auto factory = PluginFactorySP::AcquirePlugin( + cls, params, repo); + auto* manip = PluginManip::AcquirePlugin(cls, {}, repo); + check(factory, manip, allowed); + } + for (const char* cls : {"CSPPMemTabTable", "OffsetSkipListTable"}) { + SCOPED_TRACE(cls); + auto factory = PluginFactorySP::AcquirePlugin(cls, params, repo); + auto* manip = PluginManip::AcquirePlugin(cls, {}, repo); + check(factory, manip, allowed); + } + } +} + +TEST_F(DBCsppCrashSafeTest, DangerousUpdateAffectsOnlyNewMemtables) { + Close(); + const SidePluginRepo repo; + InternalKeyComparator icmp(BytewiseComparator()); + MemTable::KeyComparator cmp(icmp); + for (const char* cls : {"CSPPMemTab", "OffsetSkipList"}) { + SCOPED_TRACE(cls); + Arena arena; + auto factory = PluginFactorySP::AcquirePlugin( + cls, {{"mem_cap", 16777216}, {"allow_dangerous_update", true}}, repo); + auto* manip = PluginManip::AcquirePlugin(cls, {}, repo); + std::unique_ptr before( + factory->CreateMemTableRep(cmp, &arena, nullptr, nullptr)); + ASSERT_FALSE(before->SupportConvertToSST()); + manip->Update(factory.get(), {}, {{"convert_to_sst", "kDumpMem"}}, repo); + std::unique_ptr after( + factory->CreateMemTableRep(cmp, &arena, nullptr, nullptr)); + ASSERT_FALSE(before->SupportConvertToSST()); + ASSERT_TRUE(after->SupportConvertToSST()); + } +} + +TEST_F(DBCsppCrashSafeTest, RecoveredTableUsesFileSequenceBound) { + Close(); + for (bool osl : {false, true}) { + SCOPED_TRACE(osl ? "OSL" : "CSPP"); + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) { + SetupOsl(&options, true); + } + Destroy(options); + ASSERT_OK(env_->CreateDirIfMissing(dbname_)); + options.cf_paths = {{dbname_, 0}}; + InternalKeyComparator icmp(options.comparator); + ImmutableOptions ioptions(options); + MutableCFOptions moptions(options); + WriteBufferManager wb(options.db_write_buffer_size); + std::unique_ptr mem(new MemTable( + icmp, ioptions, moptions, &wb, kMaxSequenceNumber, 0)); + ASSERT_OK(mem->Add(1, kTypeValue, "key", "old", nullptr)); + ASSERT_OK(mem->Add(3, kTypeValue, "key", "new", nullptr)); + ASSERT_OK(mem->Add(4, kTypeValue, "ghost", "unpublished", nullptr)); + mem->MarkImmutable(); + const auto leftovers = ListLeftovers(options, dbname_); + ASSERT_EQ(leftovers.size(), 1U); + const std::string leftover = leftovers[0] + ".recovery"; + CopyFile(leftovers[0], leftover); + mem.reset(); + + IntTblPropCollectorFactories collectors; + TableBuilderOptions tbo(ioptions, moptions, icmp, &collectors, + options.compression, options.compression_opts, 0, + "default", 0); + FileMetaData meta; + meta.fd = FileDescriptor(1, 0, 0); + meta.fd.smallest_seqno = 0; + // The published bound need not be the sequence of any physical entry. + meta.fd.largest_seqno = 2; + ASSERT_OK(options.memtable_factory->RecoverCrashSafeMemTableToSST( + leftover, &meta, tbo)); + ASSERT_GT(meta.fd.GetFileSize(), 0U); + ASSERT_EQ(meta.fd.smallest_seqno, 0U); + ASSERT_EQ(meta.fd.largest_seqno, 2U); + + const std::string fname = TableFileName(options.cf_paths, 1, 0); + for (SequenceNumber limit : {meta.fd.largest_seqno, SequenceNumber(4), + SequenceNumber(0)}) { + SCOPED_TRACE(limit); + std::unique_ptr file; + ASSERT_OK(env_->GetFileSystem()->NewRandomAccessFile( + fname, FileOptions(), &file, nullptr)); + std::unique_ptr reader( + new RandomAccessFileReader(std::move(file), fname)); + EnvOptions env_options; + TableReaderOptions tro(ioptions, options.prefix_extractor, env_options, + icmp, 0); + // Reopen the same bytes with a different FileDescriptor visibility bound. + FileDescriptor fd = meta.fd; + fd.largest_seqno = limit; + tro.largest_seqno = fd.largest_seqno; + std::unique_ptr table; + ASSERT_OK(options.table_factory->NewTableReader( + ReadOptions(), tro, std::move(reader), fd.GetFileSize(), &table, + true)); + for (const char* key : {"key", "ghost"}) { + PinnableSlice value; + GetContext get_context( + options.comparator, nullptr, nullptr, nullptr, GetContext::kNotFound, + key, &value, nullptr, nullptr, nullptr, true, nullptr, nullptr); + InternalKey ikey(key, kMaxSequenceNumber, kTypeValue); + ASSERT_OK(table->Get(ReadOptions(), ikey.Encode(), &get_context, + nullptr)); + const bool visible = limit == 4 || (limit == 2 && key[0] == 'k'); + ASSERT_EQ(get_context.State(), + visible ? GetContext::kFound : GetContext::kNotFound); + if (visible) { + ASSERT_EQ(value.ToString(), key[0] == 'g' ? "unpublished" + : limit == 2 ? "old" : "new"); + } + } + } + } +} + +TEST_F(DBCsppCrashSafeTest, RecoveredTableIteratorUsesFileSequenceBound) { + Close(); + for (bool osl : {false, true}) { + for (bool reverse : {false, true}) { + SCOPED_TRACE(osl ? "OSL" : "CSPP"); + SCOPED_TRACE(reverse); + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + if (reverse) options.comparator = ReverseBytewiseComparator(); + Destroy(options); + ASSERT_OK(env_->CreateDirIfMissing(dbname_)); + options.cf_paths = {{dbname_, 0}}; + InternalKeyComparator icmp(options.comparator); + ImmutableOptions ioptions(options); + MutableCFOptions moptions(options); + WriteBufferManager wb(options.db_write_buffer_size); + auto mem = std::make_unique( + icmp, ioptions, moptions, &wb, kMaxSequenceNumber, 0); + std::vector physical; + for (const auto& entry : {std::make_pair("b", 1), {"d", 2}, {"b", 3}, + {"d", 5}, {"c", 6}, {"b", 7}, {"a", 8}, + {"z", 9}}) { + ASSERT_OK(mem->Add(entry.second, kTypeValue, entry.first, + std::to_string(entry.second), nullptr)); + physical.push_back( + InternalKey(entry.first, entry.second, kTypeValue).Encode().ToString()); + } + auto less = [&](const std::string& a, const std::string& b) { + return icmp.Compare(a, b) < 0; + }; + std::sort(physical.begin(), physical.end(), less); + mem->MarkImmutable(); + const auto leftovers = ListLeftovers(options, dbname_); + ASSERT_EQ(leftovers.size(), 1U); + const std::string leftover = leftovers[0] + ".iterator"; + CopyFile(leftovers[0], leftover); + mem.reset(); + IntTblPropCollectorFactories collectors; + TableBuilderOptions tbo(ioptions, moptions, icmp, &collectors, + options.compression, options.compression_opts, 0, + "default", 0); + FileMetaData meta; + meta.fd = FileDescriptor(1, 0, 0); + meta.fd.smallest_seqno = 0; + meta.fd.largest_seqno = 4; + ASSERT_OK(options.memtable_factory->RecoverCrashSafeMemTableToSST( + leftover, &meta, tbo)); + const std::string fname = TableFileName(options.cf_paths, 1, 0); + for (SequenceNumber limit : {SequenceNumber(0), SequenceNumber(4), + SequenceNumber(9), kMaxSequenceNumber}) { + SCOPED_TRACE(limit); + std::unique_ptr file; + ASSERT_OK(env_->GetFileSystem()->NewRandomAccessFile( + fname, FileOptions(), &file, nullptr)); + auto reader = std::make_unique(std::move(file), fname); + EnvOptions env_options; + TableReaderOptions tro(ioptions, options.prefix_extractor, env_options, + icmp, 0); + tro.largest_seqno = limit; + std::unique_ptr table; + ASSERT_OK(options.table_factory->NewTableReader( + ReadOptions(), tro, std::move(reader), meta.fd.GetFileSize(), + &table, true)); + std::vector expected; + for (const auto& key : physical) { + if (GetInternalKeySeqno(key) <= limit) expected.push_back(key); + } + for (bool use_arena : {false, true}) { + SCOPED_TRACE(use_arena); + Arena arena; + auto destroy = [&](InternalIterator* p) { + if (use_arena) p->~InternalIterator(); + else delete p; + }; + std::unique_ptr it( + table->NewIterator(ReadOptions(), nullptr, + use_arena ? &arena : nullptr, false, + TableReaderCaller::kUserIterator), destroy); + auto check = [&](size_t pos) { + ASSERT_EQ(it->Valid(), pos < expected.size()); + if (it->Valid()) { + ASSERT_EQ(it->key().ToString(), expected[pos]); + ASSERT_EQ(it->value().ToString(), + std::to_string(GetInternalKeySeqno(expected[pos]))); + } + }; + // Exercise every forward entry point, including the fast result API. + for (int advance = 0; advance < 3; ++advance) { + it->SeekToFirst(); + for (size_t i = 0; i < expected.size(); ++i) { + check(i); + ASSERT_TRUE(it->Valid()); + if (advance == 0) { + it->Next(); + } else if (advance == 1) { + ASSERT_EQ(it->NextAndCheckValid(), i + 1 < expected.size()); + } else { + IterateResult result; + ASSERT_EQ(it->NextAndGetResult(&result), i + 1 < expected.size()); + ASSERT_EQ(result.is_valid, i + 1 < expected.size()); + if (result.is_valid) { + ASSERT_EQ(result.key().ToString(), expected[i + 1]); + } + } + } + check(expected.size()); + } + for (bool fast : {false, true}) { + it->SeekToLast(); + for (size_t i = expected.size(); i > 0; --i) { + check(i - 1); + ASSERT_TRUE(it->Valid()); + if (fast) { + ASSERT_EQ(it->PrevAndCheckValid(), i > 1); + } else { + it->Prev(); + } + } + check(expected.size()); + } + auto* mem_it = static_cast(it.get()); + std::vector targets = physical; + for (const char* key : {"", "a", "b", "bb", "c", "d", "zz"}) { + for (SequenceNumber seq : {SequenceNumber(0), SequenceNumber(4), + kMaxSequenceNumber}) { + targets.push_back(InternalKey(key, seq, kTypeValue).Encode().ToString()); + } + } + for (const auto& target : targets) { + const size_t next = std::lower_bound( + expected.begin(), expected.end(), target, less) - expected.begin(); + const size_t upper = std::upper_bound( + expected.begin(), expected.end(), target, less) - expected.begin(); + const size_t prev = upper ? upper - 1 : expected.size(); + it->Seek(target); + check(next); + it->SeekForPrev(target); + check(prev); + std::string encoded; + const char* memkey = EncodeKey(&encoded, target); + mem_it->Seek(target, memkey); + check(next); + mem_it->SeekForPrev(target, memkey); + check(prev); + } + if (osl) { + for (int i = 0; i < 20; ++i) { + mem_it->RandomSeek(); + if (it->Valid()) { + ASSERT_LE(GetInternalKeySeqno(it->key()), limit); + } + } + } + } + } + } + } +} + +TEST_F(DBCsppCrashSafeTest, ConvertedTableVisibilityFilter) { + Close(); + for (bool osl : {false, true}) { + for (bool file_mmap : {false, true}) { + for (const auto& marker : { + std::make_pair("", false), {"VisFilter:0", false}, + {"VisFilter:10", false}, {"XVisFilter:1", false}, + {"Other:VisFilter:1", false}, {"VisFilter:1x", false}, + {"VisFilter:1", true}, {"Other:0;VisFilter:1", true}, + {"Other:0;VisFilter:1;Tail:0", true}}) { + SCOPED_TRACE(osl ? "OSL" : "CSPP"); + SCOPED_TRACE(file_mmap); + SCOPED_TRACE(marker.first); + Options options = BaseCrashSafeOptions(dbname_, true, false); + const char* js = file_mmap + ? R"({"mem_cap":16777216,"convert_to_sst":"kFileMmap"})" + : R"({"mem_cap":16777216,"convert_to_sst":"kDumpMem"})"; + if (osl) { + options.memtable_factory = EasyNewMemTableRep("OffsetSkipList", js); + } else { + options.memtable_factory.reset(NewCSPPMemTabForPlain(js)); + } + const SidePluginRepo repo; + options.table_factory = PluginFactorySP::AcquirePlugin( + osl ? "OffsetSkipListTable" : "CSPPMemTabTable", json::parse(js), repo); + Destroy(options); + ASSERT_OK(env_->CreateDirIfMissing(dbname_)); + options.cf_paths = {{dbname_, 0}}; + InternalKeyComparator icmp(options.comparator); + ImmutableOptions ioptions(options); + MutableCFOptions moptions(options); + WriteBufferManager wb(options.db_write_buffer_size); + auto mem = std::make_unique( + icmp, ioptions, moptions, &wb, kMaxSequenceNumber, 0); + ASSERT_OK(mem->Add(1, kTypeValue, "a", "1", nullptr)); + ASSERT_OK(mem->Add(2, kTypeValue, "b", "2", nullptr)); + ASSERT_OK(mem->Add(3, kTypeValue, "b", "3", nullptr)); + mem->MarkImmutable(); + IntTblPropCollectorFactories collectors; + TableBuilderOptions tbo(ioptions, moptions, icmp, &collectors, + options.compression, options.compression_opts, 0, + "default", 0); + FileMetaData meta; + meta.fd = FileDescriptor(1, 0, 0); + meta.fd.smallest_seqno = 1; + meta.fd.largest_seqno = 3; + SyncPoint::GetInstance()->SetCallBack( + "PropertyBlockBuilder::AddTableProperty:Start", [marker](void* p) { + auto& opts = static_cast(p)->compression_options; + ASSERT_EQ(opts.find("VisFilter:"), std::string::npos); + opts += marker.first; + }); + SyncPoint::GetInstance()->EnableProcessing(); + const Status converted = mem->ConvertToSST(&meta, tbo); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_OK(converted); + ASSERT_GT(meta.fd.GetFileSize(), 0U); + mem.reset(); + + const std::string fname = TableFileName(options.cf_paths, 1, 0); + const std::type_info* unfiltered_type = nullptr; + for (SequenceNumber limit : {kMaxSequenceNumber, meta.fd.largest_seqno}) { + std::unique_ptr file; + ASSERT_OK(env_->GetFileSystem()->NewRandomAccessFile( + fname, FileOptions(), &file, nullptr)); + auto reader = std::make_unique(std::move(file), fname); + EnvOptions env_options; + TableReaderOptions tro(ioptions, options.prefix_extractor, env_options, + icmp, 0); + tro.largest_seqno = limit; + std::unique_ptr table; + ASSERT_OK(options.table_factory->NewTableReader( + ReadOptions(), tro, std::move(reader), meta.fd.GetFileSize(), + &table, true)); + std::unique_ptr it(table->NewIterator( + ReadOptions(), nullptr, nullptr, false, + TableReaderCaller::kUserIterator)); + if (limit == kMaxSequenceNumber) { + unfiltered_type = &typeid(*it); + } else { + ASSERT_NE(unfiltered_type, nullptr); + // A finite file maximum alone must not select VisibleIter. + ASSERT_EQ(typeid(*it) == *unfiltered_type, !marker.second); + } + size_t count = 0; + for (it->SeekToFirst(); it->Valid(); it->Next()) ++count; + ASSERT_EQ(count, 3U); + } + } + } + } +} + +TEST_F(DBCsppCrashSafeTest, OslRecoveryUnmapsWholeFile) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + SetupOsl(&options, true); + Destroy(options); + ASSERT_OK(env_->CreateDirIfMissing(dbname_)); + options.cf_paths = {{dbname_, 0}}; + InternalKeyComparator icmp(options.comparator); + ImmutableOptions ioptions(options); + MutableCFOptions moptions(options); + WriteBufferManager wb(options.db_write_buffer_size); + std::unique_ptr mem(new MemTable( + icmp, ioptions, moptions, &wb, kMaxSequenceNumber, 0)); + ASSERT_OK(mem->Add(1, kTypeValue, "key", "value", nullptr)); + const auto leftovers = ListLeftovers(options, dbname_); + ASSERT_EQ(leftovers.size(), 1U); + const std::string leftover = leftovers[0] + ".review-leak"; + CopyFile(leftovers[0], leftover); + mem.reset(); + IntTblPropCollectorFactories collectors; + TableBuilderOptions tbo(ioptions, moptions, icmp, &collectors, + options.compression, options.compression_opts, 0, + "default", 0); + FileMetaData meta; + meta.fd = FileDescriptor(1, 0, 0); + meta.fd.smallest_seqno = 0; + meta.fd.largest_seqno = 1; + ASSERT_OK(options.memtable_factory->RecoverCrashSafeMemTableToSST( + leftover, &meta, tbo)); + std::ifstream maps("/proc/self/maps"); + ASSERT_TRUE(maps.good()); + const auto fname = TableFileName(options.cf_paths, 1, 0); + for (std::string line; std::getline(maps, line);) { + EXPECT_EQ(line.find(fname), std::string::npos) << "leaked mapping: " << line; + } +} + +#if !defined(OS_WIN) +TEST_F(CrashChild, DISABLED_SecondCrashSeed) { + ASSERT_EQ(arg_.size(), 3U); + Options options = BaseCrashSafeOptions(dbname_, true, arg_[1] == '1'); + if (arg_[0] == '1') SetupOsl(&options, true); + DB* db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &db)); + ASSERT_OK(db->Put(WriteOptions(), "a", "1")); + ASSERT_OK(static_cast(db)->TEST_SwitchMemtable()); + ASSERT_OK(db->Put(WriteOptions(), "b", "2")); + ASSERT_OK(static_cast(db)->TEST_SwitchMemtable()); + ASSERT_OK(db->Put(WriteOptions(), "c", "3")); + ::_exit(42); +} + +TEST_F(CrashChild, DISABLED_SecondCrashRecover) { + ASSERT_EQ(arg_.size(), 3U); + Options options = BaseCrashSafeOptions(dbname_, true, arg_[1] == '1'); + if (arg_[0] == '1') SetupOsl(&options, true); + SyncPoint::GetInstance()->SetCallBack( + "CrashSafeRecover::ConvertToSST:InjectStatus", [](void* arg) { + *static_cast(arg) = Status::IOError("review: failed conversion"); + }); + SyncPoint::GetInstance()->SetCallBack( + arg_[2] == '1' ? "DBImpl::Open:Opened" + : "DBImpl::RecoverLogFiles:BeforeReadWal", + [](void*) { ::_exit(43); }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* db = nullptr; + DB::Open(options, dbname_, &db).PermitUncheckedError(); + ::_exit(17); +} +#endif + +TEST_F(DBCsppCrashSafeTest, SecondCrashAfterConvertFailure) { + Close(); + for (bool osl : {false, true}) { + for (bool log_index : {false, true}) { + for (bool after_open : {false, true}) { + SCOPED_TRACE(osl ? "OSL" : "CSPP"); + SCOPED_TRACE(log_index); + SCOPED_TRACE(after_open); + Options options = BaseCrashSafeOptions(dbname_, true, log_index); + if (osl) SetupOsl(&options, true); + Destroy(options); + const std::string child_options = + std::to_string(osl) + std::to_string(log_index) + + std::to_string(after_open); + ASSERT_EQ(RunCrashChild(dbname_, "SecondCrashSeed", child_options), 42); + ASSERT_GE(ListLeftovers(options, dbname_).size(), 3U); + ASSERT_EQ(RunCrashChild(dbname_, "SecondCrashRecover", child_options), 43); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.generation & 1, after_open ? 0U : 1U); + ASSERT_OK(TryReopen(options)); + EXPECT_EQ(Get("a"), "1"); + EXPECT_EQ(Get("b"), "2"); + EXPECT_EQ(Get("c"), "3"); + ASSERT_OK(Put("after", "recovery")); + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.generation & 1, 0U); + Reopen(options); + EXPECT_EQ(Get("a"), "1"); + EXPECT_EQ(Get("b"), "2"); + EXPECT_EQ(Get("c"), "3"); + EXPECT_EQ(Get("after"), "recovery"); + Close(); + } + } + } +} + +#if !defined(OS_WIN) +TEST_F(CrashChild, DISABLED_PatriciaSingleWriterMmapConstructor) { + const auto mode = static_cast( + std::stoi(arg_)); + const std::string path = dbname_ + "/review-patricia-" + arg_; + alignas(terark::MainPatricia) unsigned char storage[sizeof(terark::MainPatricia)]; + std::memset(storage, 0xa5, sizeof(storage)); + auto* trie = new (storage) terark::MainPatricia( + 4, 16 << 20, mode, terark::fstring(path)); + trie->~MainPatricia(); + ::_exit(0); +} +#endif + +TEST_F(DBCsppCrashSafeTest, PatriciaSingleWriterMmapConstructor) { + Close(); + ASSERT_OK(env_->CreateDirIfMissing(dbname_)); + for (auto mode : {terark::Patricia::SingleThreadStrict, + terark::Patricia::SingleThreadShared, + terark::Patricia::OneWriteMultiRead}) { + SCOPED_TRACE(int(mode)); + ASSERT_EQ(RunCrashChild(dbname_, "PatriciaSingleWriterMmapConstructor", + std::to_string(int(mode))), 0); + } +} + +TEST_F(DBCsppCrashSafeTest, RecoverOffDoesNotCreatePublishedSeq) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, false, false); + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "v")); + Close(); + ASSERT_TRUE(env_->FileExists(CrashSafePubSeqFileName(dbname_)).IsNotFound()); +} + +TEST_F(DBCsppCrashSafeTest, RecoverOnCreatesPublishedSeqAndFlushWalBuffer) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.manual_wal_flush = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_FALSE(dbfull()->GetDBOptions().manual_wal_flush); + ASSERT_OK(Put("k", "v")); + ASSERT_TRUE(dbfull()->WALBufferIsEmpty()); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.generation % 2, 0U); + ASSERT_GT(rec.pubseq, 0U); + ASSERT_GT(rec.wal_offset, 0U); + Close(); + ASSERT_OK(env_->FileExists(CrashSafePubSeqFileName(dbname_))); +} + +TEST_F(DBCsppCrashSafeTest, SkipListPlusRecoverOpensAndUsesFullWal) { + Close(); + Options options = CurrentOptions(); + options.memtable_crash_safe_recover = true; + options.create_if_missing = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("a", "1")); + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("a"), "1"); +} + +#if !defined(OS_WIN) +static Options LogRefCrashOptions(const std::string& dbname, + const std::string& config) { + Options options = BaseCrashSafeOptions(dbname, true, true); + const bool osl = config[0] == 'O'; + const json params = { + {"mem_cap", 16777216}, {"convert_to_sst", "kFileMmap"}, + {"log_ref_format", config[1] == 'P' ? "kPlainLogRef" : "kShortLogRef"}}; + options.memtable_factory = EasyNewMemTableRep( + osl ? "OffsetSkipList" : "CSPPMemTab", params.dump()); + const SidePluginRepo repo; + options.table_factory = PluginFactorySP::AcquirePlugin( + osl ? "OffsetSkipListTable" : "CSPPMemTabTable", params, repo); + return options; +} + +TEST_F(CrashChild, DISABLED_LogRefRecoveryIgnoresCounters) { + Options options = LogRefCrashOptions(dbname_, arg_); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + const size_t count = arg_[2] == 'L' ? 5000 : 3; + for (size_t i = 0; i < count; ++i) { + ASSERT_OK(child_db->Put(WriteOptions(), std::to_string(i % 2), + std::string(128, 'v'))); + } + ASSERT_OK(child_db->Put(WriteOptions(), "inline", "v")); + ::_exit(42); +} + +TEST_F(DBCsppCrashSafeTest, LogRefRecoveryIgnoresCounters) { + for (const char* config : {"CPS", "CPL", "CSS", "CSL", + "OPS", "OPL", "OSS", "OSL"}) { + SCOPED_TRACE(config); + Close(); + Options options = LogRefCrashOptions(dbname_, config); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "LogRefRecoveryIgnoresCounters", config), 42); + const auto leftovers = ListLeftovers(options, dbname_); + ASSERT_EQ(leftovers.size(), 1U); + const int fd = ::open(leftovers[0].c_str(), O_RDONLY); + ASSERT_GE(fd, 0); + // Header statistics are not a source of truth for WAL references. + const size_t offset = config[0] == 'O' + ? offsetof(terark::OSL_MmapHeader, reserved) + 2 * sizeof(uint32_t) + : offsetof(terark::DFA_MmapHeader, reserved) + 4 * sizeof(uint32_t); + uint64_t wal[3]; // fileno, cnt, bytes + const ssize_t n = ::pread(fd, wal, sizeof(wal), offset); + ::close(fd); + ASSERT_EQ(n, static_cast(sizeof(wal))); + ASSERT_NE(wal[0], 0U); + ASSERT_EQ(wal[1], 0U); + ASSERT_EQ(wal[2], 0U); + uint64_t wal_size = 0; + ASSERT_OK(env_->GetFileSize(LogFileName(options.wal_dir, wal[0]), &wal_size)); + for (int reopen = 0; reopen < 2; ++reopen) { + ASSERT_OK(TryReopen(options)); + ASSERT_GT(CountL0(db_), 0); + ColumnFamilyMetaData cf_meta; + db_->GetColumnFamilyMetaData(&cf_meta); + ASSERT_EQ(cf_meta.blob_files.size(), 1U); + ASSERT_EQ(cf_meta.blob_files[0].total_blob_count, 1U); + ASSERT_EQ(cf_meta.blob_files[0].total_blob_bytes, wal_size); + ASSERT_EQ(Get("0"), std::string(128, 'v')); + ASSERT_EQ(Get("1"), std::string(128, 'v')); + ASSERT_EQ(Get("inline"), "v"); + Close(); + } + } +} + +TEST_F(CrashChild, DISABLED_ChangedFactoryWithOtherCfLeftover) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + Options other = options; + if (arg_ == "OSL") { + SetupOsl(&options, true); + } else { + SetupOsl(&other, true); + } + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ColumnFamilyHandle* handle = nullptr; + ASSERT_OK(child_db->CreateColumnFamily(other, "other", &handle)); + ASSERT_OK(child_db->Put(WriteOptions(), "key", "primary")); + ASSERT_OK(child_db->Put(WriteOptions(), handle, "key", "other")); + ::_exit(1); +} + +TEST_F(DBCsppCrashSafeTest, ChangedFactoryWithOtherCfLeftoverUsesFullWal) { + for (bool was_osl : {false, true}) { + SCOPED_TRACE(was_osl); + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (!was_osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "ChangedFactoryWithOtherCfLeftover", + was_osl ? "OSL" : "CSPP"), 1); + // Both CFs now use the other's factory, so default's old leftover is absent. + ASSERT_OK(TryReopenWithColumnFamilies({"default", "other"}, options)); + ASSERT_EQ(Get(0, "key"), "primary"); + ASSERT_EQ(Get(1, "key"), "other"); + } +} + +TEST_F(CrashChild, DISABLED_OslLeftoverWithWriterLock) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + SetupOsl(&options, true); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + const std::string key = "locked-osl-key"; + ASSERT_OK(child_db->Put(WriteOptions(), key, "value")); + const auto leftovers = ListLeftovers(options, dbname_); + ASSERT_EQ(leftovers.size(), 1U); + int fd = ::open(leftovers[0].c_str(), O_RDWR); + ASSERT_GE(fd, 0); + char data[4096]; + const ssize_t n = ::pread(fd, data, sizeof(data), 0); + ASSERT_GT(n, 0); + const char* p = std::search(data, data + n, key.begin(), key.end()); + ASSERT_NE(p, data + n); + ASSERT_GE(p - data, 12); + ASSERT_EQ(DecodeFixed32(p - 4), key.size()); + // ValueVec precedes the length-prefixed key. Simulate InsertDup holding its + // lock while the published COW array is still available to readers. + ASSERT_EQ(DecodeFixed32(p - 12), 1U); + char locked_num[4]; + EncodeFixed32(locked_num, 0x80000001U); + ASSERT_EQ(::pwrite(fd, locked_num, sizeof(locked_num), p - 12 - data), 4); + ::close(fd); + std::string value; + ASSERT_OK(child_db->Get(ReadOptions(), key, &value)); + ASSERT_EQ(value, "value"); + std::unique_ptr it(child_db->NewIterator(ReadOptions())); + it->SeekToFirst(); + ASSERT_TRUE(it->Valid()); + ASSERT_EQ(it->key(), key); + ASSERT_EQ(it->value(), "value"); + ::_exit(1); +} + +TEST_F(DBCsppCrashSafeTest, OslLeftoverWithWriterLock) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + SetupOsl(&options, true); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "OslLeftoverWithWriterLock"), 1); + for (int i = 0; i != 2; ++i) { + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("locked-osl-key"), "value"); + std::unique_ptr it(db_->NewIterator(ReadOptions())); + it->SeekToFirst(); + ASSERT_TRUE(it->Valid()); + ASSERT_EQ(it->key(), "locked-osl-key"); + ASSERT_EQ(it->value(), "value"); + it.reset(); + Close(); + } +} + +TEST_F(CrashChild, DISABLED_MixedDumpMemUsesFullWal) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + const bool osl = arg_ == "OSL"; + if (osl) SetupOsl(&options, true); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + Options dump = options; + dump.memtable_factory = EasyNewMemTableRep( + osl ? "OffsetSkipList" : "CSPPMemTab", + R"({"mem_cap":16777216,"convert_to_sst":"kDumpMem"})"); + ColumnFamilyHandle* handle = nullptr; + ASSERT_OK(child_db->CreateColumnFamily(dump, "dump", &handle)); + ASSERT_OK(child_db->Put(WriteOptions(), "mmap", "1")); + ASSERT_OK(child_db->Put(WriteOptions(), handle, "dump", "2")); + ::_exit(1); +} + +TEST_F(DBCsppCrashSafeTest, MixedDumpMemUsesFullWal) { + for (bool osl : {false, true}) { + SCOPED_TRACE(osl); + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "MixedDumpMemUsesFullWal", + osl ? "OSL" : "CSPP"), 1); + Options dump = options; + dump.memtable_factory = EasyNewMemTableRep( + osl ? "OffsetSkipList" : "CSPPMemTab", + R"({"mem_cap":16777216,"convert_to_sst":"kDumpMem"})"); + ASSERT_OK(TryReopenWithColumnFamilies( + {"default", "dump"}, std::vector{options, dump})); + ASSERT_EQ(Get(0, "mmap"), "1"); + ASSERT_EQ(Get(1, "dump"), "2"); + } +} + +TEST_F(CrashChild, DISABLED_CloseConvertsLeftovers) { + ASSERT_EQ(arg_.size(), 2U); + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (arg_[0] == '1') SetupOsl(&options, true); + options.atomic_flush = arg_[1] == '1'; + options.min_write_buffer_number_to_merge = 3; + DB* db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &db)); + const auto closing_thread = std::this_thread::get_id(); + std::atomic converts{0}; + SyncPoint::GetInstance()->SetCallBack( + "FlushJob::ConvertToSST:Status", [&](void*) { + EXPECT_NE(std::this_thread::get_id(), closing_thread); + converts.fetch_add(1); + }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(db->Put(WriteOptions(), "k1", "v1")); + ASSERT_OK(static_cast(db)->TEST_SwitchMemtable()); + ASSERT_OK(db->Put(WriteOptions(), "k2", "v2")); + ASSERT_OK(db->Close()); + delete db; + ASSERT_GE(converts.load(), 2); + ASSERT_FALSE(HasFailure()); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, CloseConvertsLeftovers) { + Close(); + for (bool osl : {false, true}) { + for (bool atomic : {false, true}) { + SCOPED_TRACE(osl); + SCOPED_TRACE(atomic); + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + options.atomic_flush = atomic; + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "CloseConvertsLeftovers", + std::to_string(osl) + std::to_string(atomic)), 0); + ASSERT_TRUE(ListLeftovers(options, dbname_).empty()); + ASSERT_OK(env_->FileExists(CrashSafePubSeqFileName(dbname_))); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("k1"), "v1"); + ASSERT_EQ(Get("k2"), "v2"); + ASSERT_GE(CountL0(db_), 1); + Close(); + ASSERT_TRUE(ListLeftovers(options, dbname_).empty()); + } + } +} +#endif + +TEST_F(DBCsppCrashSafeTest, AvoidFlushDuringShutdownLeavesNoLeftoverThenWal) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.avoid_flush_during_shutdown = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "v")); + Close(); + ASSERT_TRUE(ListLeftovers(options, dbname_).empty()); + ASSERT_OK(env_->FileExists(CrashSafePubSeqFileName(dbname_))); + options.avoid_flush_during_shutdown = false; + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("k"), "v"); +} + +TEST_F(DBCsppCrashSafeTest, AvoidFlushCloseReopenDoesNotProbeWal) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.avoid_flush_during_shutdown = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "v")); + Close(); + int probes = 0; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::RecoverLogFiles:ProbeWalFormat", + [&probes](void*) { probes++; }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(options)); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_EQ(probes, 0); + ASSERT_EQ(Get("k"), "v"); +} + +TEST_F(DBCsppCrashSafeTest, FreshSidecarProbesOnlyOlderWal) { + Close(); + Options off = BaseCrashSafeOptions(dbname_, false, false); + off.avoid_flush_during_shutdown = true; + Destroy(off); + ASSERT_OK(TryReopen(off)); + ASSERT_OK(Put("a", "1")); + Close(); + Options on = BaseCrashSafeOptions(dbname_, true, false); + on.avoid_flush_during_shutdown = true; + std::vector probed; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::RecoverLogFiles:ProbeWalFormat", [&probed](void* arg) { + probed.push_back(*static_cast(arg)); + }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(on)); + ASSERT_EQ(probed.size(), 1U); + ASSERT_OK(Put("b", "2")); + Close(); + probed.clear(); + ASSERT_OK(TryReopen(on)); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_GT(rec.kind_since_wal, 0U); + ASSERT_EQ(probed.size(), 1U); + ASSERT_LT(probed[0], rec.kind_since_wal); + ASSERT_EQ(Get("a"), "1"); + ASSERT_EQ(Get("b"), "2"); +} + +TEST_F(DBCsppCrashSafeTest, ClassicWalToLogIndexWithoutSidecarIsNotSupported) { + Close(); + std::string value(100000, '\0'); + for (size_t i = 0; i < value.size(); ++i) { + value[i] = static_cast('a' + i % 23); + } + for (bool osl : {false, true}) { + SCOPED_TRACE(osl ? "OSL" : "CSPP"); + for (bool paranoid : {false, true}) { + SCOPED_TRACE(paranoid); + Options classic = BaseCrashSafeOptions(dbname_, false, false); + if (osl) SetupOsl(&classic, true); + classic.avoid_flush_during_shutdown = true; + classic.paranoid_checks = paranoid; + Destroy(classic); + ASSERT_OK(TryReopen(classic)); + ASSERT_OK(Put("large", value)); + Close(); + ASSERT_TRUE(env_->FileExists(CrashSafePubSeqFileName(dbname_)).IsNotFound()); + + Options log_index = classic; + log_index.memtable_crash_safe_recover = true; + log_index.memtable_as_log_index = true; + for (int attempt = 0; attempt < 2; ++attempt) { + const Status s = TryReopen(log_index); + ASSERT_TRUE(s.IsNotSupported()) << s.ToString(); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.wal_offset_kind, 2U); + ASSERT_EQ(rec.kind_since_wal, 0U); + ASSERT_EQ(rec.generation & 1, 1U); + } + + // Failed Open must not retain the unestablished log-index kind. + // Reopen in the original format without deleting the sidecar. + classic.memtable_crash_safe_recover = true; + ASSERT_OK(TryReopen(classic)); + ASSERT_EQ(Get("large"), value); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.wal_offset_kind, 1U); + ASSERT_GT(rec.kind_since_wal, 0U); + Close(); + } + } +} + +TEST_F(DBCsppCrashSafeTest, LogIndexWalWithoutSidecarRecoversAsClassic) { + Close(); + const std::string value(100000, 'v'); + for (bool osl : {false, true}) { + SCOPED_TRACE(osl ? "OSL" : "CSPP"); + Options options = BaseCrashSafeOptions(dbname_, false, true); + if (osl) SetupOsl(&options, true); + options.avoid_flush_during_shutdown = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("large", value)); + Close(); + ASSERT_TRUE(env_->FileExists(CrashSafePubSeqFileName(dbname_)).IsNotFound()); + + options.memtable_crash_safe_recover = true; + options.memtable_as_log_index = false; + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("large"), value); + Close(); + } +} + +TEST_F(DBCsppCrashSafeTest, RecoverOffDeletesStaleSidecar) { + Close(); + Options on = BaseCrashSafeOptions(dbname_, true, false); + on.avoid_flush_during_shutdown = true; + Destroy(on); + ASSERT_OK(TryReopen(on)); + ASSERT_OK(Put("k", "v")); + Close(); + ASSERT_OK(env_->FileExists(CrashSafePubSeqFileName(dbname_))); + Options off = BaseCrashSafeOptions(dbname_, false, false); + ASSERT_OK(TryReopen(off)); + ASSERT_TRUE(env_->FileExists(CrashSafePubSeqFileName(dbname_)).IsNotFound()); + ASSERT_EQ(Get("k"), "v"); +} + +TEST_F(DBCsppCrashSafeTest, OddGenerationPublishesEvenAfterWalRecovery) { + for (bool log_index : {false, true}) { + SCOPED_TRACE(log_index); + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, log_index); + options.avoid_flush_during_shutdown = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("before", "recovery")); + Close(); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_TRUE(SetPublishedSeqGeneration(dbname_, rec.generation | 1)); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("before"), "recovery"); +#if !defined(__AVX__) || defined(__clang__) + int odd = 0; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterOddGeneration", [&](void*) { + PublishedSeqOnDisk in_progress; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &in_progress)); + ASSERT_EQ(in_progress.generation % 2, 1U); + ++odd; + }); + SyncPoint::GetInstance()->EnableProcessing(); +#endif + ASSERT_OK(Put("after", "recovery")); +#if !defined(__AVX__) || defined(__clang__) + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_EQ(odd, 1); +#endif + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.generation % 2, 0U); + ASSERT_EQ(rec.pubseq, dbfull()->GetLatestSequenceNumber()); + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("before"), "recovery"); + ASSERT_EQ(Get("after"), "recovery"); + } +} + +TEST_F(DBCsppCrashSafeTest, LogIndexOffRecoverStillConverts) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("only", "recover")); + ASSERT_OK(dbfull()->TEST_SwitchMemtable()); + ASSERT_OK(Put("tail", "wal")); + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("only"), "recover"); + ASSERT_EQ(Get("tail"), "wal"); +} + +TEST_F(DBCsppCrashSafeTest, AtomicFlushCloseConverts) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.atomic_flush = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "v")); + Close(); + ASSERT_TRUE(ListLeftovers(options, dbname_).empty()); + ASSERT_OK(env_->FileExists(CrashSafePubSeqFileName(dbname_))); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("k"), "v"); + ASSERT_GE(CountL0(db_), 1); +} + +#if !defined(OS_WIN) +TEST_F(CrashChild, DISABLED_AtomicFlushAfterCommitExitConverts) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.atomic_flush = true; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", + [](void* /*arg*/) { ::_exit(1); }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "pub", "yes")); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, AtomicFlushAfterCommitExitConverts) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.atomic_flush = true; + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "AtomicFlushAfterCommitExitConverts"), 1); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("pub"), "yes"); + ASSERT_GE(CountL0(db_), 1); +} + +TEST_F(CrashChild, DISABLED_AfterCommitExitConvertsAndReopens) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", + [](void* /*arg*/) { ::_exit(1); }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "pub", "yes")); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, AfterCommitExitConvertsAndReopens) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "AfterCommitExitConvertsAndReopens"), 1); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("pub"), "yes"); + ColumnFamilyMetaData metadata; + db_->GetColumnFamilyMetaData(&metadata); + ASSERT_EQ(metadata.levels[0].files.size(), 1U); + ASSERT_TRUE(metadata.levels[0].files[0].marked_for_compaction); +} + +// Clang is excluded from the AVX store and still publishes through the odd +// generation, so it runs the same crash-injection tests as a non-AVX build. +#if !defined(__AVX__) || defined(__clang__) +TEST_F(CrashChild, DISABLED_AfterOddGenerationFallsBackToWal) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterOddGeneration", + [](void* /*arg*/) { ::_exit(1); }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "odd", "wal")); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, AfterOddGenerationFallsBackToWal) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "AfterOddGenerationFallsBackToWal"), 1); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("odd"), "wal"); +} +#endif // !__AVX__ || __clang__ + +TEST_F(CrashChild, DISABLED_AfterWriteToWALBeforePublishKeepsWalTail) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::WriteImpl:AfterWriteToWALBeforePublish", + [](void*) { ::_exit(1); }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(child_db->Put(WriteOptions(), "tail", "2")); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, AfterWriteToWALBeforePublishKeepsWalTail) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + { + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("first", "1")); + Close(); + } + ASSERT_EQ(RunCrashChild(dbname_, "AfterWriteToWALBeforePublishKeepsWalTail"), 1); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("first"), "1"); + ASSERT_EQ(Get("tail"), "2"); +} + +TEST_F(CrashChild, DISABLED_ConvertInjectFailureFallsBackToWal) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", + [](void*) { ::_exit(1); }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "inj", "ok")); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, ConvertInjectFailureFallsBackToWal) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "ConvertInjectFailureFallsBackToWal"), 1); + SyncPoint::GetInstance()->EnableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + SyncPoint::GetInstance()->SetCallBack( + "CrashSafeRecover::ConvertToSST:InjectStatus", [](void* arg) { + *static_cast(arg) = Status::IOError("inject convert"); + }); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("inj"), "ok"); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); +} + +TEST_F(CrashChild, DISABLED_SeekInjectFailureStillConverts) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", + [](void*) { ::_exit(1); }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "seek", "ok")); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, SeekInjectFailureStillConverts) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "SeekInjectFailureStillConverts"), 1); + SyncPoint::GetInstance()->EnableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + SyncPoint::GetInstance()->SetCallBack( + "CrashSafeRecover::SeekToFileOffset:InjectStatus", [](void* arg) { + *static_cast(arg) = IOStatus::IOError("inject seek"); + }); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("seek"), "ok"); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); +} + +class SeekFailureFS : public FileSystemWrapper { + public: + SeekFailureFS(const std::shared_ptr& fs, std::string path, + bool fail_read) + : FileSystemWrapper(fs), path_(std::move(path)), fail_read_(fail_read) {} + const char* Name() const override { return "SeekFailureFS"; } + bool armed = false; + int failures = 0; + int wal_opens = 0; + + class File : public FSSequentialFileOwnerWrapper { + public: + File(std::unique_ptr&& file, SeekFailureFS* fs) + : FSSequentialFileOwnerWrapper(std::move(file)), fs_(fs) {} + IOStatus Read(size_t n, const IOOptions& opts, Slice* result, char* scratch, + IODebugContext* dbg) override { + if (fs_->armed && fs_->fail_read_) { + fs_->armed = false; + IOStatus s = FSSequentialFileOwnerWrapper::Read( + std::min(n, 4), opts, result, scratch, dbg); + if (!s.ok()) return s; + fs_->failures++; + return IOStatus::IOError("seek read failed after consuming bytes"); + } + return FSSequentialFileOwnerWrapper::Read(n, opts, result, scratch, dbg); + } + IOStatus Skip(uint64_t n) override { + IOStatus s = FSSequentialFileOwnerWrapper::Skip(n); + if (s.ok() && n != 0 && fs_->armed && !fs_->fail_read_) { + fs_->armed = false; + fs_->failures++; + return IOStatus::IOError("seek skip failed after advancing file"); + } + return s; + } + private: + SeekFailureFS* fs_; + }; + + IOStatus NewSequentialFile(const std::string& fname, const FileOptions& opts, + std::unique_ptr* file, + IODebugContext* dbg) override { + IOStatus s = FileSystemWrapper::NewSequentialFile(fname, opts, file, dbg); + if (s.ok() && fname == path_) { + wal_opens++; + *file = std::make_unique(std::move(*file), this); + } + return s; + } + private: + const std::string path_; + const bool fail_read_; +}; + +TEST_F(CrashChild, DISABLED_SeekIOFailureReopensWal) { + Options options = BaseCrashSafeOptions(dbname_, true, arg_ == "log-index"); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "prefix", + std::string(log::kBlockSize * 2, 'p'))); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::WriteImpl:AfterWriteToWALBeforePublish", + [](void*) { ::_exit(42); }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(child_db->Put(WriteOptions(), "tail", "unpublished")); + ::_exit(1); +} + +TEST_F(DBCsppCrashSafeTest, SeekIOFailureReopensWal) { + Close(); + for (const std::string mode : {"read", "skip", "log-index"}) { + SCOPED_TRACE(mode); + Options options = BaseCrashSafeOptions(dbname_, true, mode == "log-index"); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "SeekIOFailureReopensWal", mode), 42); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + auto fs = std::make_shared( + options.env->GetFileSystem(), LogFileName(options.wal_dir, rec.wal_number), + mode == "read"); + auto fault_env = NewCompositeEnv(fs); + options.env = fault_env.get(); + options.log_readahead_size = 0; + SyncPoint::GetInstance()->SetCallBack( + "CrashSafeRecover::SeekToFileOffset:InjectStatus", + [&](void*) { fs->armed = true; }); + SyncPoint::GetInstance()->EnableProcessing(); + const Status s = TryReopen(options); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + if (s.ok()) { + EXPECT_EQ(Get("prefix"), std::string(log::kBlockSize * 2, 'p')); + EXPECT_EQ(Get("tail"), "unpublished"); + } + Close(); + ASSERT_OK(s); + ASSERT_EQ(fs->failures, 1); + ASSERT_GE(fs->wal_opens, 2); + } +} + +TEST_F(DBCsppCrashSafeTest, TornStampKindZeroRestampsAndProbesWal) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.avoid_flush_during_shutdown = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "v")); + Close(); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.magic, 0x5145534255505343ULL); + ASSERT_EQ(rec.wal_offset_kind, 1U); + ASSERT_TRUE(SetPublishedSeqWalOffsetKind(dbname_, 0)); + int probes = 0; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::RecoverLogFiles:ProbeWalFormat", + [&probes](void*) { probes++; }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(options)); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_GT(probes, 0); + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.wal_offset_kind, 1U); + ASSERT_EQ(Get("k"), "v"); +} + +TEST_F(CrashChild, DISABLED_UninitializedPublishedSeqStillOpens) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::MapPublishedSeqFile:AfterMmapBeforeStamp", + [](void*) { ::_exit(1); }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, UninitializedPublishedSeqStillOpens) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "UninitializedPublishedSeqStillOpens"), 1); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("after", "stamp")); + ASSERT_EQ(Get("after"), "stamp"); +} + +TEST_F(CrashChild, DISABLED_ZeroPublishedSeqWithLeftoverFallsBackToWal) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", + [](void*) { ::_exit(1); }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "k", "v")); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, ZeroPublishedSeqWithLeftoverFallsBackToWal) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "ZeroPublishedSeqWithLeftoverFallsBackToWal"), 1); + const auto before = ListLeftovers(options, dbname_); + ASSERT_FALSE(before.empty()); + ASSERT_TRUE(ZeroPublishedSeqFile(dbname_)); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("k"), "v"); + for (const auto& path : before) { + ASSERT_TRUE(env_->FileExists(path).IsNotFound()); + } +} + +TEST_F(CrashChild, DISABLED_LogIndexSeekInjectKeepsUnpublishedWalTail) { + Options options = BaseCrashSafeOptions(dbname_, true, true); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::WriteImpl:AfterWriteToWALBeforePublish", + [](void*) { ::_exit(1); }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(child_db->Put(WriteOptions(), "b", "2")); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, LogIndexSeekInjectKeepsUnpublishedWalTail) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, true); + Destroy(options); + { + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("a", "1")); + Close(); + } + ASSERT_EQ(RunCrashChild(dbname_, "LogIndexSeekInjectKeepsUnpublishedWalTail"), 1); + SyncPoint::GetInstance()->EnableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + SyncPoint::GetInstance()->SetCallBack( + "CrashSafeRecover::SeekToFileOffset:InjectStatus", [](void* arg) { + *static_cast(arg) = IOStatus::IOError("inject seek"); + }); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("a"), "1"); + ASSERT_EQ(Get("b"), "2"); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); +} +#endif // !OS_WIN + +TEST_F(DBCsppCrashSafeTest, TwoWriteQueuesWalTail) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.two_write_queues = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("q", "1")); + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("q"), "1"); +} + +TEST_F(DBCsppCrashSafeTest, WriteCommittedSkipPrepare) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + TransactionDBOptions txn_opts; + txn_opts.write_policy = TxnDBWritePolicy::WRITE_COMMITTED; + TransactionDB* txn_db = nullptr; + ASSERT_OK(TransactionDB::Open(options, txn_opts, dbname_, &txn_db)); + Transaction* txn = txn_db->BeginTransaction(WriteOptions()); + ASSERT_OK(txn->Put("wc", "1")); + ASSERT_OK(txn->Commit()); + delete txn; + delete txn_db; + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("wc"), "1"); +} + +TEST_F(DBCsppCrashSafeTest, OslFileMmapRecoverRoundTrip) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + SetupOsl(&options, true); + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("osl", "v")); + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("osl"), "v"); +} + +TEST_F(DBCsppCrashSafeTest, LogIndexHeaderHasWalRefWithoutRecover) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, false, true); + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", std::string(64, 'v'))); + auto leftovers = ListLeftovers(options, dbname_); + ASSERT_FALSE(leftovers.empty()); + const int fd = ::open(leftovers[0].c_str(), O_RDONLY); + ASSERT_GE(fd, 0); + terark::DFA_MmapHeader hdr{}; + ASSERT_EQ(::pread(fd, &hdr, sizeof(hdr), 0), + static_cast(sizeof(hdr))); + ::close(fd); + const auto* cs = + reinterpret_cast(hdr.reserved); + ASSERT_EQ(cs[0], 0x50505343U); + ASSERT_NE(cs[1], 0U); + ASSERT_GT(cs[2], 0U); +} + +#if !defined(OS_WIN) +TEST_F(CrashChild, DISABLED_DisableWalIsForcedOff) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + std::atomic hits{0}; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::WriteImpl:AfterWriteToWALBeforePublish", [&hits](void*) { + if (hits.fetch_add(1) >= 1) { + ::_exit(1); + } + }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "pad", "p")); + WriteOptions child_wo; + child_wo.disableWAL = true; + ASSERT_OK(child_db->Put(child_wo, "d", "4")); + ::_exit(0); +} +#endif + +TEST_F(DBCsppCrashSafeTest, DisableWalIsForcedOff) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("a", "1")); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + const uint64_t wal_number = rec.wal_number; + const uint64_t wal_offset = rec.wal_offset; + const uint64_t pubseq = rec.pubseq; + ASSERT_GT(wal_number, 0U); + ASSERT_GT(wal_offset, 0U); + WriteOptions wo; + wo.disableWAL = true; + ASSERT_OK(Put("b", "2", wo)); + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.wal_number, wal_number); + ASSERT_GT(rec.wal_offset, wal_offset); + ASSERT_GT(rec.pubseq, pubseq); + ASSERT_EQ(Get("a"), "1"); + ASSERT_EQ(Get("b"), "2"); +#if !defined(OS_WIN) + Close(); + Destroy(options); + { + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("c", "3")); + Close(); + } + ASSERT_EQ(RunCrashChild(dbname_, "DisableWalIsForcedOff"), 1); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("c"), "3"); + ASSERT_EQ(Get("pad"), "p"); + ASSERT_EQ(Get("d"), "4"); +#endif +} + +#if !defined(OS_WIN) +TEST_F(CrashChild, DISABLED_UnorderedWritePublishesPreviousWalCursor) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.unordered_write = true; + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "pad", "p")); + ASSERT_OK(child_db->Put(WriteOptions(), "d", "4")); + ::_exit(1); +} +#endif + +TEST_F(DBCsppCrashSafeTest, UnorderedWritePublishesPreviousWalCursor) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.unordered_write = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("a", "1")); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.pubseq, 0U); + ASSERT_OK(Put("b", "2")); + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_GT(rec.pubseq, 0U); + ASSERT_GT(rec.wal_offset, 0U); + const uint64_t after_b_seq = rec.pubseq; + const uint64_t after_b_off = rec.wal_offset; + ASSERT_OK(Put("c", "3")); + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_GT(rec.pubseq, after_b_seq); + ASSERT_GT(rec.wal_offset, after_b_off); + ASSERT_EQ(Get("a"), "1"); + ASSERT_EQ(Get("b"), "2"); + ASSERT_EQ(Get("c"), "3"); +#if !defined(OS_WIN) + Close(); + Destroy(options); + { + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "0")); + Close(); + } + ASSERT_EQ(RunCrashChild(dbname_, "UnorderedWritePublishesPreviousWalCursor"), 1); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("k"), "0"); + ASSERT_EQ(Get("pad"), "p"); + ASSERT_EQ(Get("d"), "4"); +#endif +} + +TEST_F(DBCsppCrashSafeTest, UnorderedWriteSyncFailureDoesNotBlockFlushOrClose) { + Close(); + for (bool recover : {false, true}) { + for (bool osl : {false, true}) { + SCOPED_TRACE(recover); + SCOPED_TRACE(osl); + Options options = BaseCrashSafeOptions(dbname_, recover, false); + if (osl) SetupOsl(&options, true); + options.unordered_write = true; + Destroy(options); + auto fs = std::make_shared(FileSystem::Default()); + auto fault_env = NewCompositeEnv(fs); + options.env = fault_env.get(); + DB* raw_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &raw_db)); + std::unique_ptr db(raw_db); + ASSERT_OK(db->Put(WriteOptions(), "seed", "value")); + SyncPoint::GetInstance()->SetCallBack("DBImpl::SyncWAL:Begin", [&](void*) { + fs->SetFilesystemActive(false, IOStatus::IOError("WAL sync failure")); + }); + SyncPoint::GetInstance()->EnableProcessing(); + WriteOptions wo; + wo.sync = true; + Status s = db->Put(wo, "failed", "value"); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + fs->SetFilesystemActive(true); + ASSERT_TRUE(s.IsIOError()); + // A WAL sync error stops the DB, but Flush and Close must still return. + ASSERT_TRUE(db->Flush(FlushOptions()).IsIOError()); + ASSERT_TRUE(db->Close().IsIOError()); + } + } +} + +TEST_F(DBCsppCrashSafeTest, UnorderedWritePreReleaseFailureAccountsWholeGroup) { + class FailPreRelease : public PreReleaseCallback { + public: + size_t calls = 0; + Status Callback(SequenceNumber, bool, uint64_t, size_t, size_t total) override { + EXPECT_EQ(total, 2U); + ++calls; + return Status::Busy("pre-release failure"); + } + }; + for (bool recover : {false, true}) { + for (bool osl : {false, true}) { + SCOPED_TRACE(recover); + SCOPED_TRACE(osl); + Close(); + Options options = BaseCrashSafeOptions(dbname_, recover, false); + if (osl) SetupOsl(&options, true); + options.unordered_write = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("seed", "value")); + // Hold the leader until the other writer has joined the same group. + FailPreRelease callback; + SyncPoint::GetInstance()->LoadDependency({ + {"UnorderedWriteFailure:FollowerJoined", "UnorderedWriteFailure:Leader"}}); + SyncPoint::GetInstance()->SetCallBack( + "WriteThread::JoinBatchGroup:Wait", [&callback](void* arg) { + auto* w = static_cast(arg); + w->pre_release_callback = &callback; + if (w->state == WriteThread::STATE_GROUP_LEADER) { + TEST_SYNC_POINT("UnorderedWriteFailure:Leader"); + } else { + TEST_SYNC_POINT("UnorderedWriteFailure:FollowerJoined"); + } + }); + SyncPoint::GetInstance()->EnableProcessing(); + Status status[2]; + std::thread writers[2]; + for (size_t i = 0; i != 2; ++i) { + writers[i] = std::thread([&, i] { + WriteBatch batch; + ASSERT_OK(batch.Put("failed" + std::to_string(i), "value")); + status[i] = db_->Write(WriteOptions(), &batch); + }); + } + for (auto& writer : writers) writer.join(); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + SyncPoint::GetInstance()->LoadDependency({}); + ASSERT_EQ(callback.calls, 1U); + for (const auto& s : status) ASSERT_TRUE(s.IsBusy()); + ASSERT_EQ(Get("failed0"), "NOT_FOUND"); + ASSERT_EQ(Get("failed1"), "NOT_FOUND"); + ASSERT_OK(Put("after", "value")); + ASSERT_OK(Put("tail", "value")); + if (recover) { + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.pubseq, db_->GetLatestSequenceNumber() - 1); + } + ASSERT_OK(Flush()); + ASSERT_OK(db_->Close()); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("seed"), "value"); + ASSERT_EQ(Get("after"), "value"); + ASSERT_EQ(Get("tail"), "value"); + } + } +} + +#if !defined(OS_WIN) +TEST_F(CrashChild, DISABLED_PipelinedWriteUsesStagedWalCursor) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.enable_pipelined_write = true; + options.merge_operator = MergeOperators::CreateStringAppendOperator(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "pad", "p")); + ASSERT_OK(child_db->Merge(WriteOptions(), "m2", "y")); + ASSERT_OK(child_db->Put(WriteOptions(), "d", "4")); + ::_exit(1); +} +#endif + +TEST_F(DBCsppCrashSafeTest, PipelinedWriteUsesStagedWalCursor) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.enable_pipelined_write = true; + options.merge_operator = MergeOperators::CreateStringAppendOperator(); + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("a", "1")); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.pubseq, 0U); + { + std::thread t1([&] { ASSERT_OK(Merge("m", "x")); }); + std::thread t2([&] { ASSERT_OK(Put("b", "2")); }); + t1.join(); + t2.join(); + } + ASSERT_OK(Put("c", "3")); + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_GT(rec.pubseq, 0U); + ASSERT_GT(rec.wal_number, 0U); + ASSERT_GT(rec.wal_offset, 0U); + ASSERT_EQ(Get("a"), "1"); + ASSERT_EQ(Get("b"), "2"); + ASSERT_EQ(Get("c"), "3"); + ASSERT_EQ(Get("m"), "x"); +#if !defined(OS_WIN) + Close(); + Destroy(options); + { + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "0")); + Close(); + } + ASSERT_EQ(RunCrashChild(dbname_, "PipelinedWriteUsesStagedWalCursor"), 1); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("k"), "0"); + ASSERT_EQ(Get("pad"), "p"); + ASSERT_EQ(Get("m2"), "y"); + ASSERT_EQ(Get("d"), "4"); +#endif +} + +#if !defined(OS_WIN) +TEST_F(CrashChild, DISABLED_LeftoverOnDbPathNotCfPaths0) { + const std::string l0 = dbname_ + "/l0data"; + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.db_paths = {{l0, 1ULL << 30}}; + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "pad", "p")); + ASSERT_OK(child_db->Put(WriteOptions(), "x", "1")); + ::_exit(1); +} +#endif + +TEST_F(DBCsppCrashSafeTest, LeftoverOnDbPathNotCfPaths0) { + Close(); + const std::string l0 = dbname_ + "/l0data"; + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.db_paths = {{l0, 1ULL << 30}}; + Destroy(options); + ASSERT_OK(env_->CreateDirIfMissing(dbname_)); + ASSERT_OK(env_->CreateDirIfMissing(l0)); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "0")); + Close(); + ASSERT_TRUE(ListLeftovers(options, dbname_).empty()); +#if !defined(OS_WIN) + ASSERT_EQ(RunCrashChild(dbname_, "LeftoverOnDbPathNotCfPaths0"), 1); + auto leftovers_l0 = ListLeftovers(options, l0); + ASSERT_FALSE(leftovers_l0.empty()); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("k"), "0"); + ASSERT_EQ(Get("pad"), "p"); + ASSERT_EQ(Get("x"), "1"); + for (const auto& leftover_path : leftovers_l0) { + ASSERT_TRUE(env_->FileExists(leftover_path).IsNotFound()); + } +#endif +} + +TEST_F(DBCsppCrashSafeTest, EmptyWriteUpdatesWalCursor) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "v")); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + const uint64_t pubseq = rec.pubseq; + const uint64_t wal_number = rec.wal_number; + const uint64_t wal_offset = rec.wal_offset; + ASSERT_GT(pubseq, 0U); + WriteBatch empty; + ASSERT_OK(db_->Write(WriteOptions(), &empty)); + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.pubseq, pubseq); + ASSERT_EQ(rec.wal_number, wal_number); + ASSERT_GT(rec.wal_offset, wal_offset); +} + +TEST_F(DBCsppCrashSafeTest, PublishedSeqFieldsAfterPut) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + ASSERT_OK(TryReopen(options)); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.wal_offset_kind, 1U); // kPhysical, stamped on create + ASSERT_EQ(rec.pubseq, 0U); + ASSERT_OK(Put("k", "v")); + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.generation % 2, 0U); + ASSERT_EQ(rec.pubseq, dbfull()->GetLatestSequenceNumber()); + ASSERT_GT(rec.wal_number, 0U); + ASSERT_GT(rec.wal_offset, 0U); + ASSERT_EQ(rec.wal_offset_kind, 1U); // Persist does not rewrite kind +} + +TEST_F(DBCsppCrashSafeTest, DontConvertFallsBackToWal) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + SetupCspp(&options, false); + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "v")); + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("k"), "v"); +} + +TEST_F(DBCsppCrashSafeTest, BestEffortsRecoveryFallsBackToWal) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.best_efforts_recovery = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "v")); + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("k"), "v"); +} + +TEST_F(DBCsppCrashSafeTest, LogIndexOnRecoverOffUsesWal) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, false, true); + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "v")); + Close(); + ASSERT_TRUE(env_->FileExists(CrashSafePubSeqFileName(dbname_)).IsNotFound()); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("k"), "v"); +} + +TEST_F(DBCsppCrashSafeTest, ListLeftoversAdvancesFileNumber) { + for (bool osl : {false, true}) { + SCOPED_TRACE(osl ? "OSL" : "CSPP"); + Close(); + Options options = BaseCrashSafeOptions(dbname_, false, false); + if (osl) { + SetupOsl(&options, true); + } + Destroy(options); + ASSERT_OK(env_->CreateDirIfMissing(dbname_)); + const std::string prefix = dbname_ + (osl ? "/OffsetSkipList-" : "/cspp-"); + const std::string high = prefix + "000100.memtab-0"; + const std::string low = prefix + "000010.memtab-0"; + ASSERT_OK(WriteStringToFile(env_, "", high)); + ASSERT_OK(WriteStringToFile(env_, "", low)); + ASSERT_EQ(ListLeftovers(options, dbname_).size(), 2U); + // Remove both so collision checks cannot hide a stale counter. + ASSERT_OK(env_->DeleteFile(high)); + ASSERT_OK(env_->DeleteFile(low)); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(ListLeftovers(options, dbname_), + std::vector{prefix + "000101.memtab-0"}); + Close(); + ASSERT_TRUE(ListLeftovers(options, dbname_).empty()); + // Empty and lower-number scans must not move the counter backwards. + ASSERT_OK(WriteStringToFile(env_, "", low)); + ASSERT_EQ(ListLeftovers(options, dbname_).size(), 1U); + ASSERT_OK(env_->DeleteFile(low)); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(ListLeftovers(options, dbname_), + std::vector{prefix + "000102.memtab-0"}); + Close(); + Destroy(options); + } +} + +TEST_F(DBCsppCrashSafeTest, CreateMemTableRepDoesNotOverwriteLeftover) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("keep", "me")); + auto first = ListLeftovers(options, dbname_); + ASSERT_EQ(first.size(), 1U); + std::string before; + ASSERT_OK(ReadFileToString(env_, first[0], &before)); + ASSERT_OK(dbfull()->TEST_SwitchMemtable()); + ASSERT_OK(Put("other", "x")); + auto after_list = ListLeftovers(options, dbname_); + ASSERT_GE(after_list.size(), 2U); + std::string after; + ASSERT_OK(ReadFileToString(env_, first[0], &after)); + ASSERT_EQ(before, after); +} + +TEST_F(DBCsppCrashSafeTest, Allow2pcAloneStillConverts) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.allow_2pc = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "v")); + ASSERT_OK(dbfull()->TEST_SwitchMemtable()); + Close(); + ASSERT_TRUE(ListLeftovers(options, dbname_).empty()); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("k"), "v"); +} + +#if !defined(OS_WIN) +TEST_F(CrashChild, DISABLED_WalFilterFallsBackToWal) { + Options options = BaseCrashSafeOptions(dbname_, true, arg_[1] == '1'); + if (arg_[0] == '1') SetupOsl(&options, true); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "a", std::string(128, 'a'))); + ASSERT_OK(child_db->Put(WriteOptions(), "b", std::string(128, 'b'))); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, WalFilterFallsBackToWal) { + class IgnoreRecords final : public WalFilter { + public: + int calls = 0; + const char* Name() const override { return "IgnoreRecords"; } + WalProcessingOption LogRecordFound(unsigned long long, const std::string&, + const WriteBatch&, WriteBatch*, + bool*) override { + ++calls; + return WalProcessingOption::kIgnoreCurrentRecord; + } + }; + for (bool osl : {false, true}) { + for (bool log_index : {false, true}) { + SCOPED_TRACE(osl); + SCOPED_TRACE(log_index); + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, log_index); + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "WalFilterFallsBackToWal", + std::string(osl ? "1" : "0") + + (log_index ? "1" : "0")), 0); + ASSERT_FALSE(ListLeftovers(options, dbname_).empty()); + IgnoreRecords filter; + options.wal_filter = &filter; + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(filter.calls, 2); + ASSERT_EQ(Get("a"), "NOT_FOUND"); + ASSERT_EQ(Get("b"), "NOT_FOUND"); + ASSERT_EQ(CountL0(db_), 0); + Close(); + } + } +} + +TEST_F(CrashChild, DISABLED_LeftoverNoMagicFallsBackToWal) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", + [](void*) { ::_exit(1); }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "bad", "hdr")); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, LeftoverNoMagicFallsBackToWal) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "LeftoverNoMagicFallsBackToWal"), 1); + auto leftovers = ListLeftovers(options, dbname_); + ASSERT_FALSE(leftovers.empty()); + const int fd = ::open(leftovers[0].c_str(), O_RDWR); + ASSERT_GE(fd, 0); + terark::DFA_MmapHeader hdr{}; + ASSERT_EQ(::pread(fd, &hdr, sizeof(hdr), 0), + static_cast(sizeof(hdr))); + std::memset(hdr.reserved, 0, sizeof(hdr.reserved)); + ASSERT_EQ(::pwrite(fd, &hdr, sizeof(hdr), 0), + static_cast(sizeof(hdr))); + ::close(fd); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("bad"), "hdr"); +} + +TEST_F(CrashChild, DISABLED_DualLeftoverSecondConvertFails) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + std::atomic pubs{0}; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", [&pubs](void*) { + if (pubs.fetch_add(1) >= 1) { + ::_exit(1); + } + }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "a", "1")); + ASSERT_OK(static_cast(child_db)->TEST_SwitchMemtable()); + ASSERT_OK(child_db->Put(WriteOptions(), "b", "2")); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, DualLeftoverSecondConvertFails) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "DualLeftoverSecondConvertFails"), 1); + SyncPoint::GetInstance()->EnableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + std::atomic converts{0}; + SyncPoint::GetInstance()->SetCallBack( + "CrashSafeRecover::ConvertToSST:InjectStatus", [&converts](void* arg) { + if (converts.fetch_add(1) >= 1) { + *static_cast(arg) = Status::IOError("second leftover"); + } + }); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("a"), "1"); + ASSERT_EQ(Get("b"), "2"); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); +} + +TEST_F(CrashChild, DISABLED_TruncateInjectFailureFallsBackToWal) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", + [](void*) { ::_exit(1); }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "tr", "ok")); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, TruncateInjectFailureFallsBackToWal) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "TruncateInjectFailureFallsBackToWal"), 1); + SyncPoint::GetInstance()->EnableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + SyncPoint::GetInstance()->SetCallBack( + "CrashSafeRecover::Truncate:InjectStatus", [](void* arg) { + *static_cast(arg) = Status::IOError("inject truncate"); + }); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("tr"), "ok"); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); +} + +TEST_F(CrashChild, DISABLED_LinkFileInjectFailureFallsBackToWal) { + Options options = BaseCrashSafeOptions(dbname_, true, true); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", + [](void*) { ::_exit(1); }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "lk", "ok")); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, LinkFileInjectFailureFallsBackToWal) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, true); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "LinkFileInjectFailureFallsBackToWal"), 1); + SyncPoint::GetInstance()->EnableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + SyncPoint::GetInstance()->SetCallBack( + "CrashSafeRecover::LinkFile:InjectStatus", [](void* arg) { + *static_cast(arg) = Status::IOError("inject link"); + }); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("lk"), "ok"); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); +} + +TEST_F(CrashChild, DISABLED_AfterRenameBeforeAddFileFallsBackToWal) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", + [](void*) { ::_exit(1); }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "or", "phan")); + ::_exit(0); +} + +TEST_F(CrashChild, DISABLED_AfterRenameBeforeAddFileFallsBackToWalRecover) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + SyncPoint::GetInstance()->SetCallBack( + "CrashSafeRecover::AfterRenameBeforeAddFile", + [](void*) { ::_exit(1); }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* recover_db = nullptr; + DB::Open(options, dbname_, &recover_db); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, AfterRenameBeforeAddFileFallsBackToWal) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "AfterRenameBeforeAddFileFallsBackToWal"), 1); + SyncPoint::GetInstance()->EnableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_EQ(RunCrashChild(dbname_, "AfterRenameBeforeAddFileFallsBackToWalRecover"), 1); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("or"), "phan"); +} + +TEST_F(DBCsppCrashSafeTest, CrashSafeOrLogIndexDisablesWalCompression) { + for (bool crash_safe : {false, true}) { + for (bool log_index : {false, true}) { + SCOPED_TRACE(crash_safe); + SCOPED_TRACE(log_index); + Close(); + Options options = BaseCrashSafeOptions(dbname_, crash_safe, log_index); + options.wal_compression = kZSTD; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(db_->GetDBOptions().wal_compression, + crash_safe || log_index ? kNoCompression : kZSTD); + ASSERT_OK(Put("k", "v")); + Reopen(options); + ASSERT_EQ(Get("k"), "v"); + } + } +} + +TEST_F(CrashChild, DISABLED_AfterWriteToWALGhostCopyHidden) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::WriteImpl:AfterWriteToWALBeforePublish", + [](void*) { ::_exit(1); }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(child_db->Put(WriteOptions(), "tail", "2")); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, AfterWriteToWALGhostCopyHidden) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + { + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("first", "1")); + Close(); + } + ASSERT_EQ(RunCrashChild(dbname_, "AfterWriteToWALGhostCopyHidden"), 1); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("first"), "1"); + ASSERT_EQ(Get("tail"), "2"); + ASSERT_OK(Put("bump", "seq")); + ASSERT_EQ(Get("tail"), "2"); + int n = 0; + std::unique_ptr it(db_->NewIterator(ReadOptions())); + for (it->Seek("tail"); it->Valid() && it->key() == "tail"; it->Next()) { + n++; + } + ASSERT_OK(it->status()); + ASSERT_EQ(n, 1); +} + +TEST_F(CrashChild, DISABLED_WriteCommittedPrepareInDoubt) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + TransactionDBOptions txn_opts; + txn_opts.write_policy = TxnDBWritePolicy::WRITE_COMMITTED; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", + [](void*) { ::_exit(1); }); + SyncPoint::GetInstance()->EnableProcessing(); + TransactionDB* txn_db = nullptr; + ASSERT_OK(TransactionDB::Open(options, txn_opts, dbname_, &txn_db)); + TransactionOptions to; + to.skip_prepare = false; + Transaction* txn = txn_db->BeginTransaction(WriteOptions(), to); + ASSERT_OK(txn->SetName("xid1")); + ASSERT_OK(txn->Put("prep", "v")); + ASSERT_OK(txn->Prepare()); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, WriteCommittedPrepareInDoubt) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + TransactionDBOptions txn_opts; + txn_opts.write_policy = TxnDBWritePolicy::WRITE_COMMITTED; + ASSERT_EQ(RunCrashChild(dbname_, "WriteCommittedPrepareInDoubt"), 1); + TransactionDB* txn_db = nullptr; + ASSERT_OK(TransactionDB::Open(options, txn_opts, dbname_, &txn_db)); + std::vector prepared; + txn_db->GetAllPreparedTransactions(&prepared); + ASSERT_EQ(prepared.size(), 1U); + ASSERT_OK(prepared[0]->Rollback()); + delete prepared[0]; + delete txn_db; +} + +TEST_F(CrashChild, DISABLED_WriteCommittedPrepareAndCommit) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + TransactionDBOptions txn_opts; + txn_opts.write_policy = TxnDBWritePolicy::WRITE_COMMITTED; + std::atomic prepare_done{false}; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", [&prepare_done](void*) { + if (prepare_done.load(std::memory_order_acquire)) { + ::_exit(1); + } + }); + SyncPoint::GetInstance()->EnableProcessing(); + TransactionDB* txn_db = nullptr; + ASSERT_OK(TransactionDB::Open(options, txn_opts, dbname_, &txn_db)); + TransactionOptions to; + to.skip_prepare = false; + Transaction* txn = txn_db->BeginTransaction(WriteOptions(), to); + ASSERT_OK(txn->SetName("xid2")); + ASSERT_OK(txn->Put("pc", "v")); + ASSERT_OK(txn->Prepare()); + prepare_done.store(true, std::memory_order_release); + ASSERT_OK(txn->Commit()); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, WriteCommittedPrepareAndCommit) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + TransactionDBOptions txn_opts; + txn_opts.write_policy = TxnDBWritePolicy::WRITE_COMMITTED; + ASSERT_EQ(RunCrashChild(dbname_, "WriteCommittedPrepareAndCommit"), 1); + TransactionDB* txn_db = nullptr; + ASSERT_OK(TransactionDB::Open(options, txn_opts, dbname_, &txn_db)); + std::vector prepared; + txn_db->GetAllPreparedTransactions(&prepared); + ASSERT_TRUE(prepared.empty()); + std::string v; + ASSERT_OK(txn_db->Get(ReadOptions(), "pc", &v)); + ASSERT_EQ(v, "v"); + delete txn_db; +} + +TEST_F(DBCsppCrashSafeTest, WriteCommittedPrepareTwoQueuesPublishes) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.two_write_queues = true; + Destroy(options); + TransactionDBOptions txn_opts; + txn_opts.write_policy = TxnDBWritePolicy::WRITE_COMMITTED; + TransactionDB* txn_db = nullptr; + ASSERT_OK(TransactionDB::Open(options, txn_opts, dbname_, &txn_db)); + TransactionOptions to; + to.skip_prepare = false; + Transaction* txn = txn_db->BeginTransaction(WriteOptions(), to); + ASSERT_OK(txn->SetName("xidq")); + ASSERT_OK(txn->Put("pq", "1")); + ASSERT_OK(txn->Prepare()); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_GT(rec.wal_number, 0U); + ASSERT_GT(rec.wal_offset, 0U); + ASSERT_OK(txn->Rollback()); + delete txn; + delete txn_db; +} + +#if !defined(OS_WIN) +TEST_F(CrashChild, DISABLED_WriteCommittedPrepareTwoQueuesInDoubt) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.two_write_queues = true; + TransactionDBOptions txn_opts; + txn_opts.write_policy = TxnDBWritePolicy::WRITE_COMMITTED; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", + [](void*) { ::_exit(1); }); + SyncPoint::GetInstance()->EnableProcessing(); + TransactionDB* txn_db = nullptr; + ASSERT_OK(TransactionDB::Open(options, txn_opts, dbname_, &txn_db)); + TransactionOptions to; + to.skip_prepare = false; + Transaction* txn = txn_db->BeginTransaction(WriteOptions(), to); + ASSERT_OK(txn->SetName("xidq1")); + ASSERT_OK(txn->Put("prepq", "v")); + ASSERT_OK(txn->Prepare()); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, WriteCommittedPrepareTwoQueuesInDoubt) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.two_write_queues = true; + Destroy(options); + TransactionDBOptions txn_opts; + txn_opts.write_policy = TxnDBWritePolicy::WRITE_COMMITTED; + ASSERT_EQ(RunCrashChild(dbname_, "WriteCommittedPrepareTwoQueuesInDoubt"), 1); + TransactionDB* txn_db = nullptr; + ASSERT_OK(TransactionDB::Open(options, txn_opts, dbname_, &txn_db)); + std::vector prepared; + txn_db->GetAllPreparedTransactions(&prepared); + ASSERT_EQ(prepared.size(), 1U); + ASSERT_OK(prepared[0]->Rollback()); + delete prepared[0]; + delete txn_db; +} + +TEST_F(CrashChild, DISABLED_WriteCommittedPrepareTwoQueuesAndCommit) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.two_write_queues = true; + TransactionDBOptions txn_opts; + txn_opts.write_policy = TxnDBWritePolicy::WRITE_COMMITTED; + std::atomic prepare_done{false}; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", [&prepare_done](void*) { + if (prepare_done.load(std::memory_order_acquire)) { + ::_exit(1); + } + }); + SyncPoint::GetInstance()->EnableProcessing(); + TransactionDB* txn_db = nullptr; + ASSERT_OK(TransactionDB::Open(options, txn_opts, dbname_, &txn_db)); + TransactionOptions to; + to.skip_prepare = false; + Transaction* txn = txn_db->BeginTransaction(WriteOptions(), to); + ASSERT_OK(txn->SetName("xidq2")); + ASSERT_OK(txn->Put("pcq", "v")); + ASSERT_OK(txn->Prepare()); + prepare_done.store(true, std::memory_order_release); + ASSERT_OK(txn->Commit()); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, WriteCommittedPrepareTwoQueuesAndCommit) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.two_write_queues = true; + Destroy(options); + TransactionDBOptions txn_opts; + txn_opts.write_policy = TxnDBWritePolicy::WRITE_COMMITTED; + ASSERT_EQ(RunCrashChild(dbname_, "WriteCommittedPrepareTwoQueuesAndCommit"), 1); + TransactionDB* txn_db = nullptr; + ASSERT_OK(TransactionDB::Open(options, txn_opts, dbname_, &txn_db)); + std::vector prepared; + txn_db->GetAllPreparedTransactions(&prepared); + ASSERT_TRUE(prepared.empty()); + std::string v; + ASSERT_OK(txn_db->Get(ReadOptions(), "pcq", &v)); + ASSERT_EQ(v, "v"); + delete txn_db; +} +#endif + +TEST_F(DBCsppCrashSafeTest, WritePreparedFallsBackToWal) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.two_write_queues = true; + Destroy(options); + TransactionDBOptions txn_opts; + txn_opts.write_policy = TxnDBWritePolicy::WRITE_PREPARED; + TransactionDB* txn_db = nullptr; + ASSERT_OK(TransactionDB::Open(options, txn_opts, dbname_, &txn_db)); + Transaction* txn = txn_db->BeginTransaction(WriteOptions()); + ASSERT_OK(txn->Put("wp", "1")); + ASSERT_OK(txn->Commit()); + delete txn; + delete txn_db; + ASSERT_OK(TransactionDB::Open(options, txn_opts, dbname_, &txn_db)); + std::string v; + ASSERT_OK(txn_db->Get(ReadOptions(), "wp", &v)); + ASSERT_EQ(v, "1"); + delete txn_db; +} + +TEST_F(DBCsppCrashSafeTest, AfterRenameCloseSecondFlushInject) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("a", "1")); + ASSERT_OK(dbfull()->TEST_SwitchMemtable()); + ASSERT_OK(Put("b", "2")); + ASSERT_OK(dbfull()->TEST_SwitchMemtable()); + ASSERT_OK(Put("c", "3")); + std::atomic converts{0}; + SyncPoint::GetInstance()->SetCallBack( + "FlushJob::ConvertToSST:Status", [&converts](void* arg) { + if (converts.fetch_add(1) >= 1) { + *static_cast(arg) = Status::IOError("close second"); + } + }); + SyncPoint::GetInstance()->EnableProcessing(); + Close(); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_OK(env_->FileExists(CrashSafePubSeqFileName(dbname_))); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("a"), "1"); + ASSERT_EQ(Get("b"), "2"); + ASSERT_EQ(Get("c"), "3"); +} + +TEST_F(CrashChild, DISABLED_DualCfLeftoverConvertsBoth) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + std::atomic pubs{0}; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", [&pubs](void*) { + if (pubs.fetch_add(1) >= 1) { + ::_exit(1); + } + }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + std::vector hs; + std::vector cfs = { + {kDefaultColumnFamilyName, options}, {"one", options}}; + ASSERT_OK(DB::Open(options, dbname_, cfs, &hs, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), hs[0], "d", "1")); + ASSERT_OK(child_db->Put(WriteOptions(), hs[1], "c", "2")); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, DualCfLeftoverConvertsBoth) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + ASSERT_OK(TryReopen(options)); + CreateAndReopenWithCF({"one"}, options); + Close(); + ASSERT_EQ(RunCrashChild(dbname_, "DualCfLeftoverConvertsBoth"), 1); + ASSERT_OK(TryReopenWithColumnFamilies({kDefaultColumnFamilyName, "one"}, + options)); + ASSERT_EQ(Get(0, "d"), "1"); + ASSERT_EQ(Get(1, "c"), "2"); + ASSERT_GE(CountL0(db_, kDefaultColumnFamilyName), 1); + ASSERT_GE(CountL0(db_, "one"), 1); +} + +TEST_F(CrashChild, DISABLED_AtomicFlushDualCfLeftoverConvertsBoth) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.atomic_flush = true; + std::atomic pubs{0}; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", [&pubs](void*) { + if (pubs.fetch_add(1) >= 1) { + ::_exit(1); + } + }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + std::vector hs; + std::vector cfs = { + {kDefaultColumnFamilyName, options}, {"one", options}}; + ASSERT_OK(DB::Open(options, dbname_, cfs, &hs, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), hs[0], "d", "1")); + ASSERT_OK(child_db->Put(WriteOptions(), hs[1], "c", "2")); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, AtomicFlushDualCfLeftoverConvertsBoth) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.atomic_flush = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + CreateAndReopenWithCF({"one"}, options); + Close(); + ASSERT_EQ(RunCrashChild(dbname_, "AtomicFlushDualCfLeftoverConvertsBoth"), 1); + ASSERT_OK(TryReopenWithColumnFamilies({kDefaultColumnFamilyName, "one"}, + options)); + ASSERT_EQ(Get(0, "d"), "1"); + ASSERT_EQ(Get(1, "c"), "2"); + ASSERT_GE(CountL0(db_, kDefaultColumnFamilyName), 1); + ASSERT_GE(CountL0(db_, "one"), 1); +} + +TEST_F(CrashChild, DISABLED_DroppedCfLeftoverSkipped) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + std::atomic pubs{0}; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", [&pubs](void*) { + if (pubs.fetch_add(1) >= 1) { + ::_exit(1); + } + }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + std::vector hs; + std::vector cfs = { + {kDefaultColumnFamilyName, options}, {"one", options}}; + ASSERT_OK(DB::Open(options, dbname_, cfs, &hs, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), hs[0], "keep", "1")); + ASSERT_OK(child_db->Put(WriteOptions(), hs[1], "drop", "2")); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, DroppedCfLeftoverSkipped) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + ASSERT_OK(TryReopen(options)); + CreateAndReopenWithCF({"one"}, options); + Close(); + ASSERT_EQ(RunCrashChild(dbname_, "DroppedCfLeftoverSkipped"), 1); + ASSERT_OK(TryReopenWithColumnFamilies({kDefaultColumnFamilyName, "one"}, + options)); + ASSERT_EQ(Get(0, "keep"), "1"); + ASSERT_EQ(Get(1, "drop"), "2"); + ASSERT_OK(Flush(0)); + std::string left1; + for (const auto& p : ListLeftovers(options, dbname_)) { + if (p.find(".memtab-1") != std::string::npos) { + left1 = p; + break; + } + } + ASSERT_FALSE(left1.empty()); + const std::string bak = left1 + ".bak"; + CopyFile(left1, bak); + ASSERT_OK(db_->DropColumnFamily(handles_[1])); + Close(); + if (::access(left1.c_str(), F_OK) != 0) { + ASSERT_EQ(::rename(bak.c_str(), left1.c_str()), 0); + } else { + ::unlink(bak.c_str()); + } + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("keep"), "1"); + bool dropped_left = false; + for (const auto& p : ListLeftovers(options, dbname_)) { + if (p.find(".memtab-1") != std::string::npos) { + dropped_left = true; + } + } + ASSERT_TRUE(dropped_left); +} + +TEST_F(CrashChild, DISABLED_MultiChunkAfterCommitStillReadable) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.memtable_factory.reset(NewCSPPMemTabForPlain( + R"({"mem_cap":16777216,"convert_to_sst":"kFileMmap","chunk_size":4096})")); + std::atomic pubs{0}; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", [&pubs](void*) { + if (pubs.fetch_add(1) >= 199) { + ::_exit(1); + } + }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + for (int i = 0; i < 200; ++i) { + ASSERT_OK(child_db->Put(WriteOptions(), "k" + std::to_string(i), + std::string(80, 'v'))); + } + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, MultiChunkAfterCommitStillReadable) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.memtable_factory.reset(NewCSPPMemTabForPlain( + R"({"mem_cap":16777216,"convert_to_sst":"kFileMmap","chunk_size":4096})")); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "MultiChunkAfterCommitStillReadable"), 1); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("k0").size(), 80U); + ASSERT_EQ(Get("k199").size(), 80U); +} + +TEST_F(CrashChild, DISABLED_OslMultiChunkAfterCommitStillReadable) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + SetupOsl(&options, true); + std::atomic pubs{0}; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", [&pubs](void*) { + if (pubs.fetch_add(1) >= 399) { + ::_exit(1); + } + }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + for (int i = 0; i < 400; ++i) { + ASSERT_OK(child_db->Put(WriteOptions(), "ok" + std::to_string(i), + std::string(8192, 'o'))); + } + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, OslMultiChunkAfterCommitStillReadable) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + SetupOsl(&options, true); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "OslMultiChunkAfterCommitStillReadable"), 1); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("ok0").size(), 8192U); + ASSERT_EQ(Get("ok399").size(), 8192U); +} + +TEST_F(CrashChild, DISABLED_OslFirstChunkAfterCommitReadable) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + SetupOsl(&options, true); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", + [](void*) { ::_exit(1); }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "osl1", "v")); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, OslFirstChunkAfterCommitReadable) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + SetupOsl(&options, true); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "OslFirstChunkAfterCommitReadable"), 1); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("osl1"), "v"); + ColumnFamilyMetaData metadata; + db_->GetColumnFamilyMetaData(&metadata); + ASSERT_EQ(metadata.levels[0].files.size(), 1U); + ASSERT_TRUE(metadata.levels[0].files[0].marked_for_compaction); +} + +TEST_F(CrashChild, DISABLED_WalSwitchKeepsTail) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + std::atomic pubs{0}; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", [&pubs](void*) { + if (pubs.fetch_add(1) >= 1) { + ::_exit(1); + } + }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "old", "1")); + ASSERT_OK(static_cast(child_db)->TEST_SwitchWAL()); + ASSERT_OK(child_db->Put(WriteOptions(), "neu", "2")); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, WalSwitchKeepsTail) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "WalSwitchKeepsTail"), 1); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("old"), "1"); + ASSERT_EQ(Get("neu"), "2"); +} + +TEST_F(CrashChild, DISABLED_LeftoverLogRefUnbindFallsBackToWal) { + Options options = BaseCrashSafeOptions(dbname_, true, true); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", + [](void*) { ::_exit(1); }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "rb", "ok")); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, LeftoverLogRefUnbindFallsBackToWal) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, true); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "LeftoverLogRefUnbindFallsBackToWal"), 1); + auto leftovers = ListLeftovers(options, dbname_); + ASSERT_FALSE(leftovers.empty()); + const int fd = ::open(leftovers[0].c_str(), O_RDWR); + ASSERT_GE(fd, 0); + terark::DFA_MmapHeader hdr{}; + ASSERT_EQ(::pread(fd, &hdr, sizeof(hdr), 0), + static_cast(sizeof(hdr))); + auto* cs = reinterpret_cast(hdr.reserved); + ASSERT_EQ(cs[0], 0x50505343U); + uint64_t fake_fileno = 999999; + const size_t wal0 = offsetof(terark::DFA_MmapHeader, reserved) + 16; + ASSERT_EQ(::pwrite(fd, &fake_fileno, sizeof(fake_fileno), + static_cast(wal0)), + static_cast(sizeof(fake_fileno))); + ::close(fd); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("rb"), "ok"); +} + +// Clang is excluded from the AVX store and still hits AfterOddGeneration. +#if !defined(__AVX__) || defined(__clang__) +TEST_F(DBCsppCrashSafeTest, CloseWaitsForStatsPublication) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.stats_dump_period_sec = 0; + options.stats_persist_period_sec = 1; + options.persist_stats_to_disk = true; + options.statistics = CreateDBStatistics(); + Destroy(options); + ASSERT_OK(TryReopen(options)); + SyncPoint::GetInstance()->LoadDependency({ + {"CloseWaitsForStatsPublication:Publishing", + "CloseWaitsForStatsPublication:BeginClose"}, + {"Timer::WaitForTaskCompleteIfNecessary:TaskExecuting", + "CloseWaitsForStatsPublication:Resume"}, + }); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterOddGeneration", [](void*) { + TEST_SYNC_POINT("CloseWaitsForStatsPublication:Publishing"); + TEST_SYNC_POINT("CloseWaitsForStatsPublication:Resume"); + }); + SyncPoint::GetInstance()->EnableProcessing(); + TEST_SYNC_POINT("CloseWaitsForStatsPublication:BeginClose"); + ASSERT_OK(db_->Close()); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + SyncPoint::GetInstance()->LoadDependency({}); + Close(); +} +#endif // !__AVX__ || __clang__ + +TEST_F(DBCsppCrashSafeTest, InFlightFlushThenClose) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.max_background_flushes = 1; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("if", "1")); + ASSERT_OK(dbfull()->TEST_SwitchMemtable()); + ASSERT_OK(Put("tail", "2")); + test::SleepingBackgroundTask sleeping_task; + env_->Schedule(&test::SleepingBackgroundTask::DoSleepTask, &sleeping_task, + Env::Priority::HIGH); + sleeping_task.WaitUntilSleeping(); + FlushOptions fo; + fo.wait = false; + ASSERT_OK(db_->Flush(fo)); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::FlushMemTable:AfterScheduleFlush", + [&](void*) { sleeping_task.WakeUp(); }); + SyncPoint::GetInstance()->EnableProcessing(); + std::thread closer([&] { Close(); }); + sleeping_task.WaitUntilDone(); + closer.join(); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("if"), "1"); + ASSERT_EQ(Get("tail"), "2"); +} +#endif // !OS_WIN + +} // namespace ROCKSDB_NAMESPACE + +int main(int argc, char** argv) { + ROCKSDB_NAMESPACE::port::InstallStackTraceHandler(); + ::testing::InitGoogleTest(&argc, argv); +#if !defined(OS_WIN) + if (argc == 5 && std::strcmp(argv[1], "--crash-child") == 0) { + ROCKSDB_NAMESPACE::crash_child_db = argv[3]; + ROCKSDB_NAMESPACE::crash_child_arg = argv[4]; + ::testing::GTEST_FLAG(filter) = std::string("CrashChild.DISABLED_") + argv[2]; + ::testing::GTEST_FLAG(also_run_disabled_tests) = true; + ::alarm(30); + const int result = RUN_ALL_TESTS(); + // Every child action must reach its explicit _exit, not just finish a test. + return result == 0 ? 126 : 125; + } +#endif + return RUN_ALL_TESTS(); +} diff --git a/db/db_impl/db_impl.cc b/db/db_impl/db_impl.cc index 48bb09ce7..4b139aab1 100644 --- a/db/db_impl/db_impl.cc +++ b/db/db_impl/db_impl.cc @@ -668,6 +668,8 @@ Status DBImpl::CloseHelper() { // marker. After this we do a variant of the waiting and unschedule work // (to consider: moving all the waiting into CancelAllBackgroundWork(true)) CancelAllBackgroundWork(false); + // Periodic PersistStats writes must finish before unmapping the cursor. + UnmapPublishedSeqFile(); // Cancel manual compaction if there's any if (HasPendingManualCompaction()) { @@ -5919,6 +5921,12 @@ Status DestroyDB(const std::string& dbname, const Options& options, } } } + env->DeleteFile(CrashSafePubSeqFileName(dbname)).PermitUncheckedError(); + for (const auto& fname : filenames) { + if (fname.find(".memtab-") != std::string::npos) { + env->DeleteFile(dbname + "/" + fname).PermitUncheckedError(); + } + } std::set paths; for (const DbPath& db_path : options.db_paths) { diff --git a/db/db_impl/db_impl.h b/db/db_impl/db_impl.h index ad9fece98..f314c7878 100644 --- a/db/db_impl/db_impl.h +++ b/db/db_impl/db_impl.h @@ -78,6 +78,7 @@ class TaskLimiterToken; class Version; class VersionEdit; class VersionSet; +struct PublishedSeqMmapHeader; class WriteCallback; struct JobContext; struct ExternalSstFileInfo; @@ -1427,6 +1428,15 @@ class DBImpl : public DB { // such a file's absolute path to its parent directory. std::unordered_map files_to_delete_; bool is_new_db_ = false; + // Restore the saved WAL cursor only after recovery edits are installed. + bool restore_published_seq_ = false; + // WAL tail cursor for this Recover. RecoverLogFiles reads it from here. + bool crash_safe_wal_tail_replay_ = false; + uint8_t crash_safe_wal_offset_kind_ = 0; + uint64_t crash_safe_wal_number_ = 0; + uint64_t crash_safe_wal_offset_ = 0; + // WALs numbered below this predate CSPUBSEQ: probe their format. + uint64_t crash_safe_probe_wal_below_ = 0; }; // Persist options to options file. Must be holding options_mutex_. @@ -1931,6 +1941,25 @@ class DBImpl : public DB { bool* corrupted_log_found, RecoveryContext* recovery_ctx); + // *probe_wal_below: WALs numbered below it have unknown format. + Status MapPublishedSeqFile(uint64_t* probe_wal_below); + void UnmapPublishedSeqFile(); + void PersistPublishedSequence(SequenceNumber seq, uint64_t wal_number, + uint64_t wal_offset); + void MaybePersistPublishedSequence(SequenceNumber seq, + const WriteThread::WriteGroup& write_group); + void PersistStagedPublishedWal(); + void StagePublishedWal(SequenceNumber seq, uint64_t wal_number, + uint64_t wal_offset); + void AccountPendingMemtableWrites(size_t n); + bool CanConvertLeftoverForCrashSafeRecover( + const std::vector& leftover_snapshot, + SequenceNumber mmap_pubseq, uint64_t mmap_wal_number, + std::string* fail_reason); + Status ConvertLeftoverMemtables( + const std::vector& leftover_snapshot, + SequenceNumber max_visible_seq, RecoveryContext* recovery_ctx); + // The following two methods are used to flush a memtable to // storage. The first one is used at database RecoveryTime (when the // database is opened) and is heavyweight because it holds the mutex @@ -2744,6 +2773,19 @@ class DBImpl : public DB { // It contains the implementations for each periodic task. std::map periodic_task_functions_; + PublishedSeqMmapHeader* pubseq_mmap_ = nullptr; + size_t pubseq_mmap_size_ = 0; + intptr_t pubseq_fd_ = -1; + SequenceNumber staged_pub_seq_ = 0; + uint64_t staged_pub_wal_number_ = 0; + uint64_t staged_pub_wal_offset_ = 0; + + // Staged WAL cursor shared by unordered_write and pipelined_write. + // After this group's WriteWAL, persist the previously staged cursor if + // pending_memtable_writes_ == 0, then stage this group's cursor (mmap + // trails one WAL). + bool staged_pub_valid_ = false; + // When set, we use a separate queue for writes that don't write to memtable. // In 2PC these are the writes at Prepare phase. const bool two_write_queues_; diff --git a/db/db_impl/db_impl_open.cc b/db/db_impl/db_impl_open.cc index da378816c..f70a4ca64 100644 --- a/db/db_impl/db_impl_open.cc +++ b/db/db_impl/db_impl_open.cc @@ -7,6 +7,15 @@ // Use of this source code is governed by a BSD-style license that can be // found in the LICENSE file. See the AUTHORS file for names of contributors. #include +#ifndef _MSC_VER +#include +#include +#endif +// Clang splits _mm256_store_si256 into two vmovdqa xmm stores, which tears +// the record across a process crash. Keep Clang on the non-AVX path. +#if defined(__AVX__) && !defined(__clang__) +#include +#endif #include "db/builder.h" #include "db/db_impl/db_impl.h" @@ -25,7 +34,9 @@ #include "rocksdb/wal_filter.h" #include "test_util/sync_point.h" #include "util/rate_limiter_impl.h" +#include #include +#include #include "util/string_util.h" #include "util/udt_util.h" @@ -443,12 +454,467 @@ IOStatus Directories::SetDirectories(FileSystem* fs, const std::string& dbname, return IOStatus::OK(); } +struct alignas(32) PublishedSeqRecord { + uint64_t pubseq = 0; + uint64_t wal_number = 0; + uint64_t wal_offset = 0; + uint64_t padding = 0; +}; + +struct alignas(CACHE_LINE_SIZE) PublishedSeqMmapHeader { + uint64_t magic = 0; + uint32_t version = 0; + uint32_t header_size = 0; + // 1 physical, 2 log-index. Written by Stamp, not by each publish. + uint32_t wal_offset_kind = 0; + // First WAL written under wal_offset_kind; 0 until that WAL exists. + // Older WALs predate this sidecar and have unknown format. + uint32_t kind_since_wal = 0; + uint64_t generation = 0; + PublishedSeqRecord rec; + + void Stamp(bool memtable_as_log_index); +}; + +namespace { + +constexpr size_t kPublishedSeqMmapSize = 4096; +constexpr uint64_t kPublishedSeqMagic = 0x5145534255505343ULL; // 'CSPUBSEQ' +constexpr uint32_t kPublishedSeqVersion = 1; + +enum class PublishedWalOffsetKind : uint32_t { + kNone = 0, + kPhysical = 1, + kLogIndex = 2, +}; + +static_assert(std::atomic::is_always_lock_free); +static_assert(sizeof(std::atomic) == sizeof(uint64_t)); +static_assert(sizeof(PublishedSeqMmapHeader) == CACHE_LINE_SIZE); +static_assert(alignof(PublishedSeqMmapHeader) == CACHE_LINE_SIZE); +static_assert(sizeof(PublishedSeqMmapHeader) <= kPublishedSeqMmapSize); +static_assert(sizeof(PublishedSeqRecord) == 32); +static_assert(offsetof(PublishedSeqMmapHeader, rec) % 32 == 0); + +bool PublishedSeqHeaderValid(const PublishedSeqMmapHeader* hdr) { + return hdr->magic == kPublishedSeqMagic && + hdr->version == kPublishedSeqVersion && + hdr->header_size == sizeof(PublishedSeqMmapHeader) && + (hdr->wal_offset_kind == 1 || // kPhysical + hdr->wal_offset_kind == 2); // kLogIndex +} + +uint64_t PublishedWalRecordStart(uint64_t offset, uint64_t kind) { + if (kind == 2) { // PublishedWalOffsetKind::kLogIndex + return offset >= sizeof(log::RawRecHeader) + ? offset - sizeof(log::RawRecHeader) + : 0; + } + return offset; +} + +bool ReadPublishedSeqRecord(const PublishedSeqMmapHeader* hdr, + PublishedSeqRecord* out) { + if ((hdr->generation & 1) == 0) { + *out = hdr->rec; + return true; + } else { + return false; + } +} + +uint32_t ParseLeftoverCfId(const std::string& path) { + const auto pos = path.rfind(".memtab-"); + if (pos == std::string::npos) { + return std::numeric_limits::max(); + } + const char* p = path.c_str() + pos + 8; + char* end = nullptr; + const unsigned long id = std::strtoul(p, &end, 10); + if (end == p) { + return std::numeric_limits::max(); + } + return static_cast(id); +} + +} // namespace + +void PublishedSeqMmapHeader::Stamp(bool memtable_as_log_index) { + magic = kPublishedSeqMagic; + version = kPublishedSeqVersion; + header_size = static_cast(sizeof(PublishedSeqMmapHeader)); + wal_offset_kind = memtable_as_log_index ? 2 : 1; // kLogIndex : kPhysical + generation = 0; + rec = PublishedSeqRecord{}; + kind_since_wal = 0; +} + +Status DBImpl::MapPublishedSeqFile(uint64_t* probe_wal_below) { + *probe_wal_below = std::numeric_limits::max(); + if (!immutable_db_options_.memtable_crash_safe_recover) { + return Status::OK(); + } + if (pubseq_mmap_ != nullptr) { + return Status::OK(); + } + const std::string path = CrashSafePubSeqFileName(dbname_); + size_t sz = 0; // new or empty file: 4K; existing file: mapped whole + intptr_t fd = -1; + void* p = nullptr; + try { + p = terark::mmap_write(path, &sz, &fd); + if (sz < kPublishedSeqMmapSize) { + terark::mmap_close(p, sz, fd); + sz = kPublishedSeqMmapSize; + p = terark::mmap_write(path, &sz, &fd); + } + } catch (const std::exception& ex) { + return Status::IOError(path, ex.what()); + } + pubseq_mmap_ = static_cast(p); + pubseq_mmap_size_ = sz; + pubseq_fd_ = fd; + if (!PublishedSeqHeaderValid(pubseq_mmap_) || + pubseq_mmap_->kind_since_wal == 0) { + TEST_SYNC_POINT("DBImpl::MapPublishedSeqFile:AfterMmapBeforeStamp"); + pubseq_mmap_->Stamp(immutable_db_options_.memtable_as_log_index); + return Status::OK(); + } + if (pubseq_mmap_->kind_since_wal != 0) { + *probe_wal_below = pubseq_mmap_->kind_since_wal; + } + return Status::OK(); +} + +void DBImpl::UnmapPublishedSeqFile() { + if (pubseq_mmap_ != nullptr) { + terark::mmap_close(pubseq_mmap_, pubseq_mmap_size_, pubseq_fd_); + pubseq_mmap_ = nullptr; + pubseq_mmap_size_ = 0; + pubseq_fd_ = -1; + } +} + +void DBImpl::PersistPublishedSequence(SequenceNumber seq, uint64_t wal_number, + uint64_t wal_offset) { + if (pubseq_mmap_ == nullptr) { + return; + } + auto* rec = &pubseq_mmap_->rec; + // DO NOT REWRITE. Do not split this back into + // `if (seq < pubseq) return; if (seq == pubseq) SameSeq;`. + // That form compares twice on the hot path. Here the hot path is one + // UNLIKELY(seq <= pubseq). seq == pubseq is not an error: it falls + // through and publishes. Only seq < pubseq logs and returns. Do not + // return on <=. + if (UNLIKELY(seq <= rec->pubseq)) { + if (UNLIKELY(seq < rec->pubseq)) { + ROCKS_LOG_ERROR(immutable_db_options_.info_log, + "PersistPublishedSequence seq %" PRIu64 + " < published %" PRIu64 " wal #%" PRIu64 + " offset %" PRIu64, + seq, rec->pubseq, wal_number, wal_offset); + return; + } + TEST_SYNC_POINT("DBImpl::PersistPublishedSequence:SameSeq"); + } + // Recovery may leave an odd generation until the first new publication. + const uint64_t g = pubseq_mmap_->generation | 1; + // Clang lowers _mm256_store_si256 to two vmovdqa xmm stores. A crash + // between them publishes a mixed record. GCC emits one vmovdqa ymm. +#if defined(__AVX__) && !defined(__clang__) + // One aligned 32-byte store. A process crash falls between instructions, + // so the record is all old or all new. Publish an even generation only + // after this store, including when recovery left it odd. A host without it + // uses the odd/even bracket below after the file is moved. + // PublishedSeqRecord next; + // next.pubseq = seq; + // next.wal_number = wal_number; + // next.wal_offset = wal_offset; + // next.padding = 0; + // _mm256_store_si256((__m256i*)rec, _mm256_load_si256((const __m256i*)&next)); + // Lowest 64 bits are pubseq, matching PublishedSeqRecord field order. + __m256i packed = _mm256_set_epi64x(0, wal_offset, wal_number, seq); + _mm256_store_si256((__m256i*)rec, packed); +#else + // Keep generation odd throughout the field stores, even if already odd. + // Acquire keeps the field stores after this exchange; the release below + // keeps them before publication of the next even generation. + terark::as_atomic(pubseq_mmap_->generation) + .exchange(g, std::memory_order_acquire); + TEST_SYNC_POINT("DBImpl::PersistPublishedSequence:AfterOddGeneration"); + rec->wal_number = wal_number; + rec->wal_offset = wal_offset; + rec->pubseq = seq; +#endif + terark::as_atomic(pubseq_mmap_->generation) + .store(g + 1, std::memory_order_release); + TEST_SYNC_POINT("DBImpl::PersistPublishedSequence:AfterCommit"); +} + +void DBImpl::MaybePersistPublishedSequence( + SequenceNumber seq, const WriteThread::WriteGroup& write_group) { + if (pubseq_mmap_ == nullptr) { + return; + } + TEST_SYNC_POINT("DBImpl::WriteImpl:AfterWriteToWALBeforePublish"); + PersistPublishedSequence(seq, write_group.wal_number, write_group.wal_offset); +} + +void DBImpl::PersistStagedPublishedWal() { + if (pubseq_mmap_ == nullptr || !staged_pub_valid_) { + return; + } + if (pending_memtable_writes_.load(std::memory_order_acquire) != 0) { + return; + } + PersistPublishedSequence(staged_pub_seq_, staged_pub_wal_number_, + staged_pub_wal_offset_); +} + +void DBImpl::StagePublishedWal(SequenceNumber seq, uint64_t wal_number, + uint64_t wal_offset) { + PersistStagedPublishedWal(); + staged_pub_seq_ = seq; + staged_pub_wal_number_ = wal_number; + staged_pub_wal_offset_ = wal_offset; + staged_pub_valid_ = true; +} + +void DBImpl::AccountPendingMemtableWrites(size_t n) { + if (n == 0) { + return; + } + const size_t pending_cnt = pending_memtable_writes_.fetch_sub(n) - n; + if (pending_cnt == 0) { + std::lock_guard lck(switch_mutex_); + switch_cv_.notify_all(); + } +} + +bool DBImpl::CanConvertLeftoverForCrashSafeRecover( + const std::vector& leftover_snapshot, + SequenceNumber mmap_pubseq, uint64_t mmap_wal_number, + std::string* fail_reason) { + mutex_.AssertHeld(); + auto fail = [&](const std::string& reason) { + *fail_reason = reason; + return false; + }; + if (immutable_db_options_.wal_filter != nullptr) { + return fail("wal_filter requires full WAL replay"); + } + if (!last_seq_same_as_publish_seq_) { + return fail("seq_per_batch && two_write_queues"); + } + if (immutable_db_options_.best_efforts_recovery) { + return fail("best_efforts_recovery"); + } + for (auto* cfd : *versions_->GetColumnFamilySet()) { + if (cfd->IsDropped()) { + continue; + } + auto* fac = cfd->ioptions()->memtable_factory.get(); + if (fac == nullptr || !fac->SupportCrashSafe()) { + return fail("CF " + cfd->GetName() + " factory !SupportCrashSafe"); + } + if (cfd->mem() != nullptr && !cfd->mem()->SupportConvertToSST()) { + return fail("CF " + cfd->GetName() + " mem !SupportConvertToSST"); + } + if (!cfd->imm()->UnflushedMemtablesSupportConvertToSST()) { + return fail("CF " + cfd->GetName() + " imm !SupportConvertToSST"); + } + } + + // Other tiers may be slow and contain many files; memtables use cf_paths[0]. + std::vector paths; + for (const auto* cfd : *versions_->GetColumnFamilySet()) { + if (!cfd->IsDropped()) { + paths.push_back( + NormalizePath(cfd->ioptions()->cf_paths[0].path + + std::string(1, kFilePathSeparator))); + } + } + std::sort(paths.begin(), paths.end()); + paths.erase(std::unique(paths.begin(), paths.end()), paths.end()); + const uint64_t next_file_number = versions_->current_next_file_number(); + for (const auto& path : paths) { + std::vector files; + const Status ls = env_->GetChildren(path, &files); + if (!ls.ok()) { + continue; + } + for (const auto& fname : files) { + uint64_t number = 0; + FileType type; + if (!ParseFileName(fname, &number, &type)) { + continue; + } + // Only numbers still ahead of MANIFEST next_file (crash mid-Open + // before LogAndApply). A Convert that failed after this Open already + // advanced next_file is invisible here; ConvertLeftover deletes those. + if (type == kTableFile && number >= next_file_number) { + return fail("orphan SST " + path + fname); + } + } + } + + std::unordered_set leftover_cfs; + for (const auto& leftover_path : leftover_snapshot) { + const uint32_t cf_id = ParseLeftoverCfId(leftover_path); + ColumnFamilyData* cfd = + versions_->GetColumnFamilySet()->GetColumnFamily(cf_id); + if (cfd == nullptr) { + ROCKS_LOG_WARN(immutable_db_options_.info_log, + "Crash-safe leftover %s belongs to dropped CF %u, skip", + leftover_path.c_str(), cf_id); + continue; + } + auto* fac = cfd->ioptions()->memtable_factory.get(); + const Status ps = + fac->ProbeCrashSafeLeftover(leftover_path, + immutable_db_options_.GetWalDir()); + if (!ps.ok()) { + return fail("probe " + leftover_path + ": " + ps.ToString()); + } + leftover_cfs.insert(cf_id); + } + if (mmap_pubseq > versions_->LastSequence() && leftover_cfs.empty()) { + return fail("CSPUBSEQ > MANIFEST LastSequence and no leftover"); + } + if (!leftover_cfs.empty() && mmap_pubseq == 0) { + return fail("CSPUBSEQ pubseq 0 with leftover present"); + } + if (mmap_wal_number != 0) { + for (const auto* cfd : *versions_->GetColumnFamilySet()) { + // Only CFs persisted past this WAL can omit their leftover safely. + if (!cfd->IsDropped() && cfd->GetLogNumber() <= mmap_wal_number && + leftover_cfs.count(cfd->GetID()) == 0) { + return fail("CF " + cfd->GetName() + " has no leftover before WAL cursor"); + } + } + } + return true; +} + +Status DBImpl::ConvertLeftoverMemtables( + const std::vector& leftover_snapshot, + SequenceNumber max_visible_seq, RecoveryContext* recovery_ctx) { + mutex_.AssertHeld(); + if (leftover_snapshot.empty()) { + return Status::OK(); + } + + struct ConvertedLeftover { + ColumnFamilyData* cfd; + FileMetaData meta; + std::vector blobs; + }; + std::vector converted; + Status s; + mutex_.Unlock(); + for (const auto& leftover_path : leftover_snapshot) { + const uint32_t cf_id = ParseLeftoverCfId(leftover_path); + ColumnFamilyData* cfd = + versions_->GetColumnFamilySet()->GetColumnFamily(cf_id); + if (cfd == nullptr) { + continue; + } + ConvertedLeftover one; + one.cfd = cfd; + const uint64_t file_num = versions_->NewFileNumber(); + one.meta.fd = FileDescriptor(file_num, 0, 0); + one.meta.fd.smallest_seqno = 0; + one.meta.fd.largest_seqno = max_visible_seq; + one.meta.epoch_number = cfd->NewEpochNumber(); + const MutableCFOptions moptions = *cfd->GetLatestMutableCFOptions(); + TableBuilderOptions tboptions( + *cfd->ioptions(), moptions, cfd->internal_comparator(), + cfd->int_tbl_prop_collector_factories(), + GetCompressionFlush(*cfd->ioptions(), moptions), + moptions.compression_opts, cfd->GetID(), cfd->GetName(), 0 /* level */, + false /* is_bottommost */, TableFileCreationReason::kRecovery, + 0 /* oldest_key_time */, 0 /* file_creation_time */, db_id_, + db_session_id_, 0 /* target_file_size */, file_num); + tboptions.generate_file_no = [this]() { return versions_->NewFileNumber(); }; + tboptions.add_blob_file = [&one](BlobFileAddition b) { + one.blobs.push_back(std::move(b)); + }; + Status one_s = + cfd->ioptions()->memtable_factory->RecoverCrashSafeMemTableToSST( + leftover_path, &one.meta, tboptions); + if (!one_s.ok()) { + s = one_s; + // Inject / rename-then-fail still leaves an SST. Include it so the + // cleanup below can unlink it; leftover is already gone. + converted.push_back(std::move(one)); + break; + } + converted.push_back(std::move(one)); + } + mutex_.Lock(); + if (!s.ok()) { + // NewFileNumber already ran. After this Open finishes, next_file is + // past these numbers, so DeleteUnreferencedSstFiles (only >= next_file, + // and it runs before Convert) will not collect them. + for (auto& one : converted) { + if (one.meta.fd.GetNumber() == 0) { + continue; + } + env_->DeleteFile(TableFileName(one.cfd->ioptions()->cf_paths, + one.meta.fd.GetNumber(), + one.meta.fd.GetPathId())) + .PermitUncheckedError(); + for (const auto& blob : one.blobs) { + env_->DeleteFile(BlobFileName(one.cfd->ioptions()->cf_paths.front().path, + blob.GetBlobFileNumber())) + .PermitUncheckedError(); + } + } + return s; + } + for (auto& one : converted) { + if (one.meta.fd.GetFileSize() == 0) { + continue; + } + VersionEdit edit; + edit.SetColumnFamily(one.cfd->GetID()); + one.meta.marked_for_compaction = true; + edit.AddFile(0, one.meta); + for (const auto& blob : one.blobs) { + edit.AddBlobFile(blob); + } + recovery_ctx->UpdateVersionEdits(one.cfd, edit); + } + return Status::OK(); +} + Status DBImpl::Recover( const std::vector& column_families, bool read_only, bool error_if_wal_file_exists, bool error_if_data_exists_in_wals, uint64_t* recovered_seq, RecoveryContext* recovery_ctx) { mutex_.AssertHeld(); + std::vector leftover_snapshot; + if (immutable_db_options_.memtable_crash_safe_recover && !read_only) { + for (const auto& desc : column_families) { + if (!desc.options.memtable_factory) { + continue; + } + const auto& paths = desc.options.cf_paths.empty() + ? immutable_db_options_.db_paths + : desc.options.cf_paths; + desc.options.memtable_factory->ListCrashSafeLeftovers( + paths[0].path, &leftover_snapshot); + } + if (!leftover_snapshot.empty()) { + std::sort(leftover_snapshot.begin(), leftover_snapshot.end()); + leftover_snapshot.erase( + std::unique(leftover_snapshot.begin(), leftover_snapshot.end()), + leftover_snapshot.end()); + } + } + bool tmp_is_new_db = false; bool& is_new_db = recovery_ctx ? recovery_ctx->is_new_db_ : tmp_is_new_db; assert(db_lock_ == nullptr); @@ -575,6 +1041,25 @@ Status DBImpl::Recover( if (!s.ok()) { return s; } + if (!read_only && immutable_db_options_.memtable_crash_safe_recover) { + s = MapPublishedSeqFile(&recovery_ctx->crash_safe_probe_wal_below_); + if (!s.ok()) { + return s; + } + } else if (!read_only) { + // This session writes WALs without updating CSPUBSEQ, so an old + // one would misstate their format to a later crash-safe Open. + const std::string path = CrashSafePubSeqFileName(dbname_); + if (env_->FileExists(path).ok()) { + s = env_->DeleteFile(path); + if (!s.ok()) { + return s; + } + ROCKS_LOG_WARN(immutable_db_options_.info_log, + "memtable_crash_safe_recover is off, deleted %s", + path.c_str()); + } + } if (s.ok() && !read_only) { for (auto cfd : *versions_->GetColumnFamilySet()) { // Try to trivially move files down the LSM tree to start from bottommost @@ -675,10 +1160,64 @@ Status DBImpl::Recover( } s = SetupDBId(read_only, recovery_ctx); ROCKS_LOG_INFO(immutable_db_options_.info_log, "DB ID: %s\n", db_id_.c_str()); + bool crash_safe_convert = false; + bool recovered_wals_ok = false; + PublishedSeqRecord mmap_rec; + bool mmap_valid = false; + if (s.ok() && !read_only && + immutable_db_options_.memtable_crash_safe_recover) { + mmap_valid = ReadPublishedSeqRecord(pubseq_mmap_, &mmap_rec); + std::string fail_reason; + if (mmap_valid) { + crash_safe_convert = CanConvertLeftoverForCrashSafeRecover( + leftover_snapshot, mmap_rec.pubseq, mmap_rec.wal_number, &fail_reason); + } else { + fail_reason = "invalid CSPUBSEQ generation"; + } + if (!crash_safe_convert) { + ROCKS_LOG_WARN(immutable_db_options_.info_log, + "Crash-safe recover check failed (%s), fallback to full " + "WAL RecoverLogFiles", + fail_reason.c_str()); + for (const auto& leftover_path : leftover_snapshot) { + ROCKS_LOG_WARN(immutable_db_options_.info_log, + "Crash-safe leftover not converted: %s", + leftover_path.c_str()); + } + } + } if (s.ok() && !read_only) { + if (pubseq_mmap_ != nullptr) { + // Recovery can consume only part of the leftover set before failing. + // Keep the cursor invalid across a second crash, including one after + // orphan/failed-conversion cleanup has removed the failure evidence. + pubseq_mmap_->generation |= 1; + recovery_ctx->restore_published_seq_ = mmap_valid; + } s = DeleteUnreferencedSstFiles(recovery_ctx); } - + if (s.ok() && crash_safe_convert) { + const Status cs = ConvertLeftoverMemtables(leftover_snapshot, mmap_rec.pubseq, + recovery_ctx); + if (!cs.ok()) { + ROCKS_LOG_WARN(immutable_db_options_.info_log, + "Crash-safe leftover Convert failed (%s), fallback to " + "full WAL RecoverLogFiles", + cs.ToString().c_str()); + crash_safe_convert = false; + } else if (mmap_rec.wal_number != 0 || mmap_rec.wal_offset != 0) { + recovery_ctx->crash_safe_wal_tail_replay_ = true; + recovery_ctx->crash_safe_wal_number_ = mmap_rec.wal_number; + recovery_ctx->crash_safe_wal_offset_ = mmap_rec.wal_offset; + recovery_ctx->crash_safe_wal_offset_kind_ = + static_cast(pubseq_mmap_->wal_offset_kind); + ROCKS_LOG_INFO(immutable_db_options_.info_log, + "Crash-safe recover Convert leftover, WAL tail from " + "#%" PRIu64 " offset %" PRIu64 " kind %u seq %" PRIu64, + mmap_rec.wal_number, mmap_rec.wal_offset, + pubseq_mmap_->wal_offset_kind, mmap_rec.pubseq); + } + } if (immutable_db_options_.paranoid_checks && s.ok()) { s = CheckConsistency(); } @@ -801,6 +1340,7 @@ Status DBImpl::Recover( bool corrupted_wal_found = false; s = RecoverLogFiles(wals, &next_sequence, read_only, &corrupted_wal_found, recovery_ctx); + recovered_wals_ok = s.ok(); if (corrupted_wal_found && recovered_seq != nullptr) { *recovered_seq = next_sequence; } @@ -814,6 +1354,16 @@ Status DBImpl::Recover( } } + if (s.ok() && recovered_wals_ok && !crash_safe_convert && + !leftover_snapshot.empty()) { + for (const auto& leftover_path : leftover_snapshot) { + ROCKS_LOG_INFO(immutable_db_options_.info_log, + "Crash-safe leftover deleted after WAL fallback: %s", + leftover_path.c_str()); + env_->DeleteFile(leftover_path).PermitUncheckedError(); + } + } + if (read_only) { // If we are opening as read-only, we need to update options_file_number_ // to reflect the most recent OPTIONS file. It does not matter for regular @@ -853,6 +1403,14 @@ Status DBImpl::Recover( versions_->options_file_size_ = options_file_size; } } + if (s.ok() && !read_only && + immutable_db_options_.memtable_crash_safe_recover) { + if (mmap_valid && mmap_rec.pubseq > versions_->LastSequence()) { + versions_->SetLastAllocatedSequence(mmap_rec.pubseq); + versions_->SetLastPublishedSequence(mmap_rec.pubseq); + versions_->SetLastSequence(mmap_rec.pubseq); + } + } return s; } @@ -984,6 +1542,9 @@ Status DBImpl::LogAndApplyForRecovery(const RecoveryContext& recovery_ctx) { } mutex_.Lock(); } + if (s.ok() && recovery_ctx.restore_published_seq_) { + pubseq_mmap_->generation += 1; + } return s; } @@ -1150,6 +1711,16 @@ Status DBImpl::RecoverLogFiles(const std::vector& wal_numbers, bool flushed = false; uint64_t corrupted_wal_number = kMaxSequenceNumber; uint64_t min_wal_number = MinLogNumberToKeep(); + const bool crash_safe_wal_tail_replay = + recovery_ctx != nullptr && recovery_ctx->crash_safe_wal_tail_replay_; + const uint64_t crash_safe_wal_number = + crash_safe_wal_tail_replay ? recovery_ctx->crash_safe_wal_number_ : 0; + const uint64_t crash_safe_wal_offset = + crash_safe_wal_tail_replay ? recovery_ctx->crash_safe_wal_offset_ : 0; + const uint8_t crash_safe_wal_offset_kind = + crash_safe_wal_tail_replay ? recovery_ctx->crash_safe_wal_offset_kind_ : 0; + const uint64_t crash_safe_probe_wal_below = + recovery_ctx != nullptr ? recovery_ctx->crash_safe_probe_wal_below_ : 0; if (!allow_2pc()) { // In non-2pc mode, we skip WALs that do not back unflushed data. min_wal_number = @@ -1167,6 +1738,14 @@ Status DBImpl::RecoverLogFiles(const std::vector& wal_numbers, // records after allocating this log number. So we manually // update the file number allocation counter in VersionSet. versions_->MarkFileNumberUsed(wal_number); + if (crash_safe_wal_tail_replay && !allow_2pc() && + wal_number < crash_safe_wal_number) { + ROCKS_LOG_INFO(immutable_db_options_.info_log, + "Skipping InsertInto for log #%" PRIu64 + " before crash-safe WAL cursor #%" PRIu64, + wal_number, crash_safe_wal_number); + continue; + } // Open the log file std::string fname = LogFileName(immutable_db_options_.GetWalDir(), wal_number); @@ -1187,7 +1766,8 @@ Status DBImpl::RecoverLogFiles(const std::vector& wal_numbers, continue; } bool wal_memtable_format = immutable_db_options_.memtable_as_log_index; - if (immutable_db_options_.check_wal_format) { + if (immutable_db_options_.check_wal_format || wal_number < crash_safe_probe_wal_below) { + TEST_SYNC_POINT_CALLBACK("DBImpl::RecoverLogFiles:ProbeWalFormat", &wal_number); if (IOStatus ios = log::Reader::IsMemTableAsLogIndexFile (*fs_, fname, &wal_memtable_format); !ios.ok()) { auto info_log = immutable_db_options_.info_log.get(); @@ -1196,6 +1776,8 @@ Status DBImpl::RecoverLogFiles(const std::vector& wal_numbers, } } + bool prefix_compare = false; + reopen_wal: std::unique_ptr file_reader; { std::unique_ptr file; @@ -1244,6 +1826,42 @@ Status DBImpl::RecoverLogFiles(const std::vector& wal_numbers, fmap->tail_pos = std::make_shared(fmap->size_); } + if (crash_safe_wal_tail_replay && !prefix_compare) { + const uint64_t cursor_wal = crash_safe_wal_number; + bool has_udt = false; + for (auto* cfd : *versions_->GetColumnFamilySet()) { + if (cfd->user_comparator()->timestamp_size() > 0) { + has_udt = true; + break; + } + } + if (allow_2pc()) { + prefix_compare = true; + } else if (wal_number == cursor_wal) { + if (has_udt) { + ROCKS_LOG_WARN(immutable_db_options_.info_log, + "Crash-safe WAL Seek skipped (UDT), " + "compare LastRecordOffset from start of log #%" PRIu64, + wal_number); + prefix_compare = true; + } else { + const uint64_t record_start = PublishedWalRecordStart( + crash_safe_wal_offset, crash_safe_wal_offset_kind); + const IOStatus seek_s = reader.SeekToFileOffset(record_start); + if (!seek_s.ok()) { + ROCKS_LOG_WARN(immutable_db_options_.info_log, + "Crash-safe SeekToFileOffset(%" PRIu64 + ") failed (%s), compare LastRecordOffset from start", + record_start, seek_s.ToString().c_str()); + prefix_compare = true; + // A failed seek may have consumed bytes. Recreate both readers + // before replaying from the beginning of this WAL. + goto reopen_wal; + } + } + } + } + // Determine if we should tolerate incomplete records at the tail end of the // Read all the records and add to a memtable std::string scratch; @@ -1344,11 +1962,26 @@ Status DBImpl::RecoverLogFiles(const std::vector& wal_numbers, if (fmap) { batch_to_use->SetWAL({fmap, wal_number, reader.LastRecordOffset()}); } + column_family_memtables_->skip_memtable_data_ = false; + if (crash_safe_wal_tail_replay && prefix_compare) { + if (wal_number < crash_safe_wal_number) { + column_family_memtables_->skip_memtable_data_ = true; + } else if (wal_number == crash_safe_wal_number && + reader.LastRecordOffset() < + PublishedWalRecordStart(crash_safe_wal_offset, + crash_safe_wal_offset_kind)) { + column_family_memtables_->skip_memtable_data_ = true; + } + } status = WriteBatchInternal::InsertInto( batch_to_use, column_family_memtables_.get(), &flush_scheduler_, &trim_history_scheduler_, true, wal_number, this, false /* concurrent_memtable_writes */, next_sequence, &has_valid_writes, seq_per_batch_, batch_per_txn_); + column_family_memtables_->skip_memtable_data_ = false; + if (status.IsNotSupported()) { + return status; + } MaybeIgnoreError(&status); if (!status.ok()) { // We are treating this as a failure while reading since we read valid @@ -2135,6 +2768,10 @@ Status DBImpl::Open(const DBOptions& db_options, const std::string& dbname, } } } + if (s.ok() && impl->pubseq_mmap_ != nullptr && + impl->pubseq_mmap_->kind_since_wal == 0) { + impl->pubseq_mmap_->kind_since_wal = impl->logfile_number_; + } if (s.ok()) { s = impl->LogAndApplyForRecovery(recovery_ctx); } diff --git a/db/db_impl/db_impl_write.cc b/db/db_impl/db_impl_write.cc index 70e0e5b65..2211106a5 100644 --- a/db/db_impl/db_impl_write.cc +++ b/db/db_impl/db_impl_write.cc @@ -369,8 +369,16 @@ Status DBImpl::WriteImpl(const WriteOptions& write_options, } } versions_->SetLastSequence(last_sequence); + MaybePersistPublishedSequence(last_sequence, *w.write_group); + if (pubseq_mmap_ != nullptr && two_write_queues_ && + !immutable_db_options_.unordered_write) { + AccountPendingMemtableWrites(1); + } MemTableInsertStatusCheck(w.status); write_thread_.ExitAsBatchGroupFollower(&w); + } else if (pubseq_mmap_ != nullptr && two_write_queues_ && + !immutable_db_options_.unordered_write) { + AccountPendingMemtableWrites(1); } assert(w.state == WriteThread::STATE_COMPLETED); // STATE_COMPLETED conditional below handles exit @@ -427,6 +435,7 @@ Status DBImpl::WriteImpl(const WriteOptions& write_options, write_thread_.EnterAsBatchGroupLeader(&w, &write_group); IOStatus io_s; + bool account_two_q_pending = false; Status pre_release_cb_status; if (status.ok()) { // Rules for when we can update the memtable concurrently @@ -530,6 +539,9 @@ Status DBImpl::WriteImpl(const WriteOptions& write_options, // wal_write_mutex_ to ensure ordered events in WAL io_s = ConcurrentWriteToWAL(write_group, log_used, &last_sequence, seq_inc); + account_two_q_pending = + io_s.ok() && pubseq_mmap_ != nullptr && + !immutable_db_options_.unordered_write; } else { // Otherwise we inc seq number for memtable writes last_sequence = versions_->FetchAddLastAllocatedSequence(seq_inc); @@ -672,9 +684,15 @@ Status DBImpl::WriteImpl(const WriteOptions& write_options, // Note: if we are to resume after non-OK statuses we need to revisit how // we reacts to non-OK statuses here. versions_->SetLastSequence(last_sequence); + MaybePersistPublishedSequence(last_sequence, write_group); + } + if (account_two_q_pending) { + AccountPendingMemtableWrites(in_parallel_group ? 1 : write_group.size); } MemTableInsertStatusCheck(w.status); write_thread_.ExitAsBatchGroupLeader(write_group, status); + } else if (in_parallel_group && account_two_q_pending) { + AccountPendingMemtableWrites(1); } if (status.ok()) { @@ -691,6 +709,7 @@ Status DBImpl::PipelinedWriteImpl(const WriteOptions& write_options, StopWatch write_sw(immutable_db_options_.clock, stats_, DB_WRITE); WriteContext write_context; + WriteThread::WriteGroup memtable_write_group; WriteThread::Writer w(write_options, my_batch, callback, log_ref, disable_memtable, /*_batch_cnt=*/0, @@ -807,15 +826,21 @@ Status DBImpl::PipelinedWriteImpl(const WriteOptions& write_options, const ReadOptions read_options; w.status = ApplyWALToManifest(read_options, &synced_wals); } + if (w.status.ok() && pubseq_mmap_ != nullptr) { + size_t memtable_write_cnt = 0; + for (auto* writer : wal_write_group) { + if (writer->ShouldWriteToMemtable()) { + memtable_write_cnt++; + } + } + StagePublishedWal(current_sequence + total_count - 1, + wal_write_group.wal_number, + wal_write_group.wal_offset); + pending_memtable_writes_ += memtable_write_cnt; + } write_thread_.ExitAsBatchGroupLeader(wal_write_group, w.status); } - // NOTE: the memtable_write_group is declared before the following - // `if` statement because its lifetime needs to be longer - // that the inner context of the `if` as a reference to it - // may be used further below within the outer _write_thread - WriteThread::WriteGroup memtable_write_group; - if (w.state == WriteThread::STATE_MEMTABLE_WRITER_LEADER) { PERF_TIMER_WITH_HISTOGRAM(write_memtable_time, MEMTAB_WRITE_KV_NANOS, stats_); assert(w.ShouldWriteToMemtable()); @@ -830,6 +855,9 @@ Status DBImpl::PipelinedWriteImpl(const WriteOptions& write_options, write_options.ignore_missing_column_families, 0 /*log_number*/, this, false /*concurrent_memtable_writes*/, seq_per_batch_, batch_per_txn_); versions_->SetLastSequence(memtable_write_group.last_sequence); + if (pubseq_mmap_ != nullptr) { + AccountPendingMemtableWrites(memtable_write_group.size); + } write_thread_.ExitAsMemTableWriter(&w, memtable_write_group); } } else { @@ -848,6 +876,9 @@ Status DBImpl::PipelinedWriteImpl(const WriteOptions& write_options, 0 /*log_number*/, this, true /*concurrent_memtable_writes*/, false /*seq_per_batch*/, 0 /*batch_cnt*/, true /*batch_per_txn*/, write_options.memtable_insert_hint_per_batch); + if (pubseq_mmap_ != nullptr) { + AccountPendingMemtableWrites(1); + } if (write_thread_.CompleteParallelMemTableWriter(&w)) { MemTableInsertStatusCheck(w.status); versions_->SetLastSequence(w.write_group->last_sequence); @@ -893,15 +924,7 @@ Status DBImpl::UnorderedWriteMemtable(const WriteOptions& write_options, } } - size_t pending_cnt = pending_memtable_writes_.fetch_sub(1) - 1; - if (pending_cnt == 0) { - // switch_cv_ waits until pending_memtable_writes_ = 0. Locking its mutex - // before notify ensures that cv is in waiting state when it is notified - // thus not missing the update to pending_memtable_writes_ even though it is - // not modified under the mutex. - std::lock_guard lck(switch_mutex_); - switch_cv_.notify_all(); - } + AccountPendingMemtableWrites(1); WriteStatusCheck(w.status); if (!w.FinalStatus().ok()) { @@ -1112,8 +1135,13 @@ Status DBImpl::WriteImplWALOnly( // Currently we only use kDoPublishLastSeq in unordered_write assert(immutable_db_options_.unordered_write); } - if (immutable_db_options_.unordered_write && status.ok()) { + if (immutable_db_options_.unordered_write && status.ok() && + pubseq_mmap_ == nullptr) { pending_memtable_writes_ += memtable_write_cnt; + } else if (immutable_db_options_.unordered_write && !status.ok() && + pubseq_mmap_ != nullptr) { + // ConcurrentWriteToWAL counted this group, but it will skip memtable writes. + AccountPendingMemtableWrites(memtable_write_cnt); } write_thread->ExitAsBatchGroupLeader(write_group, status); if (status.ok()) { @@ -1489,6 +1517,12 @@ IOStatus DBImpl::DoWriteWAL(const WriteBatch& merged_batch, total_log_size_.fetch_add(log_size_sum, std::memory_order_relaxed); log_file_number_size.AddSize(log_size_sum); log_empty_ = false; + if (io_s.ok() && pubseq_mmap_ != nullptr) { + write_group.wal_number = logfile_number_; + write_group.wal_offset = immutable_db_options_.memtable_as_log_index + ? log_writer->get_log_offset() + : log_writer->file()->GetFileSize(); + } return io_s; } @@ -1627,6 +1661,37 @@ IOStatus DBImpl::ConcurrentWriteToWAL( cached_recoverable_state_ = *to_be_cached_state; cached_recoverable_state_empty_ = false; } + if (io_s.ok() && pubseq_mmap_ != nullptr) { + const SequenceNumber pub_seq = *last_sequence + seq_inc; + if (write_group.leader->disable_memtable) { + StagePublishedWal(pub_seq, write_group.wal_number, + write_group.wal_offset); + PersistStagedPublishedWal(); + } else if (immutable_db_options_.unordered_write) { + StagePublishedWal(pub_seq, write_group.wal_number, + write_group.wal_offset); + } + } + if (io_s.ok() && pubseq_mmap_ != nullptr && + immutable_db_options_.unordered_write) { + size_t n = 0; + for (auto* writer : write_group) { + if (!writer->CallbackFailed() && !writer->disable_memtable) { + n++; + } + } + pending_memtable_writes_ += n; + } else if (io_s.ok() && pubseq_mmap_ != nullptr && two_write_queues_ && + !immutable_db_options_.unordered_write && + !write_group.leader->disable_memtable) { + size_t n = 0; + for (auto* writer : write_group) { + if (!writer->disable_memtable) { + n++; + } + } + pending_memtable_writes_ += n; + } log_write_mutex_.Unlock(); if (io_s.ok()) { diff --git a/db/db_write_test.cc b/db/db_write_test.cc index 4de24dcb7..92b810d14 100644 --- a/db/db_write_test.cc +++ b/db/db_write_test.cc @@ -39,6 +39,20 @@ class DBWriteTestUnparameterized : public DBTestBase { : DBTestBase("pipelined_write_test", /*env_do_fsync=*/false) {} }; +TEST_F(DBWriteTestUnparameterized, UnorderedWriteWithoutWALCanFlush) { + Options options = CurrentOptions(); + options.unordered_write = true; + Reopen(options); + WriteOptions write; + for (bool disable_wal : {false, true}) { + write.disableWAL = disable_wal; + ASSERT_OK(db_->Put(write, "key", disable_wal ? "no-wal" : "wal")); + ASSERT_OK(Flush()); + } + Reopen(options); + ASSERT_EQ(Get("key"), "no-wal"); +} + // It is invalid to do sync write while disabling WAL. TEST_P(DBWriteTest, SyncAndDisableWAL) { if (GetOptions().memtable_as_log_index) { diff --git a/db/write_batch.cc b/db/write_batch.cc index 5e0423aba..1542c751c 100644 --- a/db/write_batch.cc +++ b/db/write_batch.cc @@ -2300,6 +2300,7 @@ class MemTableInserter : public WriteBatch::Handler { hint_per_batch_(hint_per_batch) { assert(cf_mems_); memtable_as_log_index_ = cf_mems->GetImmutableDBOptions()->memtable_as_log_index; + skip_memtable_data_ = cf_mems->skip_memtable_data_; } bool memtable_as_log_index() const { return memtable_as_log_index_; } void SetBeginPrepareNextPtr(const char* curr) final { @@ -2313,6 +2314,7 @@ class MemTableInserter : public WriteBatch::Handler { } const WriteBatch* src_batch_ = nullptr; const char* prepare_content_begin_ = nullptr; + bool skip_memtable_data_ = false; ~MemTableInserter() override { if (dup_dectector_on_) { @@ -2397,6 +2399,10 @@ class MemTableInserter : public WriteBatch::Handler { *s = Status::OK(); return false; } + if (skip_memtable_data_) { + *s = Status::OK(); + return false; + } if (has_valid_writes_ != nullptr) { *has_valid_writes_ = true; diff --git a/db/write_batch_internal.h b/db/write_batch_internal.h index ba6f5c5d7..979da2f80 100644 --- a/db/write_batch_internal.h +++ b/db/write_batch_internal.h @@ -41,6 +41,7 @@ class ColumnFamilyMemTables { virtual ColumnFamilyHandle* GetColumnFamilyHandle() = 0; virtual ColumnFamilyData* current() { return nullptr; } virtual const ImmutableDBOptions* GetImmutableDBOptions() = 0; + bool skip_memtable_data_ = false; }; class ColumnFamilyMemTablesDefault : public ColumnFamilyMemTables { diff --git a/db/write_thread.h b/db/write_thread.h index 633d8bc2d..df2ec2e07 100644 --- a/db/write_thread.h +++ b/db/write_thread.h @@ -83,6 +83,9 @@ class WriteThread { Status status; std::atomic running; size_t size = 0; + // WAL cursor of this group after WriteToWAL (not log_file_number_size). + mutable uint64_t wal_number = 0; + mutable uint64_t wal_offset = 0; struct Iterator { Writer* writer; diff --git a/file/filename.cc b/file/filename.cc index fb7d25472..c78cc8129 100644 --- a/file/filename.cc +++ b/file/filename.cc @@ -178,6 +178,10 @@ std::string CurrentFileName(const std::string& dbname) { std::string LockFileName(const std::string& dbname) { return dbname + "/LOCK"; } +std::string CrashSafePubSeqFileName(const std::string& dbname) { + return dbname + "/CSPUBSEQ"; +} + std::string TempFileName(const std::string& dbname, uint64_t number) { return MakeFileName(dbname, number, kTempFileNameSuffix.c_str()); } diff --git a/file/filename.h b/file/filename.h index 2eb125b6a..be747e7c9 100644 --- a/file/filename.h +++ b/file/filename.h @@ -100,6 +100,9 @@ extern std::string CurrentFileName(const std::string& dbname); // "dbname". The result will be prefixed with "dbname". extern std::string LockFileName(const std::string& dbname); +// Crash-safe recover publish-seq mmap under dbname. +extern std::string CrashSafePubSeqFileName(const std::string& dbname); + // Return the name of a temporary file owned by the db named "dbname". // The result will be prefixed with "dbname". extern std::string TempFileName(const std::string& dbname, uint64_t number); diff --git a/include/rocksdb/options.h b/include/rocksdb/options.h index b80d59309..69d316dc0 100644 --- a/include/rocksdb/options.h +++ b/include/rocksdb/options.h @@ -933,9 +933,14 @@ struct DBOptions { bool memtable_as_log_index = false; - // Enable crash-safe recovery using mmap MemTables (CSPP / OffsetSkipList) + // Enable crash-safe recovery using file mmap MemTables (CSPP / OffsetSkipList) // and ConvertToSST to reduce WAL replay on the next open. // Only process crashes (kill -9 / _exit) are covered, not power loss. + // DeleteRange is not covered. + // Reuse leftover MemTable files up to the published sequence, then replay + // the WAL tail. All CFs must support crash-safe file mmap MemTables. + // Unusable leftovers or cursor metadata, wal_filter, best_efforts_recovery, + // and unsupported sequence publication modes fall back to full WAL replay. // Independent of memtable_as_log_index and Flush/Close ConvertToSST. // Requires WAL: disableWAL is ignored; manual_wal_flush and WAL recycling // are disabled, and wal_compression is forced to kNoCompression. diff --git a/src.mk b/src.mk index 25966e97d..141e953c3 100644 --- a/src.mk +++ b/src.mk @@ -483,6 +483,7 @@ TEST_MAIN_SOURCES = \ db/db_compaction_filter_test.cc \ db/db_compaction_test.cc \ db/db_clip_test.cc \ + db/db_cspp_crash_safe_test.cc \ db/db_dynamic_level_test.cc \ db/db_encryption_test.cc \ db/db_flush_test.cc \ From 45c2995390dfaa754c8480153c895f482e71bef3 Mon Sep 17 00:00:00 2001 From: leipeng Date: Sun, 27 Sep 2026 23:08:54 +0800 Subject: [PATCH 16/58] DB Recovery: Recover when the WAL kind changes MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit When the requested memtable_as_log_index differs from the saved wal_offset_kind, the previous commit did not handle this scenario. In this commit: ⇒ 1. If CSPUBSEQ has an established kind and it differs, Open first in the saved kind, flush, delete CSPUBSEQ, then Open in the requested kind. ⇒ 2. Prepared transactions recovered on that first Open stop the switch. Resolve them, then Open again. With allow_2pc and none left, the old-format WALs are retired so the next Open does not replay them into the new kind. ⇒ 3. A kind stamped but never used for a WAL is not established, so this path does not run. An odd generation does not hide the kind. ⇒ 4. DBImpl::Open still rejects the mismatch. DB::Open and TransactionDB::Open run the switch first. --- db/db_cspp_crash_safe_test.cc | 478 +++++++++++++++++- db/db_impl/db_impl.h | 16 +- db/db_impl/db_impl_open.cc | 154 +++++- include/rocksdb/options.h | 9 +- .../pessimistic_transaction_db.cc | 10 +- 5 files changed, 656 insertions(+), 11 deletions(-) diff --git a/db/db_cspp_crash_safe_test.cc b/db/db_cspp_crash_safe_test.cc index d2e901761..b36a1a7f4 100644 --- a/db/db_cspp_crash_safe_test.cc +++ b/db/db_cspp_crash_safe_test.cc @@ -208,6 +208,18 @@ class CrashChild : public ::testing::Test { const std::string arg_ = crash_child_arg ? crash_child_arg : ""; }; +const int kKindPrepChildCrashed = 42; + +TEST_F(CrashChild, DISABLED_KindPrep) { + Options log_index = BaseCrashSafeOptions(dbname_, true, true); + log_index.avoid_flush_during_shutdown = true; + SyncPoint::GetInstance()->SetCallBack( + arg_, [](void*) { ::_exit(kKindPrepChildCrashed); }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* db = nullptr; + Status s = DB::Open(log_index, dbname_, &db); + ::_exit(s.ok() ? 1 : 2); +} #endif } // namespace @@ -1185,8 +1197,8 @@ TEST_F(DBCsppCrashSafeTest, ClassicWalToLogIndexWithoutSidecarIsNotSupported) { ASSERT_EQ(rec.generation & 1, 1U); } - // Failed Open must not retain the unestablished log-index kind. - // Reopen in the original format without deleting the sidecar. + // Failed Open must not force KindPrep with the unestablished log-index + // kind. Reopen in the original format without deleting the sidecar. classic.memtable_crash_safe_recover = true; ASSERT_OK(TryReopen(classic)); ASSERT_EQ(Get("large"), value); @@ -1195,6 +1207,11 @@ TEST_F(DBCsppCrashSafeTest, ClassicWalToLogIndexWithoutSidecarIsNotSupported) { ASSERT_EQ(rec.wal_offset_kind, 1U); ASSERT_GT(rec.kind_since_wal, 0U); Close(); + + // An established sidecar still permits the existing KindPrep switch. + ASSERT_OK(TryReopen(log_index)); + ASSERT_EQ(Get("large"), value); + Close(); } } } @@ -1236,6 +1253,58 @@ TEST_F(DBCsppCrashSafeTest, RecoverOffDeletesStaleSidecar) { ASSERT_EQ(Get("k"), "v"); } +TEST_F(DBCsppCrashSafeTest, OddGenerationKindSwitchUsesKindPrep) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.avoid_flush_during_shutdown = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "v")); + Close(); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_TRUE(SetPublishedSeqGeneration(dbname_, rec.generation | 1)); + Options log_index = BaseCrashSafeOptions(dbname_, true, true); + log_index.avoid_flush_during_shutdown = true; + int prep = 0; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::Open::KindPrep:AfterOpenBeforeFlush", + [&prep](void*) { prep++; }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(log_index)); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_EQ(prep, 1); + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.wal_offset_kind, 2U); + ASSERT_EQ(Get("k"), "v"); +} + +TEST_F(DBCsppCrashSafeTest, DirectDBImplOpenKindMismatchFails) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.avoid_flush_during_shutdown = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "v")); + Close(); + Options log_index = BaseCrashSafeOptions(dbname_, true, true); + std::vector cfs = { + ColumnFamilyDescriptor(kDefaultColumnFamilyName, log_index)}; + std::vector handles; + DB* db = nullptr; + const Status s = DBImpl::Open(DBOptions(log_index), dbname_, cfs, &handles, + &db, false /*seq_per_batch*/, + true /*batch_per_txn*/); + ASSERT_TRUE(s.IsInvalidArgument()) << s.ToString(); + ASSERT_EQ(db, nullptr); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.wal_offset_kind, 1U); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("k"), "v"); +} + TEST_F(DBCsppCrashSafeTest, OddGenerationPublishesEvenAfterWalRecovery) { for (bool log_index : {false, true}) { SCOPED_TRACE(log_index); @@ -2090,6 +2159,411 @@ TEST_F(DBCsppCrashSafeTest, PublishedSeqFieldsAfterPut) { ASSERT_EQ(rec.wal_offset_kind, 1U); // Persist does not rewrite kind } +TEST_F(DBCsppCrashSafeTest, ExistingPublishedSeqKindPrepLogIndexAfterClassic) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.avoid_flush_during_shutdown = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "v")); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.wal_offset_kind, 1U); + Close(); + ASSERT_OK(env_->FileExists(CrashSafePubSeqFileName(dbname_))); + Options log_index = BaseCrashSafeOptions(dbname_, true, true); + log_index.avoid_flush_during_shutdown = true; + ASSERT_OK(TryReopen(log_index)); + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.wal_offset_kind, 2U); + ASSERT_EQ(Get("k"), "v"); + ASSERT_GE(CountL0(db_), 1); +} + +TEST_F(DBCsppCrashSafeTest, ExistingPublishedSeqKindPrepClassicAfterLogIndex) { + Close(); + Options log_index = BaseCrashSafeOptions(dbname_, true, true); + log_index.avoid_flush_during_shutdown = true; + Destroy(log_index); + ASSERT_OK(TryReopen(log_index)); + ASSERT_OK(Put("k", "v")); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.wal_offset_kind, 2U); + Close(); + ASSERT_OK(env_->FileExists(CrashSafePubSeqFileName(dbname_))); + Options classic = BaseCrashSafeOptions(dbname_, true, false); + classic.avoid_flush_during_shutdown = true; + ASSERT_OK(TryReopen(classic)); + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.wal_offset_kind, 1U); + ASSERT_EQ(Get("k"), "v"); + ASSERT_GE(CountL0(db_), 1); +} + +TEST_F(DBCsppCrashSafeTest, KindPrepSkippedWhenKindMatches) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.avoid_flush_during_shutdown = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "v")); + Close(); + int prep = 0; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::Open::KindPrep:AfterOpenBeforeFlush", + [&](void*) { prep++; }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(options)); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_EQ(prep, 0); + ASSERT_EQ(Get("k"), "v"); +} + +TEST_F(DBCsppCrashSafeTest, KindPrepFailureClearsDbPointer) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.avoid_flush_during_shutdown = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "v")); + Close(); + options.memtable_as_log_index = true; + options.error_if_exists = true; + DB* db = reinterpret_cast(uintptr_t(1)); + Status s = DB::Open(options, dbname_, &db); + ASSERT_TRUE(s.IsInvalidArgument()) << s.ToString(); + ASSERT_EQ(db, nullptr); +} + +#if !defined(OS_WIN) +TEST_F(CrashChild, DISABLED_EasyMigrateKindPrep) { + ASSERT_EQ(arg_.size(), 3U); + const bool log_index = arg_[0] == '1'; + const bool txn = arg_[1] == '1'; + const bool recover = arg_[2] == '1'; + // EasyMigrate caches its config, so load it only in this fresh exec child. + const std::string config = dbname_ + ".easy-migrate.json"; + ASSERT_EQ(::setenv("TOPLINGDB_EASY_MIGRATE_CONF", config.c_str(), 1), 0); + Options options = BaseCrashSafeOptions(dbname_, recover, log_index); + int prep = 0; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::Open::KindPrep:AfterOpenBeforeFlush", [&](void*) { ++prep; }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* db = nullptr; + if (txn) { + TransactionDB* txn_db = nullptr; + ASSERT_OK(TransactionDB::Open(options, TransactionDBOptions(), dbname_, + &txn_db)); + db = txn_db; + } else { + ASSERT_OK(DB::Open(options, dbname_, &db)); + } + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_EQ(prep, 1); + std::string value; + ASSERT_OK(db->Get(ReadOptions(), "k", &value)); + ASSERT_EQ(value, "v"); + ASSERT_OK(db->Put(WriteOptions(), "after", "switch")); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.wal_offset_kind, log_index ? 2U : 1U); + ASSERT_OK(db->Close()); + delete db; + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, EasyMigrateKindPrep) { + for (bool log_index : {false, true}) { + for (bool txn : {false, true}) { + for (bool recover : {false, true}) { + const std::string mode = {char('0' + log_index), char('0' + txn), + char('0' + recover)}; + SCOPED_TRACE(mode); + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, !log_index); + options.avoid_flush_during_shutdown = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "v")); + // This convert-only table factory cannot BuildTable during 2PC replay. + ASSERT_OK(Flush()); + Close(); + const std::string config = dbname_ + ".easy-migrate.json"; + const json conf = { + {"DBOptions", {{"default", {{"memtable_crash_safe_recover", true}, + {"memtable_as_log_index", log_index}}}}}, + {"http", {{"auto_start_http", false}}}}; + ASSERT_OK(WriteStringToFile(env_, conf.dump(), config)); + ASSERT_EQ(RunCrashChild(dbname_, "EasyMigrateKindPrep", mode), 0); + ASSERT_OK(env_->DeleteFile(config)); + options.memtable_as_log_index = log_index; + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("k"), "v"); + ASSERT_EQ(Get("after"), "switch"); + } + } + } +} + +TEST_F(CrashChild, DISABLED_KindPrepKeepsUnpublishedWalTail) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.avoid_flush_during_shutdown = true; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::WriteImpl:AfterWriteToWALBeforePublish", + [](void*) { ::_exit(1); }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "b", "2")); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, KindPrepKeepsUnpublishedWalTail) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.avoid_flush_during_shutdown = true; + Destroy(options); + { + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("a", "1")); + Close(); + } + ASSERT_EQ(RunCrashChild(dbname_, "KindPrepKeepsUnpublishedWalTail"), 1); + Options log_index = BaseCrashSafeOptions(dbname_, true, true); + log_index.avoid_flush_during_shutdown = true; + ASSERT_OK(TryReopen(log_index)); + ASSERT_EQ(Get("a"), "1"); + ASSERT_EQ(Get("b"), "2"); +} + +TEST_F(DBCsppCrashSafeTest, KindPrepAfterFlushBeforeClose) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.avoid_flush_during_shutdown = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "v")); + Close(); + ASSERT_EQ(RunCrashChild( + dbname_, "KindPrep", "DBImpl::Open::KindPrep:AfterFlushBeforeClose"), + kKindPrepChildCrashed); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.wal_offset_kind, 1U); + Options log_index = BaseCrashSafeOptions(dbname_, true, true); + log_index.avoid_flush_during_shutdown = true; + ASSERT_OK(TryReopen(log_index)); + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.wal_offset_kind, 2U); + ASSERT_EQ(Get("k"), "v"); +} + +TEST_F(DBCsppCrashSafeTest, KindPrepAfterCloseBeforeDeleteSidecar) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.avoid_flush_during_shutdown = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "v")); + Close(); + ASSERT_EQ(RunCrashChild( + dbname_, "KindPrep", "DBImpl::Open::KindPrep:AfterCloseBeforeDeleteSidecar"), + kKindPrepChildCrashed); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.wal_offset_kind, 1U); + Options log_index = BaseCrashSafeOptions(dbname_, true, true); + log_index.avoid_flush_during_shutdown = true; + ASSERT_OK(TryReopen(log_index)); + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.wal_offset_kind, 2U); + ASSERT_EQ(Get("k"), "v"); +} + +TEST_F(CrashChild, DISABLED_TransactionDBKindPrepClassicToLogIndex) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.avoid_flush_during_shutdown = true; + TransactionDBOptions txn_opts; + txn_opts.write_policy = TxnDBWritePolicy::WRITE_COMMITTED; + TransactionDB* txn_db = nullptr; + ASSERT_OK(TransactionDB::Open(options, txn_opts, dbname_, &txn_db)); + Transaction* txn = txn_db->BeginTransaction(WriteOptions()); + ASSERT_OK(txn->Put("t", "1")); + ASSERT_OK(txn->Commit()); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, TransactionDBKindPrepClassicToLogIndex) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.avoid_flush_during_shutdown = true; + Destroy(options); + TransactionDBOptions txn_opts; + txn_opts.write_policy = TxnDBWritePolicy::WRITE_COMMITTED; + TransactionDB* txn_db = nullptr; + ASSERT_EQ(RunCrashChild(dbname_, "TransactionDBKindPrepClassicToLogIndex"), 0); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.wal_offset_kind, 1U); + Options log_index = BaseCrashSafeOptions(dbname_, true, true); + log_index.avoid_flush_during_shutdown = true; + int prep = 0; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::Open::KindPrep:AfterOpenBeforeFlush", + [&prep](void*) { prep++; }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TransactionDB::Open(log_index, txn_opts, dbname_, &txn_db)); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_EQ(prep, 1); + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.wal_offset_kind, 2U); + std::string v; + ASSERT_OK(txn_db->Get(ReadOptions(), "t", &v)); + ASSERT_EQ(v, "1"); + delete txn_db; +} + +TEST_F(CrashChild, DISABLED_TransactionDBKindPrepPreparedNotSupported) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.avoid_flush_during_shutdown = true; + TransactionDBOptions txn_opts; + txn_opts.write_policy = TxnDBWritePolicy::WRITE_COMMITTED; + TransactionDB* txn_db = nullptr; + ASSERT_OK(TransactionDB::Open(options, txn_opts, dbname_, &txn_db)); + TransactionOptions to; + to.skip_prepare = false; + Transaction* txn = txn_db->BeginTransaction(WriteOptions(), to); + ASSERT_OK(txn->SetName("xid1")); + ASSERT_OK(txn->Put("prep", "v")); + ASSERT_OK(txn->Prepare()); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, TransactionDBKindPrepPreparedNotSupported) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.avoid_flush_during_shutdown = true; + Destroy(options); + TransactionDBOptions txn_opts; + txn_opts.write_policy = TxnDBWritePolicy::WRITE_COMMITTED; + TransactionDB* txn_db = nullptr; + ASSERT_EQ(RunCrashChild(dbname_, "TransactionDBKindPrepPreparedNotSupported"), 0); + Options log_index = BaseCrashSafeOptions(dbname_, true, true); + log_index.avoid_flush_during_shutdown = true; + ASSERT_TRUE(TransactionDB::Open(log_index, txn_opts, dbname_, &txn_db) + .IsNotSupported()); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.wal_offset_kind, 1U); + ASSERT_OK(TransactionDB::Open(options, txn_opts, dbname_, &txn_db)); + std::vector prepared; + txn_db->GetAllPreparedTransactions(&prepared); + ASSERT_EQ(prepared.size(), 1U); + ASSERT_OK(prepared[0]->Rollback()); + delete prepared[0]; + delete txn_db; +} + +TEST_F(DBCsppCrashSafeTest, TransactionDBPreparedCloseThenSwitchKind) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + TransactionDBOptions txn_opts; + txn_opts.write_policy = TxnDBWritePolicy::WRITE_COMMITTED; + TransactionDB* txn_db = nullptr; + ASSERT_OK(TransactionDB::Open(options, txn_opts, dbname_, &txn_db)); + TransactionOptions to; + to.skip_prepare = false; + Transaction* txn = txn_db->BeginTransaction(WriteOptions(), to); + ASSERT_OK(txn->SetName("xid1")); + ASSERT_OK(txn->Put("prep", "v")); + ASSERT_OK(txn->Prepare()); + delete txn; + delete txn_db; // default avoid_flush_during_shutdown=false: Close converts + Options log_index = BaseCrashSafeOptions(dbname_, true, true); + ASSERT_TRUE(TransactionDB::Open(log_index, txn_opts, dbname_, &txn_db) + .IsNotSupported()); + ASSERT_OK(TransactionDB::Open(options, txn_opts, dbname_, &txn_db)); + std::vector prepared; + txn_db->GetAllPreparedTransactions(&prepared); + ASSERT_EQ(prepared.size(), 1U); + ASSERT_OK(prepared[0]->Rollback()); + delete prepared[0]; + delete txn_db; +} + +TEST_F(CrashChild, DISABLED_TransactionDBRollbackCloseThenSwitchKind) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + TransactionDBOptions txn_opts; + txn_opts.write_policy = TxnDBWritePolicy::WRITE_COMMITTED; + TransactionDB* txn_db = nullptr; + ASSERT_OK(TransactionDB::Open(options, txn_opts, dbname_, &txn_db)); + TransactionOptions to; + to.skip_prepare = false; + Transaction* txn = txn_db->BeginTransaction(WriteOptions(), to); + ASSERT_OK(txn->SetName("xid1")); + ASSERT_OK(txn->Put("prep", "v")); + ASSERT_OK(txn->Prepare()); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, TransactionDBRollbackCloseThenSwitchKind) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + TransactionDBOptions txn_opts; + txn_opts.write_policy = TxnDBWritePolicy::WRITE_COMMITTED; + TransactionDB* txn_db = nullptr; + ASSERT_OK(TransactionDB::Open(options, txn_opts, dbname_, &txn_db)); + ASSERT_OK(txn_db->Put(WriteOptions(), "k", "v")); + delete txn_db; + ASSERT_EQ(RunCrashChild(dbname_, "TransactionDBRollbackCloseThenSwitchKind"), 0); + ASSERT_OK(TransactionDB::Open(options, txn_opts, dbname_, &txn_db)); + std::vector prepared; + txn_db->GetAllPreparedTransactions(&prepared); + ASSERT_EQ(prepared.size(), 1U); + ASSERT_OK(prepared[0]->Rollback()); + delete prepared[0]; + delete txn_db; // default avoid_flush_during_shutdown=false: Close converts + Options log_index = BaseCrashSafeOptions(dbname_, true, true); + ASSERT_OK(TransactionDB::Open(log_index, txn_opts, dbname_, &txn_db)); + std::string v; + ASSERT_OK(txn_db->Get(ReadOptions(), "k", &v)); + ASSERT_EQ(v, "v"); + ASSERT_TRUE(txn_db->Get(ReadOptions(), "prep", &v).IsNotFound()); + delete txn_db; +} + +TEST_F(DBCsppCrashSafeTest, KindPrepConvertFailFallsBackToWal) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.avoid_flush_during_shutdown = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "v")); + Close(); + Options log_index = BaseCrashSafeOptions(dbname_, true, true); + log_index.avoid_flush_during_shutdown = true; + ASSERT_FALSE(log_index.check_wal_format); + ASSERT_OK(TryReopen(log_index)); + ASSERT_EQ(Get("k"), "v"); + Close(); + SyncPoint::GetInstance()->SetCallBack( + "CrashSafeRecover::ConvertToSST:InjectStatus", [](void* arg) { + *static_cast(arg) = Status::IOError("inject convert"); + }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(log_index)); + ASSERT_EQ(Get("k"), "v"); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); +} +#endif + TEST_F(DBCsppCrashSafeTest, DontConvertFallsBackToWal) { Close(); Options options = BaseCrashSafeOptions(dbname_, true, false); diff --git a/db/db_impl/db_impl.h b/db/db_impl/db_impl.h index f314c7878..c62f559ee 100644 --- a/db/db_impl/db_impl.h +++ b/db/db_impl/db_impl.h @@ -1077,7 +1077,16 @@ class DBImpl : public DB { static Status Open(const DBOptions& db_options, const std::string& name, const std::vector& column_families, std::vector* handles, DB** dbptr, - const bool seq_per_batch, const bool batch_per_txn); + const bool seq_per_batch, const bool batch_per_txn, + bool options_already_updated = false); + + // memtable_crash_safe_recover: if CSPUBSEQ's wal_offset_kind differs + // from memtable_as_log_index, open with the sidecar kind, flush, close and + // delete CSPUBSEQ. Call before Open() with the same seq/txn flags. + static Status PrepareCrashSafeKindForOpen( + const DBOptions& db_options, const std::string& name, + const std::vector& column_families, + bool seq_per_batch, bool batch_per_txn); static IOStatus CreateAndNewDirectory( FileSystem* fs, const std::string& dirname, @@ -1941,6 +1950,11 @@ class DBImpl : public DB { bool* corrupted_log_found, RecoveryContext* recovery_ctx); + // Kind-prep first pass under allow_2pc: NotSupported if prepared + // transactions were recovered, else persist min_log_number_to_keep past + // the old-format WALs so the next Open does not replay them. + Status RetireWalsForKindPrep(); + // *probe_wal_below: WALs numbered below it have unknown format. Status MapPublishedSeqFile(uint64_t* probe_wal_below); void UnmapPublishedSeqFile(); diff --git a/db/db_impl/db_impl_open.cc b/db/db_impl/db_impl_open.cc index f70a4ca64..79de49c20 100644 --- a/db/db_impl/db_impl_open.cc +++ b/db/db_impl/db_impl_open.cc @@ -523,6 +523,28 @@ bool ReadPublishedSeqRecord(const PublishedSeqMmapHeader* hdr, } } +bool PeekPublishedWalKind(const std::string& dbname, uint64_t* kind) { + const std::string path = CrashSafePubSeqFileName(dbname); + PublishedSeqMmapHeader hdr; + int fd = ::open(path.c_str(), O_RDONLY | O_CLOEXEC); + if (fd < 0) { + return false; + } + auto sz = ::pread(fd, &hdr, sizeof(hdr), 0); + ::close(fd); + if (sz != static_cast(sizeof(hdr))) { + return false; + } + // wal_offset_kind is written only by Stamp, so an odd generation from a + // crash mid-publish does not make it stale. + // A Stamp from an unfinished Open has not established a WAL format yet. + bool ok = PublishedSeqHeaderValid(&hdr) && hdr.kind_since_wal != 0; + if (ok) { + *kind = hdr.wal_offset_kind; + } + return ok; +} + uint32_t ParseLeftoverCfId(const std::string& path) { const auto pos = path.rfind(".memtab-"); if (pos == std::string::npos) { @@ -537,8 +559,106 @@ uint32_t ParseLeftoverCfId(const std::string& path) { return static_cast(id); } +void DestroyKindPrepDb(DB* db, std::vector* handles) { + for (auto* h : *handles) { + delete h; + } + handles->clear(); + if (db != nullptr) { + db->Close().PermitUncheckedError(); + delete db; + } +} + } // namespace +Status DBImpl::PrepareCrashSafeKindForOpen( + const DBOptions& db_options, const std::string& dbname, + const std::vector& column_families, + bool seq_per_batch, bool batch_per_txn) { + if (!db_options.memtable_crash_safe_recover) { + return Status::OK(); + } + uint64_t sidecar_kind = 0; + if (!PeekPublishedWalKind(dbname, &sidecar_kind) || sidecar_kind == 0) { + return Status::OK(); + } + const uint64_t opt_kind = db_options.memtable_as_log_index ? 2 : 1; + if (sidecar_kind == opt_kind) { + return Status::OK(); + } + DBOptions cheap = db_options; + cheap.memtable_as_log_index = sidecar_kind == 2; // kLogIndex + cheap.memtable_crash_safe_recover = true; + cheap.max_open_files = 0; + cheap.persist_stats_to_disk = false; + std::vector cheap_cfs = column_families; + for (auto& cf : cheap_cfs) { + cf.options.disable_auto_compactions = true; + } + std::vector handles; + DB* db = nullptr; + Status s = DBImpl::Open(cheap, dbname, cheap_cfs, &handles, &db, + seq_per_batch, batch_per_txn, + true /* options_already_updated */); + if (!s.ok()) { + return s; + } + TEST_SYNC_POINT("DBImpl::Open::KindPrep:AfterOpenBeforeFlush"); + FlushOptions fo; + fo.wait = true; + s = db->Flush(fo, handles); + if (s.ok() && cheap.allow_2pc) { + s = static_cast(db)->RetireWalsForKindPrep(); + } + if (!s.ok()) { + DestroyKindPrepDb(db, &handles); + return s; + } + TEST_SYNC_POINT("DBImpl::Open::KindPrep:AfterFlushBeforeClose"); + for (auto* h : handles) { + delete h; + } + handles.clear(); + s = db->Close(); + delete db; + db = nullptr; + if (!s.ok()) { + return s; + } + TEST_SYNC_POINT("DBImpl::Open::KindPrep:AfterCloseBeforeDeleteSidecar"); + Env* env = db_options.env != nullptr ? db_options.env : Env::Default(); + s = env->DeleteFile(CrashSafePubSeqFileName(dbname)); + if (!s.ok() && !s.IsNotFound()) { + return s; + } + return Status::OK(); +} + +Status DBImpl::RetireWalsForKindPrep() { + InstrumentedMutexLock l(&mutex_); + if (!recovered_transactions_.empty()) { + return Status::NotSupported( + "memtable_as_log_index change with prepared transactions", + "reopen with the previous memtable_as_log_index, commit or roll back " + "them, then switch"); + } + const uint64_t min_log = versions_->MinLogNumberWithUnflushedData(); + if (min_log <= versions_->min_log_number_to_keep()) { + return Status::OK(); + } + VersionEdit edit; + if (immutable_db_options_.track_and_verify_wals_in_manifest) { + edit.DeleteWalsBefore(min_log); + } + edit.SetMinLogNumberToKeep(min_log); + auto* cfd = versions_->GetColumnFamilySet()->GetDefault(); + const ReadOptions read_options(Env::IOActivity::kDBOpen); + return versions_->LogAndApply(cfd, *cfd->GetLatestMutableCFOptions(), + read_options, &edit, &mutex_, + directories_.GetDbDir()); +} + void PublishedSeqMmapHeader::Stamp(bool memtable_as_log_index) { magic = kPublishedSeqMagic; version = kPublishedSeqVersion; @@ -580,6 +700,15 @@ Status DBImpl::MapPublishedSeqFile(uint64_t* probe_wal_below) { pubseq_mmap_->Stamp(immutable_db_options_.memtable_as_log_index); return Status::OK(); } + const uint32_t opt_kind = immutable_db_options_.memtable_as_log_index + ? uint32_t(PublishedWalOffsetKind::kLogIndex) + : uint32_t(PublishedWalOffsetKind::kPhysical); + if (pubseq_mmap_->wal_offset_kind != opt_kind) { + // DB::Open / TransactionDB::Open switch kind before getting here. + UnmapPublishedSeqFile(); + return Status::InvalidArgument( + path, "wal_offset_kind differs from memtable_as_log_index"); + } if (pubseq_mmap_->kind_since_wal != 0) { *probe_wal_below = pubseq_mmap_->kind_since_wal; } @@ -2501,12 +2630,23 @@ Status DB::Open(const Options& options, const std::string& dbname, DB** dbptr) { Status DB::Open(const DBOptions& db_options, const std::string& dbname, const std::vector& column_families, std::vector* handles, DB** dbptr) { + *dbptr = nullptr; + MaybeOptionsUpdateFrom(const_cast(&db_options), + const_cast*>(&column_families), + dbname); const bool kSeqPerBatch = true; const bool kBatchPerTxn = true; ThreadStatusUtil::SetEnableTracking(db_options.enable_thread_tracking); ThreadStatusUtil::SetThreadOperation(ThreadStatus::OperationType::OP_DBOPEN); + Status s0 = DBImpl::PrepareCrashSafeKindForOpen( + db_options, dbname, column_families, !kSeqPerBatch, kBatchPerTxn); + if (!s0.ok()) { + ThreadStatusUtil::ResetThreadStatus(); + return s0; + } Status s = DBImpl::Open(db_options, dbname, column_families, handles, dbptr, - !kSeqPerBatch, kBatchPerTxn); + !kSeqPerBatch, kBatchPerTxn, + true /* options_already_updated */); ThreadStatusUtil::ResetThreadStatus(); return s; } @@ -2640,10 +2780,14 @@ IOStatus DBImpl::CreateWAL(uint64_t log_file_num, uint64_t recycle_log_number, Status DBImpl::Open(const DBOptions& db_options, const std::string& dbname, const std::vector& column_families, std::vector* handles, DB** dbptr, - const bool seq_per_batch, const bool batch_per_txn) { - MaybeOptionsUpdateFrom(const_cast(&db_options), - const_cast*>(&column_families), - dbname); + const bool seq_per_batch, const bool batch_per_txn, + bool options_already_updated) { + // Kind-prep supplies the old WAL kind after applying the external config. + if (!options_already_updated) { + MaybeOptionsUpdateFrom(const_cast(&db_options), + const_cast*>(&column_families), + dbname); + } *dbptr = nullptr; ROCKSDB_SCOPE_EXIT(MaybeRetainDB(*dbptr, *handles)); diff --git a/include/rocksdb/options.h b/include/rocksdb/options.h index 69d316dc0..0b6ffa295 100644 --- a/include/rocksdb/options.h +++ b/include/rocksdb/options.h @@ -946,12 +946,17 @@ struct DBOptions { // are disabled, and wal_compression is forced to kNoCompression. // Read-only Open disables this option and warns that full WAL replay may be // very slow with the much larger MemTables encouraged in crash-safe mode. + // If a valid saved WAL kind differs from memtable_as_log_index, Open first + // recovers and flushes using the old kind, then opens in the requested kind. + // TransactionDB requires resolving recovered prepared transactions before + // switching kinds. // Default: false. Not dynamically changeable through SetDBOptions(). bool memtable_crash_safe_recover = false; // If true, each WAL file is probed on DB open to auto-detect its on-disk - // format, so recovery works even when memtable_as_log_index was changed - // between runs. + // format. This selects the reader format, not the memtable format. Replaying + // a classic WAL directly into log-index memtables returns NotSupported; + // reopen with memtable_as_log_index=false and flush before switching. // // Defaults to false because the probe relies on CRC32 self-consistency // rather than a magic number to distinguish the two formats, which carries diff --git a/utilities/transactions/pessimistic_transaction_db.cc b/utilities/transactions/pessimistic_transaction_db.cc index 682ccd84f..ec6896fb8 100644 --- a/utilities/transactions/pessimistic_transaction_db.cc +++ b/utilities/transactions/pessimistic_transaction_db.cc @@ -306,8 +306,16 @@ Status TransactionDB::Open( const bool use_batch_per_txn = txn_db_options.write_policy == WRITE_COMMITTED || txn_db_options.write_policy == WRITE_PREPARED; + MaybeOptionsUpdateFrom(&db_options_2pc, &column_families_copy, dbname); + s = DBImpl::PrepareCrashSafeKindForOpen(db_options_2pc, dbname, + column_families_copy, + use_seq_per_batch, use_batch_per_txn); + if (!s.ok()) { + return s; + } s = DBImpl::Open(db_options_2pc, dbname, column_families_copy, handles, &db, - use_seq_per_batch, use_batch_per_txn); + use_seq_per_batch, use_batch_per_txn, + true /* options_already_updated */); if (s.ok()) { ROCKS_LOG_WARN(db->GetDBOptions().info_log, "Transaction write_policy is %s", From c91b5e8dbf75c19f26c366bf1c8604e30ae346fc Mon Sep 17 00:00:00 2001 From: leipeng Date: Mon, 28 Sep 2026 00:46:15 +0800 Subject: [PATCH 17/58] DB Recovery: Bench Open after an abnormal exit Fill one memtable to 75% of 2GiB, then _exit without Close. Time only DB::Open on a sparse copy of that directory. Numbers are in tools/crash_recover_bench.md. Crash-safe path, /dev/shm, 5 runs. CSPP 9,664,512 keys. Default SkipList 9,090,048 keys. Both configurations use the same ToplingDB build, not an independently built upstream RocksDB binary. | recovery path | five runs (ms) | avg (ms) | ratio | | --- | --- | ---: | ---: | | CSPP crash-safe | 6.545, 4.692, 3.820, 4.768, 5.120 | 4.989 | 1.0 | | Default SkipList / WAL replay | 6124.357, 5849.884, 5984.324, 5943.391, 6033.802 | 5987.152 | 1200.1 | --- Makefile | 3 + README-zh_cn.md | 4 + README.md | 4 + tools/crash_recover_bench.cc | 281 +++++++++++++++++++++++++++++++++ tools/crash_recover_bench.md | 30 ++++ tools/crash_recover_bench.yaml | 45 ++++++ 6 files changed, 367 insertions(+) create mode 100644 tools/crash_recover_bench.cc create mode 100644 tools/crash_recover_bench.md create mode 100644 tools/crash_recover_bench.yaml diff --git a/Makefile b/Makefile index 462fd9cba..30f7ce819 100644 --- a/Makefile +++ b/Makefile @@ -1889,6 +1889,9 @@ db_bench_rls: $(OBJ_DIR)/tools/db_bench.o $(BENCH_OBJECTS) $(TESTUTIL) $(LIBRARY $(AM_LINK) endif +crash_recover_bench: $(OBJ_DIR)/tools/crash_recover_bench.o $(LIBRARY) + $(AM_LINK) + trace_analyzer: $(OBJ_DIR)/tools/trace_analyzer.o $(ANALYZE_OBJECTS) $(TOOLS_LIBRARY) $(LIBRARY) $(AM_LINK) diff --git a/README-zh_cn.md b/README-zh_cn.md index fc17d6b5a..d96877deb 100644 --- a/README-zh_cn.md +++ b/README-zh_cn.md @@ -37,6 +37,10 @@ ToplingDB 兼容 RocksDB API 的同时,增加了很多非常重要的功能与 1. 内置 Prometheus 指标的支持,这是在[内嵌 Http](https://github.com/topling/rockside/wiki/WebView) 中实现的 1. 修复了很多 RocksDB 的 bug,我们已将其中易于合并到 RocksDB 的很多修复与改进给上游 RocksDB 发了 [Pull Request](https://github.com/facebook/rocksdb/pulls?q=is%3Apr+author%3Arockeet) +## 进程崩溃后的恢复 +恢复机制、配置方式及适用边界见 [MemTable Crash-Safe Recovery](https://github.com/topling/rockside/wiki/Crash-Safe-Recovery)。 +异常退出后 `DB::Open` 的耗时见 [crash_recover_bench.md](tools/crash_recover_bench.md)。 + ## ToplingDB 云原生数据库服务 1. [MyTopling](https://github.com/topling/mytopling)(MySQL on ToplingDB), [阿里云上的 MyTopling](https://market.aliyun.com/products?k=mytopling) 1. [Todis](https://github.com/topling/todis)(Redis on ToplingDB) diff --git a/README.md b/README.md index 639040abb..f322cf97d 100644 --- a/README.md +++ b/README.md @@ -39,6 +39,10 @@ ToplingDB has much more key features than RocksDB: 1. Builtin Prometheus metrics support, this is based on [Embedded Http Server](https://github.com/topling/sideplugin-wiki-en/wiki/WebView) 1. Many bugfixes for RocksDB, a small part of such fixes was [Pull Requested](https://github.com/facebook/rocksdb/pulls?q=is%3Apr+author%3Arockeet) to [upstream RocksDB](https://github.com/facebook/rocksdb) +## Crash-safe recovery +See the [Chinese crash-safe recovery guide](https://github.com/topling/rockside/wiki/Crash-Safe-Recovery) for the recovery mechanism, configuration, and limitations. +`DB::Open` after an abnormal exit is timed in [crash_recover_bench.md](tools/crash_recover_bench.md). + ## ToplingDB cloud native DB services 1. [MyTopling](https://github.com/topling/mytopling)(MySQL on ToplingDB), [MyTopling on aliyun](https://market.aliyun.com/products?k=mytopling) 1. [Todis](https://github.com/topling/todis)(Redis on ToplingDB) diff --git a/tools/crash_recover_bench.cc b/tools/crash_recover_bench.cc new file mode 100644 index 000000000..7da18c40c --- /dev/null +++ b/tools/crash_recover_bench.cc @@ -0,0 +1,281 @@ +// Copyright (c) 2026-present, Topling Inc. +// Abnormal-exit recovery benchmark. +// Same RocksDB API. Topling options come from TOPLINGDB_EASY_MIGRATE_CONF. +// +// Fill one memtable, _exit without Close, copy that directory aside, then +// time DB::Open on a fresh copy of the backup. Repeat and print the average. + +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "rocksdb/db.h" +#include "rocksdb/options.h" + +namespace fs = std::filesystem; +using ROCKSDB_NAMESPACE::DB; +using ROCKSDB_NAMESPACE::Options; +using ROCKSDB_NAMESPACE::ReadOptions; +using ROCKSDB_NAMESPACE::Slice; +using ROCKSDB_NAMESPACE::Status; +using ROCKSDB_NAMESPACE::WriteOptions; + +namespace { + +constexpr size_t kWriteBufferBytes = 2ull << 30; +constexpr int kFillExit = 42; + +struct Args { + std::string root = "/tmp/crash_recover_bench"; + int runs = 5; + int key_size = 16; + int value_size = 128; + // Stop filling once the active memtable reaches this fraction of 2GB, + // so the engine does not flush it before the abnormal exit. + double fill_frac = 0.75; + bool reuse_backup = false; +}; + +void Die(const std::string& msg) { + fprintf(stderr, "crash_recover_bench: %s\n", msg.c_str()); + std::_Exit(1); +} + +Args ParseArgs(int argc, char** argv) { + Args a; + for (int i = 1; i < argc; i++) { + std::string s = argv[i]; + auto need = [&](const char* name) -> std::string { + auto eq = s.find('='); + if (eq == std::string::npos || s.substr(0, eq) != name) { + return ""; + } + return s.substr(eq + 1); + }; + if (auto v = need("--root"); !v.empty()) { + a.root = v; + } else if (auto v = need("--runs"); !v.empty()) { + a.runs = std::atoi(v.c_str()); + } else if (auto v = need("--key_size"); !v.empty()) { + a.key_size = std::atoi(v.c_str()); + } else if (auto v = need("--value_size"); !v.empty()) { + a.value_size = std::atoi(v.c_str()); + } else if (auto v = need("--fill_frac"); !v.empty()) { + a.fill_frac = std::atof(v.c_str()); + } else if (s == "--reuse_backup") { + a.reuse_backup = true; + } else if (s == "--help") { + fprintf(stderr, + "Usage: crash_recover_bench [--root=DIR] [--runs=N]\n" + " [--key_size=N] [--value_size=N] [--fill_frac=0.75]\n" + " [--reuse_backup]\n" + "ToplingDB: TOPLINGDB_EASY_MIGRATE_CONF=tools/crash_recover_bench.yaml\n" + "RocksDB: leave that variable unset\n" + "write_buffer_size is 2GB either way.\n"); + std::exit(0); + } else { + Die("unknown arg " + s); + } + } + if (a.runs < 1 || a.key_size < 1 || a.value_size < 1 || a.fill_frac <= 0 || + a.fill_frac >= 1) { + Die("bad --runs/--key_size/--value_size/--fill_frac"); + } + return a; +} + +Options MakeOptions() { + Options opt; + opt.create_if_missing = true; + opt.write_buffer_size = kWriteBufferBytes; + opt.max_write_buffer_number = 4; + opt.disable_auto_compactions = true; + opt.level0_file_num_compaction_trigger = 1 << 20; + opt.avoid_flush_during_recovery = true; + return opt; +} + +std::string KeyAt(int key_size, uint64_t i) { + std::string k(static_cast(key_size), '0'); + for (int p = key_size - 1; p >= 0 && i > 0; --p) { + k[static_cast(p)] = static_cast('0' + (i % 10)); + i /= 10; + } + return k; +} + +uint64_t ActiveMemBytes(DB* db) { + std::string v; + if (!db->GetProperty("rocksdb.cur-size-active-mem-table", &v)) { + return 0; + } + return std::strtoull(v.c_str(), nullptr, 10); +} + +// Child only. Writes until the live memtable reaches the target, then _exit +// without Close so the directory is a process-crash image. +void FillAndAbort(const Args& args, const fs::path& dbpath, + const fs::path& nkeys_path) { + fs::remove_all(dbpath); + fs::create_directories(dbpath); + DB* db = nullptr; + Status s = DB::Open(MakeOptions(), dbpath.string(), &db); + if (!s.ok()) { + Die("fill open: " + s.ToString()); + } + const uint64_t target = + static_cast(kWriteBufferBytes * args.fill_frac); + WriteOptions wo; + wo.disableWAL = false; + std::string value(static_cast(args.value_size), 'v'); + uint64_t n = 0; + uint64_t mem = 0; + while (mem < target) { + std::string key = KeyAt(args.key_size, n); + s = db->Put(wo, key, value); + if (!s.ok()) { + Die("put: " + s.ToString()); + } + n++; + if ((n & 1023) == 0) { + mem = ActiveMemBytes(db); + } + } + fprintf(stderr, "filled keys=%llu active_mem=%llu target=%llu\n", + static_cast(n), + static_cast(mem), + static_cast(target)); + FILE* nf = fopen(nkeys_path.c_str(), "w"); + if (nf == nullptr || fprintf(nf, "%llu\n", static_cast(n)) < 0) { + Die("write nkeys"); + } + fclose(nf); + // Leave the DB open. _exit skips destructors, same as kill -9. + std::_Exit(kFillExit); +} + +void CopyTree(const fs::path& src, const fs::path& dst) { + std::error_code ec; + fs::remove_all(dst, ec); + fs::create_directories(dst.parent_path(), ec); + if (ec) { + Die("mkdir " + dst.parent_path().string() + ": " + ec.message()); + } + // Keep holes. A plain copy fills the memtab tail and makes ftruncate drop + // real pages. + const pid_t child = ::fork(); + if (child < 0) { + Die("fork cp: " + std::string(std::strerror(errno))); + } + if (child == 0) { + ::execlp("cp", "cp", "-a", "--sparse=always", "--", src.c_str(), + dst.c_str(), static_cast(nullptr)); + std::_Exit(127); + } + int status = 0; + pid_t waited; + do { + waited = ::waitpid(child, &status, 0); + } while (waited < 0 && errno == EINTR); + if (waited < 0 || !WIFEXITED(status) || WEXITSTATUS(status) != 0) { + Die("cp --sparse=always " + src.string() + " -> " + dst.string()); + } +} + +double OpenMillis(const fs::path& dbpath, const Args& args, uint64_t nkeys) { + DB* db = nullptr; + auto t0 = std::chrono::steady_clock::now(); + Status s = DB::Open(MakeOptions(), dbpath.string(), &db); + auto t1 = std::chrono::steady_clock::now(); + if (!s.ok()) { + Die("recover open: " + s.ToString()); + } + std::string got; + s = db->Get(ReadOptions(), KeyAt(args.key_size, 0), &got); + if (!s.ok() || got.size() != static_cast(args.value_size)) { + Die("spot check key 0: " + s.ToString()); + } + if (nkeys > 1) { + s = db->Get(ReadOptions(), KeyAt(args.key_size, nkeys - 1), &got); + if (!s.ok()) { + Die("spot check last key: " + s.ToString()); + } + } + s = db->Close(); + delete db; + if (!s.ok()) { + Die("close: " + s.ToString()); + } + return std::chrono::duration(t1 - t0).count(); +} + +uint64_t ReadNKeys(const fs::path& path) { + FILE* nf = fopen(path.c_str(), "r"); + if (nf == nullptr) { + Die("open " + path.string()); + } + unsigned long long n = 0; + if (fscanf(nf, "%llu", &n) != 1 || n == 0) { + fclose(nf); + Die("bad nkeys file"); + } + fclose(nf); + return n; +} + +} // namespace + +int main(int argc, char** argv) { + setenv("ROCKSDB_KICK_OUT_OPTIONS_FILE", "1", 1); + Args args = ParseArgs(argc, argv); + const fs::path root = args.root; + const fs::path crashed = root / "crashed"; + const fs::path backup = root / "backup"; + const fs::path nkeys_path = root / "nkeys.txt"; + fs::create_directories(root); + + const bool have_backup = args.reuse_backup && fs::exists(backup / "CURRENT"); + if (!have_backup) { + const pid_t pid = ::fork(); + if (pid < 0) { + Die("fork"); + } + if (pid == 0) { + FillAndAbort(args, crashed, nkeys_path); + } + int st = 0; + if (::waitpid(pid, &st, 0) != pid || !WIFEXITED(st) || + WEXITSTATUS(st) != kFillExit) { + Die("fill child failed"); + } + CopyTree(crashed, backup); + fprintf(stderr, "backup %s\n", backup.c_str()); + } + + const uint64_t nkeys = ReadNKeys(nkeys_path); + fprintf(stderr, "keys=%llu runs=%d write_buffer=%zuMB\n", + static_cast(nkeys), args.runs, + kWriteBufferBytes >> 20); + + double sum = 0; + for (int i = 0; i < args.runs; i++) { + const fs::path run = root / ("run-" + std::to_string(i)); + CopyTree(backup, run); + const double ms = OpenMillis(run, args, nkeys); + sum += ms; + printf("run %d open_ms %.3f\n", i, ms); + fflush(stdout); + fs::remove_all(run); + } + printf("avg_open_ms %.3f runs %d\n", sum / args.runs, args.runs); + return 0; +} diff --git a/tools/crash_recover_bench.md b/tools/crash_recover_bench.md new file mode 100644 index 000000000..3c89ec190 --- /dev/null +++ b/tools/crash_recover_bench.md @@ -0,0 +1,30 @@ +# Abnormal-exit recovery: DB::Open + +For the recovery mechanism, configuration, and limitations, see the [Chinese crash-safe recovery guide](https://github.com/topling/rockside/wiki/Crash-Safe-Recovery). + +`crash_recover_bench` fills one memtable to 75% of a 2GiB (2,147,483,648-byte) +`write_buffer_size`, targeting 1.5GiB (1,610,612,736 bytes), then `_exit`s +without `Close`. Keys are 16 bytes and values are 128 bytes. Each run copies that +directory with `cp -a --sparse=always` and times only `DB::Open`. A Get of +key 0 and the last key runs after the timer. Default is 5 runs. + +Measured on 2026-10-02, directories under `/dev/shm`. + +CSPP crash-safe sets `TOPLINGDB_EASY_MIGRATE_CONF=tools/crash_recover_bench.yaml`. +Default SkipList / WAL replay leaves that variable unset, so it does not read +the yaml. Both configurations use the same benchmark binary and the same +ToplingDB build (`librocksdb.so.8.10.2`); the control is not an independently +built upstream RocksDB binary. +`write_buffer_size` is 2GiB either way. CSPP reached 9,664,512 keys and +1,610,644,224 active-memtable bytes; default SkipList reached 9,090,048 keys +and 1,610,614,784 bytes. The active memtable stops at the same 75% +byte target; the key counts differ. + +| recovery path | five runs (ms) | avg (ms) | ratio | +| --- | --- | ---: | ---: | +| CSPP crash-safe | 6.545, 4.692, 3.820, 4.768, 5.120 | 4.989 | 1.0 | +| Default SkipList / WAL replay | 6124.357, 5849.884, 5984.324, 5943.391, 6033.802 | 5987.152 | 1200.1 | + +Both groups exited with status 0. The copied images contained no SST and had +not flushed early. Separate LOG checks confirmed CSPP leftover conversion +and WAL-tail recovery without fallback, and full WAL recovery for default SkipList. diff --git a/tools/crash_recover_bench.yaml b/tools/crash_recover_bench.yaml new file mode 100644 index 000000000..9a542ac2d --- /dev/null +++ b/tools/crash_recover_bench.yaml @@ -0,0 +1,45 @@ +# Easy-migrate config for crash_recover_bench. +# Point TOPLINGDB_EASY_MIGRATE_CONF at this file. Unset it for plain RocksDB. +http: + auto_start_http: false +setenv: + ROCKSDB_KICK_OUT_OPTIONS_FILE: + overwrite: true + value: "1" + TOPLINGDB_WARMUP_PROVIDER: + overwrite: true + value: "willneed" +MemTableRepFactory: + cspp: + class: cspp + params: + mem_cap: 1G + convert_to_sst: kFileMmap + sync_sst_file: false +TableFactory: + cspp_memtab_sst: + class: CSPPMemTabTable + params: + populate_read: false + dispatch: + class: DispatcherTable + params: + default: cspp_memtab_sst + readers: + CSPPMemTabTable: cspp_memtab_sst +CFOptions: + default: + write_buffer_size: 2G + max_write_buffer_number: 4 + memtable_factory: "${cspp}" + table_factory: dispatch + disable_auto_compactions: true + level0_file_num_compaction_trigger: 1048576 +DBOptions: + default: + create_if_missing: true + memtable_crash_safe_recover: true + memtable_as_log_index: true + avoid_flush_during_recovery: true + allow_fdatasync: false + max_background_jobs: 2 From 57d2443026318b303890cb824b90ff6ee360179f Mon Sep 17 00:00:00 2001 From: leipeng Date: Mon, 28 Sep 2026 01:43:38 +0800 Subject: [PATCH 18/58] DB Recovery: Use volatile AVX stores for pubseq publication Crash-safe publication relies on writing the 32-byte record containing pubseq in one instruction. A non-volatile AVX intrinsic does not preserve that requirement: Clang can split it into two 16-byte stores, leaving a mixed record if the process crashes between them. Clang/LLVM specifies that the backend must not split or merge target-legal volatile loads/stores. AVX provides a native 32-byte store, so volatile preserves the required instruction width and allows Clang to use the same publication path as GCC instead of the odd/even fallback. https://llvm.org/docs/LangRef.html#volatile-memory-accesses --- db/db_cspp_crash_safe_test.cc | 17 ++++++++--------- db/db_impl/db_impl_open.cc | 13 ++++++------- 2 files changed, 14 insertions(+), 16 deletions(-) diff --git a/db/db_cspp_crash_safe_test.cc b/db/db_cspp_crash_safe_test.cc index b36a1a7f4..aa0268ea4 100644 --- a/db/db_cspp_crash_safe_test.cc +++ b/db/db_cspp_crash_safe_test.cc @@ -1320,7 +1320,7 @@ TEST_F(DBCsppCrashSafeTest, OddGenerationPublishesEvenAfterWalRecovery) { ASSERT_TRUE(SetPublishedSeqGeneration(dbname_, rec.generation | 1)); ASSERT_OK(TryReopen(options)); ASSERT_EQ(Get("before"), "recovery"); -#if !defined(__AVX__) || defined(__clang__) +#if !defined(__AVX__) int odd = 0; SyncPoint::GetInstance()->SetCallBack( "DBImpl::PersistPublishedSequence:AfterOddGeneration", [&](void*) { @@ -1332,7 +1332,7 @@ TEST_F(DBCsppCrashSafeTest, OddGenerationPublishesEvenAfterWalRecovery) { SyncPoint::GetInstance()->EnableProcessing(); #endif ASSERT_OK(Put("after", "recovery")); -#if !defined(__AVX__) || defined(__clang__) +#if !defined(__AVX__) SyncPoint::GetInstance()->DisableProcessing(); SyncPoint::GetInstance()->ClearAllCallBacks(); ASSERT_EQ(odd, 1); @@ -1426,9 +1426,8 @@ TEST_F(DBCsppCrashSafeTest, AfterCommitExitConvertsAndReopens) { ASSERT_TRUE(metadata.levels[0].files[0].marked_for_compaction); } -// Clang is excluded from the AVX store and still publishes through the odd -// generation, so it runs the same crash-injection tests as a non-AVX build. -#if !defined(__AVX__) || defined(__clang__) +// Non-AVX builds publish through the odd generation. +#if !defined(__AVX__) TEST_F(CrashChild, DISABLED_AfterOddGenerationFallsBackToWal) { Options options = BaseCrashSafeOptions(dbname_, true, false); SyncPoint::GetInstance()->SetCallBack( @@ -1449,7 +1448,7 @@ TEST_F(DBCsppCrashSafeTest, AfterOddGenerationFallsBackToWal) { ASSERT_OK(TryReopen(options)); ASSERT_EQ(Get("odd"), "wal"); } -#endif // !__AVX__ || __clang__ +#endif // !__AVX__ TEST_F(CrashChild, DISABLED_AfterWriteToWALBeforePublishKeepsWalTail) { Options options = BaseCrashSafeOptions(dbname_, true, false); @@ -3462,8 +3461,8 @@ TEST_F(DBCsppCrashSafeTest, LeftoverLogRefUnbindFallsBackToWal) { ASSERT_EQ(Get("rb"), "ok"); } -// Clang is excluded from the AVX store and still hits AfterOddGeneration. -#if !defined(__AVX__) || defined(__clang__) +// Non-AVX builds hit AfterOddGeneration. +#if !defined(__AVX__) TEST_F(DBCsppCrashSafeTest, CloseWaitsForStatsPublication) { Close(); Options options = BaseCrashSafeOptions(dbname_, true, false); @@ -3492,7 +3491,7 @@ TEST_F(DBCsppCrashSafeTest, CloseWaitsForStatsPublication) { SyncPoint::GetInstance()->LoadDependency({}); Close(); } -#endif // !__AVX__ || __clang__ +#endif // !__AVX__ TEST_F(DBCsppCrashSafeTest, InFlightFlushThenClose) { Close(); diff --git a/db/db_impl/db_impl_open.cc b/db/db_impl/db_impl_open.cc index 79de49c20..5665a55ae 100644 --- a/db/db_impl/db_impl_open.cc +++ b/db/db_impl/db_impl_open.cc @@ -11,9 +11,7 @@ #include #include #endif -// Clang splits _mm256_store_si256 into two vmovdqa xmm stores, which tears -// the record across a process crash. Keep Clang on the non-AVX path. -#if defined(__AVX__) && !defined(__clang__) +#if defined(__AVX__) #include #endif @@ -749,9 +747,7 @@ void DBImpl::PersistPublishedSequence(SequenceNumber seq, uint64_t wal_number, } // Recovery may leave an odd generation until the first new publication. const uint64_t g = pubseq_mmap_->generation | 1; - // Clang lowers _mm256_store_si256 to two vmovdqa xmm stores. A crash - // between them publishes a mixed record. GCC emits one vmovdqa ymm. -#if defined(__AVX__) && !defined(__clang__) +#if defined(__AVX__) // One aligned 32-byte store. A process crash falls between instructions, // so the record is all old or all new. Publish an even generation only // after this store, including when recovery left it odd. A host without it @@ -764,7 +760,10 @@ void DBImpl::PersistPublishedSequence(SequenceNumber seq, uint64_t wal_number, // _mm256_store_si256((__m256i*)rec, _mm256_load_si256((const __m256i*)&next)); // Lowest 64 bits are pubseq, matching PublishedSeqRecord field order. __m256i packed = _mm256_set_epi64x(0, wal_offset, wal_number, seq); - _mm256_store_si256((__m256i*)rec, packed); + // Clang/LLVM specifies that the backend must not split or merge target-legal + // volatile loads/stores. This keeps the AVX store a single 32-byte instruction. + // https://llvm.org/docs/LangRef.html#volatile-memory-accesses + *(volatile __m256i*)rec = packed; #else // Keep generation odd throughout the field stores, even if already odd. // Acquire keeps the field stores after this exchange; the release below From eca9a370af5959234ea8fb574a39cca6adf57f93 Mon Sep 17 00:00:00 2001 From: leipeng Date: Fri, 2 Oct 2026 21:21:56 +0800 Subject: [PATCH 19/58] docs: link crash-safe recovery and data-structure guides The recovery guide explains database behavior, while the companion article explains why wait-free read structures can remain readable after a process crash. Make both discoverable from the main repository, with English readers directed to the new English editions. --- README-zh_cn.md | 1 + README.md | 3 ++- 2 files changed, 3 insertions(+), 1 deletion(-) diff --git a/README-zh_cn.md b/README-zh_cn.md index d96877deb..699f6a130 100644 --- a/README-zh_cn.md +++ b/README-zh_cn.md @@ -39,6 +39,7 @@ ToplingDB 兼容 RocksDB API 的同时,增加了很多非常重要的功能与 ## 进程崩溃后的恢复 恢复机制、配置方式及适用边界见 [MemTable Crash-Safe Recovery](https://github.com/topling/rockside/wiki/Crash-Safe-Recovery)。 +底层数据结构为何同时支持读侧无等待与进程崩溃后的读取,见 [读侧无等待与 Crash-Safe 的同构性](https://github.com/topling/rockside/wiki/Wait-Free-Reads-and-Crash-Safe)。 异常退出后 `DB::Open` 的耗时见 [crash_recover_bench.md](tools/crash_recover_bench.md)。 ## ToplingDB 云原生数据库服务 diff --git a/README.md b/README.md index f322cf97d..ea287b94c 100644 --- a/README.md +++ b/README.md @@ -40,7 +40,8 @@ ToplingDB has much more key features than RocksDB: 1. Many bugfixes for RocksDB, a small part of such fixes was [Pull Requested](https://github.com/facebook/rocksdb/pulls?q=is%3Apr+author%3Arockeet) to [upstream RocksDB](https://github.com/facebook/rocksdb) ## Crash-safe recovery -See the [Chinese crash-safe recovery guide](https://github.com/topling/rockside/wiki/Crash-Safe-Recovery) for the recovery mechanism, configuration, and limitations. +See the [crash-safe recovery guide](https://github.com/topling/sideplugin-wiki-en/wiki/Crash-Safe-Recovery) for the recovery mechanism, configuration, and limitations. +For the underlying data-structure principles, see [The Isomorphism Between Wait-Free Reads and Crash Safety](https://github.com/topling/sideplugin-wiki-en/wiki/Wait-Free-Reads-and-Crash-Safe). `DB::Open` after an abnormal exit is timed in [crash_recover_bench.md](tools/crash_recover_bench.md). ## ToplingDB cloud native DB services From e1e7809c096a74a177c6bf3c6872a07dcdf242d6 Mon Sep 17 00:00:00 2001 From: leipeng Date: Sat, 3 Oct 2026 20:54:16 +0800 Subject: [PATCH 20/58] Track recoverable MemTable files in the MANIFEST Finding leftover files only tells recovery which files still exist, not whether any required file has disappeared. Fast recovery needs an authoritative record of the complete set before it can safely skip WAL replay. That record must survive MANIFEST rotation and atomic edits, including the transition to an empty set. An empty tracked set is not the same as an old MANIFEST that never tracked MemTables. Required tags make older binaries reject this state instead of silently ignoring it. Backing files also occupy DB file numbers and need the same committed ownership rules as SSTs: retirement must not delete a file that an atomic conversion has just installed under the same number. --- db/version_edit.cc | 54 ++++++++++ db/version_edit.h | 22 +++- db/version_edit_handler.cc | 14 ++- db/version_edit_test.cc | 50 +++++++++ db/version_set.cc | 78 +++++++++++++- db/version_set.h | 10 ++ db/version_set_test.cc | 204 +++++++++++++++++++++++++++++++++++++ 7 files changed, 429 insertions(+), 3 deletions(-) diff --git a/db/version_edit.cc b/db/version_edit.cc index b0af0d87a..29d9c8090 100644 --- a/db/version_edit.cc +++ b/db/version_edit.cc @@ -86,6 +86,9 @@ void VersionEdit::Clear() { wal_additions_.clear(); wal_deletion_.Reset(); column_family_ = 0; + memtable_file_additions_.clear(); + memtable_file_deletions_.clear(); + has_memtable_file_tracking_ = false; is_column_family_add_ = false; is_column_family_drop_ = false; column_family_name_.clear(); @@ -288,6 +291,17 @@ bool VersionEdit::EncodeTo(std::string* dst, } // 0 is default and does not need to be explicitly written + if (has_memtable_file_tracking_) { + PutVarint32(dst, kMemTableFileTracking); + } + for (uint64_t number : memtable_file_additions_) { + if (number == 0 || number > kFileNumberMask) return false; + PutVarint32Varint64(dst, kMemTableFileAddition, number); + } + for (uint64_t number : memtable_file_deletions_) { + if (number == 0 || number > kFileNumberMask) return false; + PutVarint32Varint64(dst, kMemTableFileDeletion, number); + } if (column_family_ != 0) { PutVarint32Varint32(dst, kColumnFamily, column_family_); } @@ -768,6 +782,24 @@ Status VersionEdit::DecodeFrom(const Slice& src) { is_column_family_drop_ = true; break; + case kMemTableFileTracking: + has_memtable_file_tracking_ = true; + break; + + case kMemTableFileAddition: + case kMemTableFileDeletion: { + uint64_t number; + if (!GetVarint64(&input, &number) || number == 0 || + number > kFileNumberMask) { + msg = "invalid MemTable file number"; + } else if (tag == kMemTableFileAddition) { + memtable_file_additions_.insert(number); + } else { + memtable_file_deletions_.insert(number); + } + break; + } + case kInAtomicGroup: is_in_atomic_group_ = true; if (!GetVarint32(&input, &remaining_entries_)) { @@ -948,6 +980,15 @@ std::string VersionEdit::DebugString(bool hex_key) const { r.append("\n ColumnFamily: "); AppendNumberTo(&r, column_family_); + if (has_memtable_file_tracking_) r.append("\n MemTableFileTracking: true"); + for (uint64_t number : memtable_file_additions_) { + r.append("\n AddMemTableFile: "); + AppendNumberTo(&r, number); + } + for (uint64_t number : memtable_file_deletions_) { + r.append("\n DeleteMemTableFile: "); + AppendNumberTo(&r, number); + } if (is_column_family_add_) { r.append("\n ColumnFamilyAdd: "); r.append(column_family_name_); @@ -1098,6 +1139,19 @@ std::string VersionEdit::DebugJSON(int edit_num, bool hex_key) const { } jw << "ColumnFamily" << column_family_; + if (has_memtable_file_tracking_) jw << "MemTableFileTracking" << true; + if (!memtable_file_additions_.empty()) { + jw << "MemTableFileAdditions"; + jw.StartArray(); + for (uint64_t number : memtable_file_additions_) jw << number; + jw.EndArray(); + } + if (!memtable_file_deletions_.empty()) { + jw << "MemTableFileDeletions"; + jw.StartArray(); + for (uint64_t number : memtable_file_deletions_) jw << number; + jw.EndArray(); + } if (is_column_family_add_) { jw << "ColumnFamilyAdd" << column_family_name_; diff --git a/db/version_edit.h b/db/version_edit.h index ce35758d4..1df20c5f1 100644 --- a/db/version_edit.h +++ b/db/version_edit.h @@ -63,6 +63,11 @@ enum Tag : uint32_t { kBlobFileAddition = 400, kBlobFileGarbage, + // Required recovery metadata: older readers must reject these tags. + kMemTableFileAddition = 500, + kMemTableFileDeletion = 501, + kMemTableFileTracking = 502, + // Mask for an unidentified tag from the future which can be safely ignored. kTagSafeIgnoreMask = 1 << 13, @@ -652,7 +657,8 @@ class VersionEdit { size_t NumEntries() const { return new_files_.size() + deleted_files_.size() + blob_file_additions_.size() + blob_file_garbages_.size() + - wal_additions_.size() + !wal_deletion_.IsEmpty(); + wal_additions_.size() + !wal_deletion_.IsEmpty() + + memtable_file_additions_.size() + memtable_file_deletions_.size(); } void SetColumnFamily(uint32_t column_family_id) { @@ -660,6 +666,17 @@ class VersionEdit { } uint32_t GetColumnFamily() const { return column_family_; } + void AddMemTableFile(uint64_t number) { memtable_file_additions_.insert(number); } + void DeleteMemTableFile(uint64_t number) { memtable_file_deletions_.insert(number); } + const std::set& GetMemTableFileAdditions() const { + return memtable_file_additions_; + } + const std::set& GetMemTableFileDeletions() const { + return memtable_file_deletions_; + } + void SetMemTableFileTracking() { has_memtable_file_tracking_ = true; } + bool HasMemTableFileTracking() const { return has_memtable_file_tracking_; } + const std::string& GetColumnFamilyName() const { return column_family_name_; } // set column family ID by calling SetColumnFamily() @@ -774,6 +791,9 @@ class VersionEdit { // Each version edit record should have column_family_ set // If it's not set, it is default (0) uint32_t column_family_ = 0; + std::set memtable_file_additions_; + std::set memtable_file_deletions_; + bool has_memtable_file_tracking_ = false; // a version edit can be either column_family add or // column_family drop. If it's column family add, // it also includes column family name. diff --git a/db/version_edit_handler.cc b/db/version_edit_handler.cc index 395d066d7..fb32f4db4 100644 --- a/db/version_edit_handler.cc +++ b/db/version_edit_handler.cc @@ -213,6 +213,7 @@ Status VersionEditHandler::ApplyVersionEdit(VersionEdit& edit, if (s.ok()) { assert(cfd != nullptr); s = ExtractInfoFromVersionEdit(*cfd, edit); + if (s.ok()) version_set_->ApplyMemTableFileEdit(edit); } return s; } @@ -640,7 +641,18 @@ Status VersionEditHandler::ExtractInfoFromVersionEdit(ColumnFamilyData* cfd, version_edit_params_.SetPrevLogNumber(edit.GetPrevLogNumber()); } if (edit.HasNextFile()) { - version_edit_params_.SetNextFile(edit.GetNextFile()); + version_edit_params_.SetNextFile( + std::max(version_edit_params_.GetNextFile(), edit.GetNextFile())); + } + uint64_t next_file = version_edit_params_.GetNextFile(); + for (uint64_t number : edit.GetMemTableFileAdditions()) { + next_file = std::max(next_file, number + 1); + } + for (uint64_t number : edit.GetMemTableFileDeletions()) { + next_file = std::max(next_file, number + 1); + } + if (next_file != version_edit_params_.GetNextFile()) { + version_edit_params_.SetNextFile(next_file); } if (edit.HasMaxColumnFamily()) { version_edit_params_.SetMaxColumnFamily(edit.GetMaxColumnFamily()); diff --git a/db/version_edit_test.cc b/db/version_edit_test.cc index 252352069..8767ef49e 100644 --- a/db/version_edit_test.cc +++ b/db/version_edit_test.cc @@ -40,6 +40,56 @@ static void TestEncodeDecode(const VersionEdit& edit) { class VersionEditTest : public testing::Test {}; +TEST_F(VersionEditTest, MemTableFilesEncodeDecodeAndClear) { + VersionEdit edit; + edit.SetColumnFamily(7); + edit.SetMemTableFileTracking(); + edit.AddMemTableFile(123); + edit.AddMemTableFile(kFileNumberMask); + edit.DeleteMemTableFile(42); + edit.MarkAtomicGroup(0); + std::string encoded; + ASSERT_TRUE(edit.EncodeTo(&encoded)); + VersionEdit parsed; + ASSERT_OK(parsed.DecodeFrom(encoded)); + ASSERT_EQ(parsed.GetColumnFamily(), 7U); + ASSERT_TRUE(parsed.HasMemTableFileTracking()); + ASSERT_EQ(parsed.GetMemTableFileAdditions(), edit.GetMemTableFileAdditions()); + ASSERT_EQ(parsed.GetMemTableFileDeletions(), edit.GetMemTableFileDeletions()); + ASSERT_TRUE(parsed.IsInAtomicGroup()); + ASSERT_EQ(parsed.GetRemainingEntries(), 0U); + parsed.Clear(); + ASSERT_FALSE(parsed.HasMemTableFileTracking()); + ASSERT_TRUE(parsed.GetMemTableFileAdditions().empty()); + ASSERT_TRUE(parsed.GetMemTableFileDeletions().empty()); + ASSERT_FALSE(parsed.IsInAtomicGroup()); +} + +TEST_F(VersionEditTest, MemTableFilesRejectInvalidNumbers) { + for (uint32_t tag : {uint32_t(kMemTableFileAddition), + uint32_t(kMemTableFileDeletion)}) { + for (uint64_t number : {uint64_t(0), kFileNumberMask + 1}) { + std::string encoded; + PutVarint32Varint64(&encoded, tag, number); + VersionEdit parsed; + ASSERT_TRUE(parsed.DecodeFrom(encoded).IsCorruption()); + VersionEdit invalid; + if (tag == kMemTableFileAddition) invalid.AddMemTableFile(number); + else invalid.DeleteMemTableFile(number); + encoded.clear(); + ASSERT_FALSE(invalid.EncodeTo(&encoded)); + } + std::string truncated; + PutVarint32(&truncated, tag); + truncated.push_back(char(0x80)); + VersionEdit parsed; + ASSERT_TRUE(parsed.DecodeFrom(truncated).IsCorruption()); + } + ASSERT_EQ(kMemTableFileAddition & kTagSafeIgnoreMask, 0U); + ASSERT_EQ(kMemTableFileDeletion & kTagSafeIgnoreMask, 0U); + ASSERT_EQ(kMemTableFileTracking & kTagSafeIgnoreMask, 0U); +} + TEST_F(VersionEditTest, EncodeDecode) { static const uint64_t kBig = 1ull << 50; static const uint32_t kBig32Bit = 1ull << 30; diff --git a/db/version_set.cc b/db/version_set.cc index 573c57e97..ad13602ab 100644 --- a/db/version_set.cc +++ b/db/version_set.cc @@ -5505,6 +5505,8 @@ void VersionSet::Reset() { obsolete_files_.clear(); obsolete_manifests_.clear(); wals_.Reset(); + memtable_files_.clear(); + has_memtable_file_tracking_ = false; } void VersionSet::AppendVersion(ColumnFamilyData* column_family_data, @@ -5743,6 +5745,7 @@ Status VersionSet::ProcessManifestWrites( std::unordered_map curr_state; VersionEdit wal_additions; if (new_descriptor_log) { + if (has_memtable_file_tracking_) wal_additions.SetMemTableFileTracking(); pending_manifest_file_number_ = NewFileNumber(); batch_edits.back()->SetNextFile(next_file_number_.load()); @@ -5757,6 +5760,10 @@ Status VersionSet::ProcessManifestWrites( curr_state.emplace( cfd->GetID(), MutableCFState(cfd->GetLogNumber(), cfd->GetFullHistoryTsLow())); + auto memtable_iter = memtable_files_.find(cfd->GetID()); + if (memtable_iter != memtable_files_.end()) { + curr_state.at(cfd->GetID()).memtable_files = memtable_iter->second; + } } for (const auto& wal : wals_.GetWals()) { @@ -5956,6 +5963,45 @@ Status VersionSet::ProcessManifestWrites( // Install the new versions if (s.ok()) { + // Only a committed live batch can retire backing files. MANIFEST replay + // applies registry edits without scheduling historical deletions. + std::map retired_memtable_files; + for (const auto* edit : batch_edits) { + auto files = memtable_files_.find(edit->GetColumnFamily()); + if (files != memtable_files_.end()) { + auto* cfd = column_family_set_->GetColumnFamily(edit->GetColumnFamily()); + assert(cfd != nullptr); + const auto& path = cfd->ioptions()->cf_paths[0].path; + if (edit->IsColumnFamilyDrop()) { + for (uint64_t number : files->second) { + retired_memtable_files.emplace(number, path); + } + } else { + for (uint64_t number : edit->GetMemTableFileDeletions()) { + if (files->second.count(number)) { + retired_memtable_files.emplace(number, path); + } + } + } + } + ApplyMemTableFileEdit(*edit); + } + if (!retired_memtable_files.empty()) { + // In-place conversion transfers ownership to the installed SST version. + for (const auto* edit : batch_edits) { + for (const auto& file : edit->GetNewFiles()) { + retired_memtable_files.erase(file.second.fd.GetNumber()); + } + } + for (const auto& cf : memtable_files_) { + for (uint64_t number : cf.second) retired_memtable_files.erase(number); + } + for (const auto& file : retired_memtable_files) { + auto* metadata = new FileMetaData(); + metadata->fd = FileDescriptor(file.first, 0, 0); + obsolete_files_.emplace_back(metadata, file.second); + } + } if (first_writer.edit_list.front()->IsColumnFamilyAdd()) { assert(batch_edits.size() == 1); assert(new_cf_options != nullptr); @@ -6850,7 +6896,8 @@ Status VersionSet::WriteCurrentStateToManifest( } // Save WALs. - if (!wal_additions.GetWalAdditions().empty()) { + if (!wal_additions.GetWalAdditions().empty() || + wal_additions.HasMemTableFileTracking()) { TEST_SYNC_POINT_CALLBACK("VersionSet::WriteCurrentStateToManifest:SaveWal", const_cast(&wal_additions)); std::string record; @@ -6916,6 +6963,10 @@ Status VersionSet::WriteCurrentStateToManifest( VersionEdit edit; edit.SetColumnFamily(cfd->GetID()); + for (uint64_t number : curr_state.at(cfd->GetID()).memtable_files) { + edit.AddMemTableFile(number); + } + const auto* current = cfd->current(); assert(current); @@ -7568,6 +7619,31 @@ uint64_t VersionSet::GetObsoleteSstFilesSize() const { return ret; } +void VersionSet::ApplyMemTableFileEdit(const VersionEdit& edit) { + has_memtable_file_tracking_ |= edit.HasMemTableFileTracking(); + // Live allocation can race with MANIFEST installation; never move the + // atomic allocator backwards when reserving recovered/persisted numbers. + auto reserve_number = [this](uint64_t number) { + uint64_t next = next_file_number_.load(std::memory_order_relaxed); + while (next <= number && + !next_file_number_.compare_exchange_weak( + next, number + 1, std::memory_order_relaxed)) {} + }; + for (uint64_t number : edit.GetMemTableFileAdditions()) reserve_number(number); + for (uint64_t number : edit.GetMemTableFileDeletions()) reserve_number(number); + if (edit.IsColumnFamilyDrop()) { + memtable_files_.erase(edit.GetColumnFamily()); + return; + } + for (uint64_t number : edit.GetMemTableFileDeletions()) { + auto iter = memtable_files_.find(edit.GetColumnFamily()); + if (iter != memtable_files_.end()) iter->second.erase(number); + } + for (uint64_t number : edit.GetMemTableFileAdditions()) { + memtable_files_[edit.GetColumnFamily()].insert(number); + } +} + ColumnFamilyData* VersionSet::CreateColumnFamily( const ColumnFamilyOptions& cf_options, const ReadOptions& read_options, const VersionEdit* edit) { diff --git a/db/version_set.h b/db/version_set.h index 2c4b3bf7e..987258d1c 100644 --- a/db/version_set.h +++ b/db/version_set.h @@ -1370,6 +1370,13 @@ class VersionSet { // Allocate and return a new file number uint64_t NewFileNumber() { return next_file_number_.fetch_add(1); } + // Access requires the DB mutex, like other MANIFEST state. + const std::map>& GetMemTableFiles() const { + return memtable_files_; + } + bool HasMemTableFileTracking() const { return has_memtable_file_tracking_; } + void ApplyMemTableFileEdit(const VersionEdit& edit); + // Fetch And Add n new file number uint64_t FetchAddFileNumber(uint64_t n) { return next_file_number_.fetch_add(n); @@ -1655,6 +1662,7 @@ class VersionSet { struct MutableCFState { uint64_t log_number; std::string full_history_ts_low; + std::set memtable_files; explicit MutableCFState() = default; explicit MutableCFState(uint64_t _log_number, std::string ts_low) @@ -1678,6 +1686,8 @@ class VersionSet { // Protected by DB mutex. WalSet wals_; + std::map> memtable_files_; + bool has_memtable_file_tracking_ = false; std::unique_ptr column_family_set_; Cache* table_cache_; diff --git a/db/version_set_test.cc b/db/version_set_test.cc index 4703b1f44..f329ccb31 100644 --- a/db/version_set_test.cc +++ b/db/version_set_test.cc @@ -1505,6 +1505,139 @@ TEST_F(VersionSetTest, SameColumnFamilyGroupCommit) { EXPECT_EQ(kGroupSize - 1, count); } +TEST_F(VersionSetTest, MemTableRegistryManifestRollover) { + NewDB(); + ASSERT_FALSE(versions_->HasMemTableFileTracking()); + VersionEdit addition; + addition.SetMemTableFileTracking(); + addition.AddMemTableFile(10000); + addition.AddMemTableFile(10001); + ASSERT_OK(LogAndApplyToDefaultCF(addition)); + ASSERT_EQ(versions_->GetMemTableFiles().at(0), + std::set({10000, 10001})); + CreateNewManifest(); + ReopenDB(); + ASSERT_TRUE(versions_->HasMemTableFileTracking()); + ASSERT_EQ(versions_->GetMemTableFiles().at(0), + std::set({10000, 10001})); + ASSERT_GT(versions_->current_next_file_number(), 10001U); + VersionEdit deletion; + deletion.DeleteMemTableFile(10000); + deletion.DeleteMemTableFile(10001); + ASSERT_OK(LogAndApplyToDefaultCF(deletion)); + CreateNewManifest(); + ReopenDB(); + ASSERT_TRUE(versions_->HasMemTableFileTracking()); + for (const auto& entry : versions_->GetMemTableFiles()) { + ASSERT_TRUE(entry.second.empty()); + } +} + +TEST_F(VersionSetTest, MemTableRegistryFailedCommitDoesNotEnableTracking) { + NewDB(); + VersionEdit invalid; + invalid.SetMemTableFileTracking(); + invalid.AddMemTableFile(0); + ASSERT_TRUE(LogAndApplyToDefaultCF(invalid).IsCorruption()); + ASSERT_FALSE(versions_->HasMemTableFileTracking()); + ASSERT_TRUE(versions_->GetMemTableFiles().empty()); +} + +TEST_F(VersionSetTest, MemTableRegistryRetirementQueuesObsoleteFile) { + NewDB(); + VersionEdit addition; + addition.SetMemTableFileTracking(); + addition.AddMemTableFile(10000); + ASSERT_OK(LogAndApplyToDefaultCF(addition)); + VersionEdit deletion; + deletion.DeleteMemTableFile(10000); + deletion.DeleteMemTableFile(10001); // Never registered; do not schedule it. + ASSERT_OK(LogAndApplyToDefaultCF(deletion)); + std::vector tables; + std::vector blobs; + std::vector manifests; + versions_->GetObsoleteFiles(&tables, &blobs, &manifests, 10000); + ASSERT_TRUE(tables.empty()); + manifests.clear(); + versions_->GetObsoleteFiles(&tables, &blobs, &manifests, 10001); + ASSERT_EQ(tables.size(), 1U); + ASSERT_EQ(tables[0].metadata->fd.GetNumber(), 10000U); + ASSERT_EQ(tables[0].path, + versions_->GetColumnFamilySet()->GetDefault()->ioptions()->cf_paths[0].path); + ASSERT_EQ(tables[0].metadata->table_reader_handle, nullptr); + tables[0].DeleteMetadata(); +} + +TEST_F(VersionSetTest, MemTableRegistryBatchReregistrationDoesNotRetire) { + NewDB(); + VersionEdit addition; + addition.SetMemTableFileTracking(); + addition.AddMemTableFile(10000); + ASSERT_OK(LogAndApplyToDefaultCF(addition)); + autovector> edits; + edits.emplace_back(new VersionEdit); + edits.back()->DeleteMemTableFile(10000); + edits.emplace_back(new VersionEdit); + edits.back()->AddMemTableFile(10000); + ASSERT_OK(LogAndApplyToDefaultCF(edits)); + ASSERT_EQ(versions_->GetMemTableFiles().at(0), std::set({10000})); + std::vector tables; + std::vector blobs; + std::vector manifests; + versions_->GetObsoleteFiles(&tables, &blobs, &manifests, 10001); + ASSERT_TRUE(tables.empty()); +} + +TEST_F(VersionSetTest, MemTableRegistryDropAndDiscardedEdit) { + NewDB(); + auto* cfd = versions_->GetColumnFamilySet()->GetColumnFamily(1); + ASSERT_NE(cfd, nullptr); + cfd->Ref(); + VersionEdit addition; + addition.SetColumnFamily(1); + addition.SetMemTableFileTracking(); + addition.AddMemTableFile(10000); + mutex_.Lock(); + Status status = versions_->LogAndApply( + cfd, mutable_cf_options_, read_options_, &addition, &mutex_, nullptr); + mutex_.Unlock(); + ASSERT_OK(status); + ASSERT_EQ(versions_->GetMemTableFiles().at(1), std::set({10000})); + VersionEdit drop; + drop.SetColumnFamily(1); + drop.DropColumnFamily(); + mutex_.Lock(); + status = versions_->LogAndApply( + cfd, mutable_cf_options_, read_options_, &drop, &mutex_, nullptr); + mutex_.Unlock(); + ASSERT_OK(status); + ASSERT_EQ(versions_->GetMemTableFiles().count(1), 0U); + std::vector retired_tables; + std::vector retired_blobs; + std::vector retired_manifests; + versions_->GetObsoleteFiles(&retired_tables, &retired_blobs, + &retired_manifests, 10001); + ASSERT_EQ(retired_tables.size(), 1U); + ASSERT_EQ(retired_tables[0].metadata->fd.GetNumber(), 10000U); + ASSERT_EQ(retired_tables[0].path, cfd->ioptions()->cf_paths[0].path); + retired_tables[0].DeleteMetadata(); + VersionEdit discarded; + discarded.SetColumnFamily(1); + discarded.AddMemTableFile(10001); + mutex_.Lock(); + status = versions_->LogAndApply( + cfd, mutable_cf_options_, read_options_, &discarded, &mutex_, nullptr); + cfd->UnrefAndTryDelete(); + mutex_.Unlock(); + ASSERT_TRUE(status.IsColumnFamilyDropped()); + ASSERT_EQ(versions_->GetMemTableFiles().count(1), 0U); + ASSERT_TRUE(versions_->HasMemTableFileTracking()); + CreateNewManifest(); + ReopenDB(); + ASSERT_TRUE(versions_->HasMemTableFileTracking()); + ASSERT_EQ(versions_->GetMemTableFiles().count(1), 0U); +} + TEST_F(VersionSetTest, PersistBlobFileStateInNewManifest) { // Initialize the database and add a couple of blob files, one with some // garbage in it, and one without any garbage. @@ -2624,6 +2757,54 @@ TEST_F(VersionSetAtomicGroupTest, EXPECT_EQ(num_initial_edits_ + kAtomicGroupSize, num_recovered_edits_); } +TEST_F(VersionSetAtomicGroupTest, MemTableRegistryAtomicTransition) { + SetupValidAtomicGroup(3); + edits_[0].SetMemTableFileTracking(); + edits_[0].AddMemTableFile(123); + edits_[1].DeleteMemTableFile(123); + edits_[1].AddMemTableFile(124); + // A later next-file record must not erase the allocator reservation. + edits_[2].SetNextFile(2); + AddNewEditsToLog(3); + ASSERT_OK(versions_->Recover(column_families_, false)); + ASSERT_TRUE(versions_->HasMemTableFileTracking()); + ASSERT_EQ(versions_->GetMemTableFiles().at(0), std::set({124})); + ASSERT_GT(versions_->current_next_file_number(), 124U); + auto* cfd = versions_->GetColumnFamilySet()->GetColumnFamily(0); + for (int level = 0; level < cfd->NumberLevels(); ++level) { + ASSERT_TRUE(cfd->current()->storage_info()->LevelFiles(level).empty()); + } + std::vector tables; + std::vector blobs; + std::vector manifests; + versions_->GetObsoleteFiles(&tables, &blobs, &manifests, 125); + ASSERT_TRUE(tables.empty()); // Historical replay must never schedule GC. +} + +TEST_F(VersionSetAtomicGroupTest, MemTableRegistryIncompleteAtomicGroup) { + SetupIncompleteTrailingAtomicGroup(3); + edits_[0].SetMemTableFileTracking(); + edits_[0].AddMemTableFile(123); + edits_[1].DeleteMemTableFile(123); + edits_[1].AddMemTableFile(124); + AddNewEditsToLog(2); + ASSERT_OK(versions_->Recover(column_families_, false)); + ASSERT_FALSE(versions_->HasMemTableFileTracking()); + ASSERT_TRUE(versions_->GetMemTableFiles().empty()); +} + +TEST_F(VersionSetAtomicGroupTest, MemTableRegistryTrackingSurvivesEmptyList) { + SetupValidAtomicGroup(3); + edits_[0].SetMemTableFileTracking(); + edits_[0].AddMemTableFile(123); + edits_[1].DeleteMemTableFile(123); + AddNewEditsToLog(3); + ASSERT_OK(versions_->Recover(column_families_, false)); + ASSERT_TRUE(versions_->HasMemTableFileTracking()); + ASSERT_TRUE(versions_->GetMemTableFiles().at(0).empty()); + ASSERT_GT(versions_->current_next_file_number(), 123U); +} + TEST_F(VersionSetAtomicGroupTest, HandleValidAtomicGroupWithReactiveVersionSetReadAndApply) { const int kAtomicGroupSize = 3; @@ -3681,6 +3862,29 @@ TEST_F(VersionSetTestMissingFiles, NoFileMissing) { } } +TEST_F(VersionSetTestMissingFiles, MemTableRegistryConversionKeepsSameNumberSst) { + NewDB(); + VersionEdit addition; + addition.SetMemTableFileTracking(); + addition.AddMemTableFile(100); + ASSERT_OK(LogAndApplyToDefaultCF(addition)); + SstInfo sst(100, kDefaultColumnFamilyName, "a", 0, 100); + std::vector file_metas; + CreateDummyTableFiles({sst}, &file_metas); + VersionEdit conversion; + conversion.DeleteMemTableFile(100); + conversion.AddFile(0, file_metas[0]); + ASSERT_OK(LogAndApplyToDefaultCF(conversion)); + ASSERT_TRUE(versions_->GetMemTableFiles().at(0).empty()); + std::vector tables; + std::vector blobs; + std::vector manifests; + versions_->GetObsoleteFiles(&tables, &blobs, &manifests, 101); + ASSERT_TRUE(tables.empty()); + ASSERT_EQ(versions_->GetColumnFamilySet()->GetDefault()->current() + ->storage_info()->LevelFiles(0).size(), 1U); +} + TEST_F(VersionSetTestMissingFiles, MinLogNumberToKeep2PC) { db_options_.allow_2pc = true; NewDB(); From cd79ca7a1fd6915c395b9403726a36fd0f07272a Mon Sep 17 00:00:00 2001 From: leipeng Date: Sat, 3 Oct 2026 20:54:17 +0800 Subject: [PATCH 21/58] Recover registered MemTables with stable DB file numbers A complete MANIFEST registry is useful only if each MemTable is registered before it accepts writes. Registration belongs to creation and installation, not the insert hot path; MemTables prepared ahead of a flush must obey the same rule before the cache can supply them. Giving the backing file its final SST number lets ConvertToSST transfer ownership without a rename gap. The MANIFEST can retire the MemTable and install its SST together. Until then the mutable file must remain protected from garbage collection without being exported as an immutable SST. Recovery can now compare against the registered set and fall back to full WAL replay when a required file is missing or unusable. Conversion of a leftover is repeatable, so interruption before the MANIFEST commit must leave the published recovery cursor usable for another attempt. --- db/column_family.cc | 82 +- db/column_family.h | 13 +- db/db_cspp_crash_safe_test.cc | 1384 +++++++++++++++++++++++++++---- db/db_impl/db_impl.cc | 18 + db/db_impl/db_impl.h | 8 +- db/db_impl/db_impl_files.cc | 9 + db/db_impl/db_impl_open.cc | 317 +++---- db/db_impl/db_impl_secondary.cc | 13 +- db/db_impl/db_impl_write.cc | 11 +- db/db_memtable_convert_test.cc | 151 ++++ db/flush_job.cc | 31 +- db/memtable.cc | 9 +- db/memtable.h | 11 +- db/memtable_list.cc | 15 +- db/version_edit_handler.cc | 3 +- db/version_set.cc | 6 +- db/version_set.h | 3 +- include/rocksdb/memtablerep.h | 21 +- 18 files changed, 1734 insertions(+), 371 deletions(-) diff --git a/db/column_family.cc b/db/column_family.cc index 2ea7cb857..c6a828f45 100644 --- a/db/column_family.cc +++ b/db/column_family.cc @@ -37,6 +37,7 @@ #include "rocksdb/convenience.h" #include "rocksdb/table.h" #include "table/merging_iterator.h" +#include "test_util/sync_point.h" #include "util/autovector.h" #include "util/cast_util.h" #include "util/compression.h" @@ -1164,7 +1165,14 @@ uint64_t ColumnFamilyData::GetLiveSstFilesSize() const { void ColumnFamilyData::PrepareNewMemtableInBackground( const MutableCFOptions& mutable_cf_options) { - #if !defined(ROCKSDB_UNIT_TEST) +#if defined(ROCKSDB_UNIT_TEST) + bool cache_enabled = false; +#else + bool cache_enabled = true; +#endif + TEST_SYNC_POINT_CALLBACK("ColumnFamilyData::MemTableCache:Enabled", + &cache_enabled); + if (!cache_enabled) return; { std::lock_guard lk(precreated_memtable_mutex_); if (precreated_memtable_list_.full()) { @@ -1173,8 +1181,11 @@ void ColumnFamilyData::PrepareNewMemtableInBackground( } } auto beg = ioptions_.clock->NowNanos(); + // dummy_versions_ remains alive for the lifetime of this CF, unlike current_. + uint64_t number = ioptions_.memtable_factory->SupportCrashSafe() + ? dummy_versions_->version_set()->NewFileNumber() : 0; auto tab = new MemTable(internal_comparator_, ioptions_, mutable_cf_options, - write_buffer_manager_, 0/*earliest_seq*/, id_); + write_buffer_manager_, 0/*earliest_seq*/, id_, number); auto end = ioptions_.clock->NowNanos(); RecordInHistogram(ioptions_.stats, MEMTAB_CONSTRUCT_NANOS, end - beg); { @@ -1191,21 +1202,67 @@ void ColumnFamilyData::PrepareNewMemtableInBackground( "precreated_memtable_list_ is full, discard the newly created memtab"); delete tab; } - #endif +} + +void ColumnFamilyData::AddPendingMemTableFileEdits(VersionEdit* edit) { + std::lock_guard lk(precreated_memtable_mutex_); + for (size_t i = 0; i < precreated_memtable_list_.size(); ++i) { + const auto& mem = *(precreated_memtable_list_.begin() + ptrdiff_t(i)); + if (mem->GetBackingFileNumber() != 0 && !mem->IsFileRegistered()) { + edit->AddMemTableFile(mem->GetBackingFileNumber()); + edit->SetMemTableFileTracking(); + } + } +} + +void ColumnFamilyData::PublishRegisteredMemTableCache() { + TEST_SYNC_POINT("FlushJob::MemTableCache:BeforePublish"); + { + std::lock_guard lk(precreated_memtable_mutex_); + const auto& files = dummy_versions_->version_set()->GetMemTableFiles(); + auto iter = files.find(id_); + if (iter != files.end()) { + for (size_t i = 0; i < precreated_memtable_list_.size(); ++i) { + const auto& mem = *(precreated_memtable_list_.begin() + ptrdiff_t(i)); + if (iter->second.count(mem->GetBackingFileNumber())) { + mem->MarkFileRegistered(); + } + } + } + } + TEST_SYNC_POINT("FlushJob::MemTableCache:AfterPublish"); +} + +void ColumnFamilyData::AddMemTableCacheFileNumbers(std::vector* live) { + std::lock_guard lk(precreated_memtable_mutex_); + for (size_t i = 0; i < precreated_memtable_list_.size(); ++i) { + const auto& mem = *(precreated_memtable_list_.begin() + ptrdiff_t(i)); + if (mem->GetBackingFileNumber() != 0) { + live->push_back(mem->GetBackingFileNumber()); + } + } } MemTable* ColumnFamilyData::ConstructNewMemtable( - const MutableCFOptions& mutable_cf_options, SequenceNumber earliest_seq) { + const MutableCFOptions& mutable_cf_options, SequenceNumber earliest_seq, + bool create_file) { MemTable* tab = nullptr; - #if !defined(ROCKSDB_UNIT_TEST) - { +#if defined(ROCKSDB_UNIT_TEST) + bool cache_enabled = false; +#else + bool cache_enabled = true; +#endif + TEST_SYNC_POINT_CALLBACK("ColumnFamilyData::MemTableCache:Enabled", + &cache_enabled); + if (cache_enabled && create_file) { std::lock_guard lk(precreated_memtable_mutex_); - if (!precreated_memtable_list_.empty()) { + if (!precreated_memtable_list_.empty() && + (precreated_memtable_list_.front()->GetBackingFileNumber() == 0 || + precreated_memtable_list_.front()->IsFileRegistered())) { tab = precreated_memtable_list_.front().release(); precreated_memtable_list_.pop_front(); } } - #endif if (tab) { tab->SetCreationSeq(earliest_seq); tab->SetEarliestSequenceNumber(earliest_seq); @@ -1213,8 +1270,10 @@ MemTable* ColumnFamilyData::ConstructNewMemtable( #if !defined(ROCKSDB_UNIT_TEST) auto beg = ioptions_.clock->NowNanos(); #endif + uint64_t number = create_file && ioptions_.memtable_factory->SupportCrashSafe() + ? dummy_versions_->version_set()->NewFileNumber() : 0; tab = new MemTable(internal_comparator_, ioptions_, mutable_cf_options, - write_buffer_manager_, earliest_seq, id_); + write_buffer_manager_, earliest_seq, id_, number); #if !defined(ROCKSDB_UNIT_TEST) auto end = ioptions_.clock->NowNanos(); RecordInHistogram(ioptions_.stats, MEMTAB_CONSTRUCT_NANOS, end - beg); @@ -1224,11 +1283,12 @@ MemTable* ColumnFamilyData::ConstructNewMemtable( } void ColumnFamilyData::CreateNewMemtable( - const MutableCFOptions& mutable_cf_options, SequenceNumber earliest_seq) { + const MutableCFOptions& mutable_cf_options, SequenceNumber earliest_seq, + bool create_file) { if (mem_ != nullptr) { delete mem_->Unref(); } - SetMemtable(ConstructNewMemtable(mutable_cf_options, earliest_seq)); + SetMemtable(ConstructNewMemtable(mutable_cf_options, earliest_seq, create_file)); mem_->Ref(); } diff --git a/db/column_family.h b/db/column_family.h index 367b94160..c8e923f91 100644 --- a/db/column_family.h +++ b/db/column_family.h @@ -37,6 +37,7 @@ namespace ROCKSDB_NAMESPACE { class Version; class VersionSet; +class VersionEdit; class VersionStorageInfo; class MemTable; class MemTableListVersion; @@ -373,12 +374,18 @@ class ColumnFamilyData { uint64_t OldestLogToKeep(); void PrepareNewMemtableInBackground(const MutableCFOptions&); + // DB mutex must be held when registering or publishing cached files. + void AddPendingMemTableFileEdits(VersionEdit* edit); + void AddMemTableCacheFileNumbers(std::vector* live); + void PublishRegisteredMemTableCache(); // See Memtable constructor for explanation of earliest_seq param. MemTable* ConstructNewMemtable(const MutableCFOptions& mutable_cf_options, - SequenceNumber earliest_seq); + SequenceNumber earliest_seq, + bool create_file = true); void CreateNewMemtable(const MutableCFOptions& mutable_cf_options, - SequenceNumber earliest_seq); + SequenceNumber earliest_seq, + bool create_file = true); TableCache* table_cache() const { return table_cache_.get(); } BlobSource* blob_source() const { return blob_source_.get(); } @@ -612,11 +619,9 @@ class ColumnFamilyData { WriteBufferManager* write_buffer_manager_; - #if !defined(ROCKSDB_UNIT_TEST) // precreated_memtable_list_.size() is normally 1 terark::fixed_circular_queue, 4> precreated_memtable_list_; std::mutex precreated_memtable_mutex_; - #endif MemTable* mem_; MemTableList imm_; diff --git a/db/db_cspp_crash_safe_test.cc b/db/db_cspp_crash_safe_test.cc index aa0268ea4..f724b95fc 100644 --- a/db/db_cspp_crash_safe_test.cc +++ b/db/db_cspp_crash_safe_test.cc @@ -10,6 +10,8 @@ #include #include #include +#include +#include #include #include #include @@ -27,6 +29,7 @@ #include "db/log_reader.h" #include "db/log_writer.h" #include "db/pre_release_callback.h" +#include "db/version_set.h" #include "file/filename.h" #include "file/file_util.h" #include "file/sequence_file_reader.h" @@ -35,6 +38,7 @@ #include "port/stack_trace.h" #include "rocksdb/io_status.h" #include "rocksdb/statistics.h" +#include "rocksdb/utilities/checkpoint.h" #include "rocksdb/utilities/transaction_db.h" #include "rocksdb/wal_filter.h" #include "table/get_context.h" @@ -107,8 +111,74 @@ Options BaseCrashSafeOptions(const std::string& dbname, bool recover, std::vector ListLeftovers(const Options& options, const std::string& dir) { std::vector leftovers; - if (options.memtable_factory) { - options.memtable_factory->ListCrashSafeLeftovers(dir, &leftovers); + Env* env = options.env ? options.env : Env::Default(); + std::string current; + Status s = env->FileExists(CurrentFileName(dir)); + if (s.IsNotFound()) return leftovers; + EXPECT_OK(s); + if (!s.ok()) return leftovers; + s = ReadFileToString(env, CurrentFileName(dir), ¤t); + if (!s.ok()) { + ADD_FAILURE() << s.ToString(); + return leftovers; + } + EXPECT_FALSE(current.empty()); + if (current.empty()) return leftovers; + if (current.back() == '\n') current.pop_back(); + const std::string manifest = dir + "/" + current; + std::unique_ptr file; + s = env->GetFileSystem()->NewSequentialFile(manifest, FileOptions(), &file, + nullptr); + EXPECT_OK(s); + if (!s.ok()) return leftovers; + struct Reporter : log::Reader::Reporter { + void Corruption(size_t, const Status& status) override { + ADD_FAILURE() << status.ToString(); + } + } reporter; + auto input = std::make_unique(std::move(file), manifest); + log::Reader reader(nullptr, std::move(input), &reporter, true, 0); + std::map> registered; + auto apply = [&](const VersionEdit& edit) { + const uint32_t cf = edit.GetColumnFamily(); + if (edit.IsColumnFamilyDrop()) { + registered.erase(cf); + return; + } + for (uint64_t number : edit.GetMemTableFileDeletions()) { + registered[cf].erase(number); + } + for (uint64_t number : edit.GetMemTableFileAdditions()) { + registered[cf].insert(number); + } + }; + AtomicGroupReadBuffer group; + Slice record; + std::string scratch; + while (reader.ReadRecord(&record, &scratch)) { + VersionEdit edit; + s = edit.DecodeFrom(record); + EXPECT_OK(s); + if (!s.ok()) break; + s = group.AddEdit(&edit); + EXPECT_OK(s); + if (!s.ok()) break; + if (!edit.IsInAtomicGroup()) { + apply(edit); + } else if (group.IsFull()) { + for (const auto& member : group.replay_buffer()) apply(member); + group.Clear(); + } + } + const std::string path = !options.cf_paths.empty() + ? options.cf_paths[0].path + : !options.db_paths.empty() + ? options.db_paths[0].path + : dir; + for (const auto& cf : registered) { + for (uint64_t number : cf.second) { + leftovers.push_back(MakeTableFileName(path, number)); + } } return leftovers; } @@ -230,6 +300,695 @@ class DBCsppCrashSafeTest : public DBTestBase { : DBTestBase("db_cspp_crash_safe_test", /*env_do_fsync=*/false) {} }; +#if !defined(OS_WIN) +TEST_F(CrashChild, DISABLED_RegisteredMemTables) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (arg_ == "osl") SetupOsl(&options, true); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "first", "1")); + ASSERT_OK(static_cast(child_db)->TEST_SwitchMemtable()); + ASSERT_OK(child_db->Put(WriteOptions(), "second", "2")); + ASSERT_OK(static_cast(child_db)->TEST_SwitchMemtable()); + // The registered empty active file must also exist for prefix recovery. + ::_exit(42); +} + +TEST_F(DBCsppCrashSafeTest, MissingRegisteredMemTableUsesFullWal) { + Close(); + for (bool osl : {false, true}) { + for (int missing : {0, 1, 2, 3}) { + SCOPED_TRACE(osl); + SCOPED_TRACE(missing); + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "RegisteredMemTables", + osl ? "osl" : "cspp"), 42); + const auto registered = ListLeftovers(options, dbname_); + ASSERT_EQ(registered.size(), 3U); + if (missing == 3) { + for (const auto& path : registered) ASSERT_OK(env_->DeleteFile(path)); + } else { + ASSERT_OK(env_->DeleteFile(registered[missing])); + } + std::atomic converted{0}; + SyncPoint::GetInstance()->SetCallBack( + "MemTableRep::ConvertToSST:After", + [&](void*) { ++converted; }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(options)); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + // Recovery may convert an intact prefix before discovering the missing + // registered file, but must discard that prefix and replay the full WAL. + ASSERT_EQ(converted.load(), missing == 3 ? 0 : missing); + ASSERT_EQ(Get("first"), "1"); + ASSERT_EQ(Get("second"), "2"); + ASSERT_EQ(NumTableFilesAtLevel(0), 0); + ASSERT_OK(Flush()); + ASSERT_EQ(NumTableFilesAtLevel(0), 1); + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("first"), "1"); + ASSERT_EQ(Get("second"), "2"); + ASSERT_EQ(NumTableFilesAtLevel(0), 1); + Close(); + } + } +} + +TEST_F(DBCsppCrashSafeTest, FailedRegistrationCannotAcceptWrites) { + Close(); + for (bool osl : {false, true}) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("before", "safe")); + std::atomic failed{0}; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::RegisterMemTableFile:BeforeLogAndApply", [&](void* p) { + ++failed; + *static_cast(p) = Status::IOError("register injection"); + }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_NOK(dbfull()->TEST_SwitchMemtable()); + ASSERT_GT(failed.load(), 0); + ASSERT_NOK(Put("unregistered", "must-not-commit")); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_EQ(Get("before"), "safe"); + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("before"), "safe"); + ASSERT_EQ(Get("unregistered"), "NOT_FOUND"); + Close(); + } +} + +TEST_F(DBCsppCrashSafeTest, FailedInitialRegistrationClearsDbPointer) { + Close(); + for (bool osl : {false, true}) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + Destroy(options); + std::atomic failed{0}; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::RegisterMemTableFile:BeforeLogAndApply", [&](void* p) { + ++failed; + *static_cast(p) = Status::IOError("initial register injection"); + }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* opened = nullptr; + const Status status = DB::Open(options, dbname_, &opened); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_NOK(status); + ASSERT_GT(failed.load(), 0); + ASSERT_EQ(opened, nullptr); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("after-failure", "safe")); + Close(); + } +} + +TEST_F(CrashChild, DISABLED_RegistrationCommitWindow) { + ASSERT_EQ(arg_.size(), 3U); + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (arg_[0] == '1') SetupOsl(&options, true); + const char* point = arg_[1] == '0' + ? "DBImpl::RegisterMemTableFile:AfterLogAndApply" + : "DBImpl::RegisterMemTableFile:BeforeInstall"; + const auto arm = [&] { + SyncPoint::GetInstance()->SetCallBack(point, [](void*) { ::_exit(42); }); + SyncPoint::GetInstance()->EnableProcessing(); + }; + if (arg_[2] == '0') arm(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "before-switch", "preserved")); + if (arg_[2] == '1') arm(); + ASSERT_OK(static_cast(child_db)->TEST_SwitchMemtable()); + ::_exit(1); +} + +TEST_F(DBCsppCrashSafeTest, RegistrationCommitCrashKeepsManifestInventory) { + Close(); + for (bool osl : {false, true}) { + for (bool marked : {false, true}) { + for (bool switching : {false, true}) { + SCOPED_TRACE(osl); + SCOPED_TRACE(marked); + SCOPED_TRACE(switching); + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + Destroy(options); + const std::string arg = std::to_string(osl) + std::to_string(marked) + + std::to_string(switching); + ASSERT_EQ(RunCrashChild(dbname_, "RegistrationCommitWindow", arg), 42); + const auto registered = ListLeftovers(options, dbname_); + ASSERT_EQ(registered.size(), switching ? 2U : 1U); + for (const auto& path : registered) ASSERT_OK(env_->FileExists(path)); + ASSERT_OK(TryReopen(options)); + ASSERT_TRUE(dbfull()->GetVersionSet()->HasMemTableFileTracking()); + ASSERT_EQ(Get("before-switch"), switching ? "preserved" : "NOT_FOUND"); + ASSERT_OK(Put("after-crash", "committed")); + Close(); + for (const auto& path : ListLeftovers(options, dbname_)) + ASSERT_OK(env_->FileExists(path)); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("before-switch"), switching ? "preserved" : "NOT_FOUND"); + ASSERT_EQ(Get("after-crash"), "committed"); + Close(); + } + } + } +} + +TEST_F(DBCsppCrashSafeTest, FailedNewColumnFamilyRegistrationRemainsReopenable) { + Close(); + for (bool osl : {false, true}) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("existing", "preserved")); + std::atomic failed{0}; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::RegisterMemTableFile:BeforeLogAndApply", [&](void* p) { + ++failed; + *static_cast(p) = Status::IOError("new CF register injection"); + }); + SyncPoint::GetInstance()->EnableProcessing(); + ColumnFamilyHandle* handle = nullptr; + const Status created = db_->CreateColumnFamily(options, "failed", &handle); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_NOK(created); + ASSERT_GT(failed.load(), 0); + ASSERT_EQ(handle, nullptr); + ASSERT_EQ(Get("existing"), "preserved"); + // Closing and reopening also exercises manifest snapshots over this CF. + Close(); + ASSERT_OK(TryReopenWithColumnFamilies({kDefaultColumnFamilyName, "failed"}, + options)); + ASSERT_EQ(Get(0, "existing"), "preserved"); + ASSERT_EQ(Get(1, "existing"), "NOT_FOUND"); + std::unique_ptr iterator(db_->NewIterator(ReadOptions(), handles_[1])); + iterator->SeekToFirst(); + ASSERT_FALSE(iterator->Valid()); + ASSERT_OK(iterator->status()); + iterator.reset(); + Close(); + } +} + +TEST_F(DBCsppCrashSafeTest, FileMmapRejectsReadOnlyAndSecondary) { + Close(); + for (bool osl : {false, true}) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "WalFilterFallsBackToWal", + osl ? "10" : "00"), 0); + std::vector before; + ASSERT_OK(env_->GetChildren(dbname_, &before)); + std::sort(before.begin(), before.end()); + for (bool secondary : {false, true}) { + SCOPED_TRACE(osl); + SCOPED_TRACE(secondary); + DB* rejected = nullptr; + const Status s = secondary + ? DB::OpenAsSecondary(options, dbname_, dbname_ + "_secondary", + &rejected) + : DB::OpenForReadOnly(options, dbname_, &rejected); + ASSERT_TRUE(s.IsInvalidArgument()) << s.ToString(); + ASSERT_EQ(rejected, nullptr); + std::vector after; + ASSERT_OK(env_->GetChildren(dbname_, &after)); + std::sort(after.begin(), after.end()); + ASSERT_EQ(after, before); + } + } +} + +TEST_F(DBCsppCrashSafeTest, ReadWriteWalRecoveryFailureDoesNotAbort) { + class CorruptSecondRecord final : public WalFilter { + public: + int calls = 0; + const char* Name() const override { return "CorruptSecondRecord"; } + WalProcessingOption LogRecordFound(unsigned long long, const std::string&, + const WriteBatch&, WriteBatch*, + bool*) override { + return ++calls == 1 ? WalProcessingOption::kContinueProcessing + : WalProcessingOption::kCorruptedRecord; + } + }; + Close(); + for (bool osl : {false, true}) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "WalFilterFallsBackToWal", + osl ? "10" : "00"), 0); + auto list_sst = [&] { + std::vector children; + EXPECT_OK(env_->GetChildren(dbname_, &children)); + std::vector files; + for (const auto& child : children) { + if (child.size() >= 4 && child.compare(child.size() - 4, 4, ".sst") == 0) + files.push_back(child); + } + std::sort(files.begin(), files.end()); + return files; + }; + const auto before = list_sst(); + ASSERT_FALSE(before.empty()); + CorruptSecondRecord filter; + options.memtable_crash_safe_recover = false; + options.wal_filter = &filter; + options.wal_recovery_mode = WALRecoveryMode::kAbsoluteConsistency; + DB* failed = nullptr; + const Status status = DB::Open(options, dbname_, &failed); + ASSERT_TRUE(status.IsCorruption()) << status.ToString(); + ASSERT_EQ(failed, nullptr); + ASSERT_EQ(filter.calls, 2); + ASSERT_EQ(list_sst(), before); + options.wal_filter = nullptr; + options.memtable_crash_safe_recover = true; + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("a"), std::string(128, 'a')); + ASSERT_EQ(Get("b"), std::string(128, 'b')); + Close(); + } +} + +TEST_F(DBCsppCrashSafeTest, ManifestRolloverPreservesMemTableRegistry) { + Close(); + for (bool osl : {false, true}) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.max_manifest_file_size = 1; + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("first", "1")); + std::string before; + ASSERT_OK(ReadFileToString(env_, CurrentFileName(dbname_), &before)); + ASSERT_OK(dbfull()->TEST_SwitchMemtable()); + ASSERT_OK(Put("second", "2")); + std::string after; + ASSERT_OK(ReadFileToString(env_, CurrentFileName(dbname_), &after)); + ASSERT_NE(before, after); + const auto registered = dbfull()->GetVersionSet()->GetMemTableFiles(); + ASSERT_EQ(registered.count(0), 1U); + ASSERT_EQ(registered.at(0).size(), 2U); + const auto disk = ListLeftovers(options, dbname_); + ASSERT_EQ(disk.size(), 2U); + for (uint64_t number : registered.at(0)) { + const auto path = MakeTableFileName(dbname_, number); + ASSERT_EQ(std::count(disk.begin(), disk.end(), path), 1); + ASSERT_OK(env_->FileExists(path)); + } + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("first"), "1"); + ASSERT_EQ(Get("second"), "2"); + Close(); + } +} + +TEST_F(CrashChild, DISABLED_LegacyManifestWal) { + Options options = BaseCrashSafeOptions(dbname_, false, false); + options.memtable_factory = std::make_shared(); + options.table_factory = Options().table_factory; + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + WriteOptions write; + write.sync = true; + ASSERT_OK(child_db->Put(write, "legacy-first", "one")); + ASSERT_OK(child_db->Put(write, "legacy-second", "two")); + ::_exit(42); +} + +TEST_F(DBCsppCrashSafeTest, LegacyManifestWithoutTrackingUsesFullWal) { + Close(); + for (bool osl : {false, true}) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "LegacyManifestWal"), 42); + ASSERT_TRUE(ListLeftovers(options, dbname_).empty()); + std::vector children; + ASSERT_OK(env_->GetChildren(dbname_, &children)); + uint64_t wal_number = 0; + for (const auto& child : children) { + uint64_t number; + FileType type; + if (ParseFileName(child, &number, &type) && type == kWalFile) + wal_number = std::max(wal_number, number); + } + ASSERT_NE(wal_number, 0U); + uint64_t wal_size = 0; + ASSERT_OK(env_->GetFileSize(LogFileName(dbname_, wal_number), &wal_size)); + ASSERT_GT(wal_size, 0U); + // Valid classic sidecar deliberately points past both keys. An inventory + // without tracking cannot justify skipping this WAL prefix. + PublishedSeqOnDisk record; + record.magic = 0x5145534255505343ULL; + record.version = 1; + record.header_size = sizeof(record); + record.wal_offset_kind = 1; + record.kind_since_wal = static_cast(wal_number); + record.generation = 2; + record.pubseq = 2; + record.wal_number = wal_number; + record.wal_offset = wal_size; + std::string sidecar(4096, '\0'); + std::memcpy(&sidecar[0], &record, sizeof(record)); + ASSERT_OK(WriteStringToFile(env_, sidecar, CrashSafePubSeqFileName(dbname_))); + std::atomic converted{0}; + std::atomic reads{0}; + SyncPoint::GetInstance()->SetCallBack( + "MemTableRep::ConvertToSST:After", [&](void*) { ++converted; }); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::RecoverLogFiles:BeforeReadWal", [&](void*) { ++reads; }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(options)); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_EQ(converted.load(), 0); + ASSERT_GT(reads.load(), 0); + ASSERT_EQ(Get("legacy-first"), "one"); + ASSERT_EQ(Get("legacy-second"), "two"); + ASSERT_TRUE(dbfull()->GetVersionSet()->HasMemTableFileTracking()); + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("legacy-first"), "one"); + ASSERT_EQ(Get("legacy-second"), "two"); + Close(); + } +} + +TEST_F(CrashChild, DISABLED_FlushManifestWindow) { + ASSERT_EQ(arg_.size(), 2U); + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (arg_[0] == '1') SetupOsl(&options, true); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "converted", "durable")); + SyncPoint::GetInstance()->SetCallBack( + arg_[1] == '0' ? "FlushJob::BeforeManifest" + : "FlushJob::AfterManifest", + [](void*) { ::_exit(42); }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(child_db->Flush(FlushOptions())); + ::_exit(1); +} + +TEST_F(DBCsppCrashSafeTest, ConversionCrashAcrossManifestCommit) { + Close(); + for (bool osl : {false, true}) { + for (bool committed : {false, true}) { + SCOPED_TRACE(osl); + SCOPED_TRACE(committed); + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + Destroy(options); + const std::string arg = std::string(osl ? "1" : "0") + + (committed ? "1" : "0"); + ASSERT_EQ(RunCrashChild(dbname_, "FlushManifestWindow", arg), 42); + const auto registered = ListLeftovers(options, dbname_); + ASSERT_FALSE(registered.empty()); + for (const auto& path : registered) ASSERT_OK(env_->FileExists(path)); + std::atomic converts{0}; + SyncPoint::GetInstance()->SetCallBack( + "MemTableRep::ConvertToSST:After", + [&](void*) { ++converts; }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(options)); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_EQ(converts.load(), committed ? 0 : 1); + ASSERT_EQ(Get("converted"), "durable"); + ASSERT_EQ(CountL0(db_), 1); + std::vector files; + db_->GetLiveFilesMetaData(&files); + ASSERT_EQ(files.size(), 1U); + if (!committed) { + const std::string path = MakeTableFileName(dbname_, files[0].file_number); + ASSERT_EQ(std::count(registered.begin(), registered.end(), path), 1); + ASSERT_OK(env_->FileExists(path)); + } + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("converted"), "durable"); + ASSERT_EQ(CountL0(db_), 1); + Close(); + } + } +} + +TEST_F(DBCsppCrashSafeTest, GarbageCollectionKeepsActiveAndCachedMemTables) { + Close(); + for (bool osl : {false, true}) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + Destroy(options); + SyncPoint::GetInstance()->SetCallBack( + "ColumnFamilyData::MemTableCache:Enabled", + [](void* p) { *static_cast(p) = true; }); + std::atomic published{0}; + SyncPoint::GetInstance()->SetCallBack( + "FlushJob::MemTableCache:BeforePublish", [&](void*) { + const auto& files = dbfull()->GetVersionSet()->GetMemTableFiles(); + std::vector expected; + for (const auto& cf : files) { + for (uint64_t number : cf.second) { + expected.push_back(MakeTableFileName(dbname_, number)); + } + } + ASSERT_EQ(ListLeftovers(options, dbname_), expected); + for (const auto& path : expected) ASSERT_OK(env_->FileExists(path)); + }); + SyncPoint::GetInstance()->SetCallBack( + "FlushJob::MemTableCache:AfterPublish", [&](void*) { ++published; }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("first", "1")); + ASSERT_OK(Flush()); + ASSERT_GT(published.load(), 0); + ASSERT_OK(Put("active", "2")); + const auto registered = dbfull()->GetVersionSet()->GetMemTableFiles(); + ASSERT_EQ(registered.count(0), 1U); + ASSERT_GE(registered.at(0).size(), 2U); + ASSERT_OK(db_->DisableFileDeletions()); + ASSERT_OK(db_->EnableFileDeletions(true)); + for (uint64_t number : registered.at(0)) { + ASSERT_OK(env_->FileExists(MakeTableFileName(dbname_, number))); + } + ASSERT_EQ(Get("first"), "1"); + ASSERT_EQ(Get("active"), "2"); + Close(); + const auto kept = ListLeftovers(options, dbname_); + ASSERT_FALSE(kept.empty()); + for (const auto& path : kept) ASSERT_OK(env_->FileExists(path)); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("first"), "1"); + ASSERT_EQ(Get("active"), "2"); + Close(); + } +} + +TEST_F(DBCsppCrashSafeTest, EmptyFlushRetiresSourceFileWithoutFullScan) { + Close(); + for (bool osl : {false, true}) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.delete_obsolete_files_period_micros = UINT64_MAX; + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_OK(TryReopen(options)); + const auto source = ListLeftovers(options, dbname_); + ASSERT_EQ(source.size(), 1U); + ASSERT_OK(env_->FileExists(source[0])); + ASSERT_OK(dbfull()->TEST_SwitchMemtable()); + ASSERT_OK(Flush()); + ASSERT_OK(dbfull()->TEST_WaitForBackgroundWork()); + ASSERT_OK(dbfull()->TEST_WaitForPurge()); + std::vector files; + db_->GetLiveFilesMetaData(&files); + ASSERT_TRUE(files.empty()); + ASSERT_TRUE(env_->FileExists(source[0]).IsNotFound()); + const auto active = ListLeftovers(options, dbname_); + ASSERT_EQ(active.size(), 1U); + for (const auto& path : active) ASSERT_OK(env_->FileExists(path)); + ASSERT_OK(Put("converted", "preserved")); + ASSERT_OK(Flush()); + ASSERT_OK(dbfull()->TEST_WaitForBackgroundWork()); + ASSERT_OK(dbfull()->TEST_WaitForPurge()); + db_->GetLiveFilesMetaData(&files); + ASSERT_EQ(files.size(), 1U); + ASSERT_EQ(MakeTableFileName(dbname_, files[0].file_number), active[0]); + ASSERT_OK(env_->FileExists(active[0])); + for (const auto& path : ListLeftovers(options, dbname_)) + ASSERT_OK(env_->FileExists(path)); + ASSERT_EQ(Get("converted"), "preserved"); + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("converted"), "preserved"); + Close(); + } +} + +TEST_F(DBCsppCrashSafeTest, DroppedCfRetiresSourceFilesWithoutFullScan) { + Close(); + for (bool osl : {false, true}) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.delete_obsolete_files_period_micros = UINT64_MAX; + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_OK(TryReopen(options)); + CreateAndReopenWithCF({"retire"}, options); + ASSERT_OK(Put(0, "keep", "one")); + ASSERT_OK(Put(1, "drop", "two")); + const auto registered = dbfull()->GetVersionSet()->GetMemTableFiles(); + ASSERT_EQ(registered.count(1), 1U); + ASSERT_EQ(registered.at(1).size(), 1U); + const std::string source = MakeTableFileName(dbname_, *registered.at(1).begin()); + ASSERT_OK(env_->FileExists(source)); + ASSERT_OK(db_->DropColumnFamily(handles_[1])); + ASSERT_OK(db_->DestroyColumnFamilyHandle(handles_[1])); + handles_.pop_back(); + // An ordinary flush provides normal obsolete-file GC, without a scan. + ASSERT_OK(Flush(0)); + ASSERT_OK(dbfull()->TEST_WaitForBackgroundWork()); + ASSERT_OK(dbfull()->TEST_WaitForPurge()); + ASSERT_TRUE(env_->FileExists(source).IsNotFound()); + ASSERT_EQ(dbfull()->GetVersionSet()->GetMemTableFiles().count(1), 0U); + for (const auto& path : ListLeftovers(options, dbname_)) + ASSERT_OK(env_->FileExists(path)); + std::vector files; + db_->GetLiveFilesMetaData(&files); + ASSERT_EQ(files.size(), 1U); + ASSERT_OK(env_->FileExists(MakeTableFileName(dbname_, files[0].file_number))); + ASSERT_EQ(Get(0, "keep"), "one"); + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("keep"), "one"); + Close(); + } +} + +TEST_F(DBCsppCrashSafeTest, CheckpointDoesNotHardLinkWritableMemTable) { + Close(); + for (bool osl : {false, true}) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("checkpoint", "original")); + const auto registered = dbfull()->GetVersionSet()->GetMemTableFiles(); + ASSERT_EQ(registered.count(0), 1U); + ASSERT_EQ(registered.at(0).size(), 1U); + const uint64_t number = *registered.at(0).begin(); + const std::string checkpoint_dir = dbname_ + ".checkpoint"; + Options copy_options = options; + copy_options.wal_dir = checkpoint_dir; + ASSERT_OK(DestroyDB(checkpoint_dir, copy_options)); + Checkpoint* raw = nullptr; + ASSERT_OK(Checkpoint::Create(db_, &raw)); + std::unique_ptr checkpoint(raw); + ASSERT_OK(checkpoint->CreateCheckpoint(checkpoint_dir, UINT64_MAX)); + // This memtable is still mutable, and hence must not be a live SST. + const auto& after = dbfull()->GetVersionSet()->GetMemTableFiles(); + auto cf = after.find(0); + if (cf != after.end() && cf->second.count(number)) { + ASSERT_TRUE(env_->FileExists(MakeTableFileName(checkpoint_dir, number)) + .IsNotFound()); + } + ASSERT_OK(Put("checkpoint", "source-changed")); + DB* copy_raw = nullptr; + ASSERT_OK(DB::Open(copy_options, checkpoint_dir, ©_raw)); + std::unique_ptr copy(copy_raw); + std::string value; + ASSERT_OK(copy->Get(ReadOptions(), "checkpoint", &value)); + ASSERT_EQ(value, "original"); + copy.reset(); + checkpoint.reset(); + ASSERT_OK(DestroyDB(checkpoint_dir, copy_options)); + Close(); + } +} + +TEST_F(DBCsppCrashSafeTest, FailedCacheRegistrationSurvivesGcAndReopen) { + Close(); + for (bool osl : {false, true}) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.max_bgerror_resume_count = 0; + if (osl) SetupOsl(&options, true); + Destroy(options); + SyncPoint::GetInstance()->SetCallBack( + "ColumnFamilyData::MemTableCache:Enabled", + [](void* p) { *static_cast(p) = true; }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("retry", "preserved")); + std::vector pending; + std::atomic pending_commit{false}; + SyncPoint::GetInstance()->SetCallBack("FlushJob::BeforeManifest", [&](void*) { + std::vector children; + ASSERT_OK(env_->GetChildren(dbname_, &children)); + for (const auto& child : children) { + if (child.size() >= 4 && child.compare(child.size() - 4, 4, ".sst") == 0) + pending.push_back(dbname_ + "/" + child); + } + pending_commit.store(true); + }); + std::atomic injected{false}; + std::atomic published{0}; + SyncPoint::GetInstance()->SetCallBack( + "VersionSet::ProcessManifestWrites:AfterSyncManifest", [&](void* p) { + if (pending_commit.load() && !injected.exchange(true)) + *static_cast(p) = IOStatus::IOError("cache register injection"); + }); + SyncPoint::GetInstance()->SetCallBack( + "FlushJob::MemTableCache:AfterPublish", [&](void*) { ++published; }); + ASSERT_NOK(Flush()); + ASSERT_TRUE(injected.load()); + ASSERT_EQ(published.load(), 0); + ASSERT_GE(pending.size(), 3U); // converting input, active, pending cache + ASSERT_OK(db_->DisableFileDeletions()); + ASSERT_OK(db_->EnableFileDeletions(true)); + for (const auto& path : pending) ASSERT_OK(env_->FileExists(path)); + SyncPoint::GetInstance()->ClearCallBack("FlushJob::BeforeManifest"); + // Plain MANIFEST IOError is fatal under the existing error policy. + // Resume preserves that error; reopening is the supported recovery path. + const Status resumed = db_->Resume(); + ASSERT_TRUE(resumed.IsIOError()); + ASSERT_EQ(Get("retry"), "preserved"); + Close(); + SyncPoint::GetInstance()->ClearCallBack( + "VersionSet::ProcessManifestWrites:AfterSyncManifest"); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("retry"), "preserved"); + published.store(0); + ASSERT_OK(Put("after-reopen", "committed")); + ASSERT_OK(Flush()); + ASSERT_GT(published.load(), 0); + ASSERT_EQ(Get("retry"), "preserved"); + Close(); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("retry"), "preserved"); + ASSERT_EQ(Get("after-reopen"), "committed"); + Close(); + } +} +#endif + TEST_F(DBCsppCrashSafeTest, CrashSafeRequiresFileMmap) { for (const char* cls : {"CSPPMemTab", "OffsetSkipList"}) { for (const char* mode : {"kDontConvert", "kDumpMem", "kFileMmap"}) { @@ -355,15 +1114,13 @@ TEST_F(DBCsppCrashSafeTest, RecoveredTableUsesFileSequenceBound) { MutableCFOptions moptions(options); WriteBufferManager wb(options.db_write_buffer_size); std::unique_ptr mem(new MemTable( - icmp, ioptions, moptions, &wb, kMaxSequenceNumber, 0)); + icmp, ioptions, moptions, &wb, kMaxSequenceNumber, 0, 1)); ASSERT_OK(mem->Add(1, kTypeValue, "key", "old", nullptr)); ASSERT_OK(mem->Add(3, kTypeValue, "key", "new", nullptr)); ASSERT_OK(mem->Add(4, kTypeValue, "ghost", "unpublished", nullptr)); mem->MarkImmutable(); - const auto leftovers = ListLeftovers(options, dbname_); - ASSERT_EQ(leftovers.size(), 1U); - const std::string leftover = leftovers[0] + ".recovery"; - CopyFile(leftovers[0], leftover); + const std::string leftover = MakeTableFileName(dbname_, 2); + CopyFile(MakeTableFileName(dbname_, 1), leftover); mem.reset(); IntTblPropCollectorFactories collectors; @@ -371,7 +1128,7 @@ TEST_F(DBCsppCrashSafeTest, RecoveredTableUsesFileSequenceBound) { options.compression, options.compression_opts, 0, "default", 0); FileMetaData meta; - meta.fd = FileDescriptor(1, 0, 0); + meta.fd = FileDescriptor(2, 0, 0); meta.fd.smallest_seqno = 0; // The published bound need not be the sequence of any physical entry. meta.fd.largest_seqno = 2; @@ -381,7 +1138,7 @@ TEST_F(DBCsppCrashSafeTest, RecoveredTableUsesFileSequenceBound) { ASSERT_EQ(meta.fd.smallest_seqno, 0U); ASSERT_EQ(meta.fd.largest_seqno, 2U); - const std::string fname = TableFileName(options.cf_paths, 1, 0); + const std::string fname = TableFileName(options.cf_paths, 2, 0); for (SequenceNumber limit : {meta.fd.largest_seqno, SequenceNumber(4), SequenceNumber(0)}) { SCOPED_TRACE(limit); @@ -438,7 +1195,7 @@ TEST_F(DBCsppCrashSafeTest, RecoveredTableIteratorUsesFileSequenceBound) { MutableCFOptions moptions(options); WriteBufferManager wb(options.db_write_buffer_size); auto mem = std::make_unique( - icmp, ioptions, moptions, &wb, kMaxSequenceNumber, 0); + icmp, ioptions, moptions, &wb, kMaxSequenceNumber, 0, 1); std::vector physical; for (const auto& entry : {std::make_pair("b", 1), {"d", 2}, {"b", 3}, {"d", 5}, {"c", 6}, {"b", 7}, {"a", 8}, @@ -453,22 +1210,20 @@ TEST_F(DBCsppCrashSafeTest, RecoveredTableIteratorUsesFileSequenceBound) { }; std::sort(physical.begin(), physical.end(), less); mem->MarkImmutable(); - const auto leftovers = ListLeftovers(options, dbname_); - ASSERT_EQ(leftovers.size(), 1U); - const std::string leftover = leftovers[0] + ".iterator"; - CopyFile(leftovers[0], leftover); + const std::string leftover = MakeTableFileName(dbname_, 2); + CopyFile(MakeTableFileName(dbname_, 1), leftover); mem.reset(); IntTblPropCollectorFactories collectors; TableBuilderOptions tbo(ioptions, moptions, icmp, &collectors, options.compression, options.compression_opts, 0, "default", 0); FileMetaData meta; - meta.fd = FileDescriptor(1, 0, 0); + meta.fd = FileDescriptor(2, 0, 0); meta.fd.smallest_seqno = 0; meta.fd.largest_seqno = 4; ASSERT_OK(options.memtable_factory->RecoverCrashSafeMemTableToSST( leftover, &meta, tbo)); - const std::string fname = TableFileName(options.cf_paths, 1, 0); + const std::string fname = TableFileName(options.cf_paths, 2, 0); for (SequenceNumber limit : {SequenceNumber(0), SequenceNumber(4), SequenceNumber(9), kMaxSequenceNumber}) { SCOPED_TRACE(limit); @@ -613,7 +1368,7 @@ TEST_F(DBCsppCrashSafeTest, ConvertedTableVisibilityFilter) { MutableCFOptions moptions(options); WriteBufferManager wb(options.db_write_buffer_size); auto mem = std::make_unique( - icmp, ioptions, moptions, &wb, kMaxSequenceNumber, 0); + icmp, ioptions, moptions, &wb, kMaxSequenceNumber, 0, 1); ASSERT_OK(mem->Add(1, kTypeValue, "a", "1", nullptr)); ASSERT_OK(mem->Add(2, kTypeValue, "b", "2", nullptr)); ASSERT_OK(mem->Add(3, kTypeValue, "b", "3", nullptr)); @@ -674,6 +1429,133 @@ TEST_F(DBCsppCrashSafeTest, ConvertedTableVisibilityFilter) { } } +TEST_F(DBCsppCrashSafeTest, CsppSelfMmapUnmapsWholeFile) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + ASSERT_OK(env_->CreateDirIfMissing(dbname_)); + options.cf_paths = {{dbname_, 0}}; + InternalKeyComparator icmp(options.comparator); + ImmutableOptions ioptions(options); + MutableCFOptions moptions(options); + WriteBufferManager wb(options.db_write_buffer_size); + std::unique_ptr mem(new MemTable( + icmp, ioptions, moptions, &wb, kMaxSequenceNumber, 0, 1)); + ASSERT_OK(mem->Add(1, kTypeValue, "key", "value", nullptr)); + const std::string path = MakeTableFileName(dbname_, 2); + CopyFile(MakeTableFileName(dbname_, 1), path); + mem.reset(); + const int fd = ::open(path.c_str(), O_RDWR); + ASSERT_GE(fd, 0); + terark::DFA_MmapHeader hdr{}; + ASSERT_EQ(::pread(fd, &hdr, sizeof(hdr), 0), + static_cast(sizeof(hdr))); + ASSERT_EQ(::ftruncate(fd, hdr.file_size + 2 * ::sysconf(_SC_PAGESIZE)), 0); + const size_t physical_size = hdr.file_size + 2 * ::sysconf(_SC_PAGESIZE); + const auto original = hdr; + std::vector buffer((physical_size + 7) / 8); + ASSERT_EQ(::pread(fd, buffer.data(), physical_size, 0), + static_cast(physical_size)); + auto check_unmapped = [&] { + std::ifstream maps("/proc/self/maps"); + ASSERT_TRUE(maps.good()); + for (std::string line; std::getline(maps, line);) { + EXPECT_EQ(line.find(path), std::string::npos) << "leaked mapping: " << line; + } + }; + for (int entry = 0; entry < 4; ++entry) { + // self(path), self(fd), load(path), load(fd). + for (int damage = 0; damage < 7; ++damage) { + SCOPED_TRACE(entry); + SCOPED_TRACE(damage); + if (damage == 6 && entry < 2) continue; // load-only format check. + hdr = original; + size_t length = physical_size; + if (damage == 1) hdr.num_blocks = 0; // finish_load_mmap failure. + if (damage == 2) length = 0; + if (damage == 3) length = sizeof(hdr) - 1; + if (damage == 4) hdr.file_size = sizeof(hdr) - 1; + if (damage == 5) hdr.file_size = physical_size + 1; + if (damage == 6) hdr.magic[0] = '!'; + ASSERT_EQ(::ftruncate(fd, physical_size), 0); + ASSERT_EQ(::pwrite(fd, buffer.data(), physical_size, 0), + static_cast(physical_size)); + ASSERT_EQ(::pwrite(fd, &hdr, sizeof(hdr), 0), + static_cast(sizeof(hdr))); + ASSERT_EQ(::ftruncate(fd, length), 0); + auto open = [&] { + if (entry < 2) { + terark::MainPatricia trie(0, 16 << 20, + terark::Patricia::NoWriteReadOnly); + if (entry == 0) trie.self_mmap(path); + else trie.self_mmap(fd, false); + EXPECT_EQ(trie.get_mmap().size(), original.file_size); + } else { + std::unique_ptr trie(entry == 2 + ? terark::BaseDFA::load_mmap(path, false) + : terark::BaseDFA::load_mmap(fd)); + EXPECT_EQ(trie->get_mmap().size(), original.file_size); + } + }; + if (damage == 0) { + ASSERT_NO_THROW(open()); + } else { + try { + open(); + FAIL() << "expected invalid_argument"; + } catch (const std::invalid_argument& ex) { + if ((entry == 0 || entry == 2) && damage >= 2 && damage <= 5) { + EXPECT_NE(std::string(ex.what()).find(path), std::string::npos); + } + } + } + ASSERT_NE(::fcntl(fd, F_GETFD), -1); // Caller retains its descriptor. + check_unmapped(); + } + } + for (bool load : {false, true}) { + for (int damage = 0; damage < 5; ++damage) { + SCOPED_TRACE(load); + SCOPED_TRACE(damage); + auto* header = reinterpret_cast(buffer.data()); + *header = original; + const void* data = buffer.data(); + size_t length = physical_size; + if (damage == 1) { data = nullptr; length = 0; } + if (damage == 2) length = sizeof(original) - 1; + if (damage == 3) header->file_size = sizeof(original) - 1; + if (damage == 4) header->file_size = physical_size + 1; + buffer.back() = 0x87654321; + auto borrow = [&] { + if (load) { + std::unique_ptr trie( + terark::BaseDFA::load_mmap_user_mem(data, length)); + ASSERT_NE(trie, nullptr); + EXPECT_EQ(trie->get_mmap().size(), original.file_size); + } else { + terark::MainPatricia trie(0, 16 << 20, + terark::Patricia::NoWriteReadOnly); + trie.self_mmap_user_mem(data, length); + EXPECT_EQ(trie.get_mmap().size(), original.file_size); + } + }; + if (damage == 0) { + ASSERT_NO_THROW(borrow()); + } else { + ASSERT_THROW(borrow(), std::invalid_argument); + } + // Borrowers neither free the buffer nor alter its logical or extra tail. + ASSERT_EQ(header->file_size, damage == 3 ? sizeof(original) - 1 + : damage == 4 ? physical_size + 1 + : original.file_size); + ASSERT_EQ(buffer.back(), 0x87654321U); + buffer.back() = 0x12345678; + ASSERT_EQ(buffer.back(), 0x12345678U); + } + } + ::close(fd); +} + TEST_F(DBCsppCrashSafeTest, OslRecoveryUnmapsWholeFile) { Close(); Options options = BaseCrashSafeOptions(dbname_, true, false); @@ -686,26 +1568,24 @@ TEST_F(DBCsppCrashSafeTest, OslRecoveryUnmapsWholeFile) { MutableCFOptions moptions(options); WriteBufferManager wb(options.db_write_buffer_size); std::unique_ptr mem(new MemTable( - icmp, ioptions, moptions, &wb, kMaxSequenceNumber, 0)); + icmp, ioptions, moptions, &wb, kMaxSequenceNumber, 0, 1)); ASSERT_OK(mem->Add(1, kTypeValue, "key", "value", nullptr)); - const auto leftovers = ListLeftovers(options, dbname_); - ASSERT_EQ(leftovers.size(), 1U); - const std::string leftover = leftovers[0] + ".review-leak"; - CopyFile(leftovers[0], leftover); + const std::string leftover = MakeTableFileName(dbname_, 2); + CopyFile(MakeTableFileName(dbname_, 1), leftover); mem.reset(); IntTblPropCollectorFactories collectors; TableBuilderOptions tbo(ioptions, moptions, icmp, &collectors, options.compression, options.compression_opts, 0, "default", 0); FileMetaData meta; - meta.fd = FileDescriptor(1, 0, 0); + meta.fd = FileDescriptor(2, 0, 0); meta.fd.smallest_seqno = 0; meta.fd.largest_seqno = 1; ASSERT_OK(options.memtable_factory->RecoverCrashSafeMemTableToSST( leftover, &meta, tbo)); std::ifstream maps("/proc/self/maps"); ASSERT_TRUE(maps.good()); - const auto fname = TableFileName(options.cf_paths, 1, 0); + const auto fname = TableFileName(options.cf_paths, 2, 0); for (std::string line; std::getline(maps, line);) { EXPECT_EQ(line.find(fname), std::string::npos) << "leaked mapping: " << line; } @@ -764,7 +1644,9 @@ TEST_F(DBCsppCrashSafeTest, SecondCrashAfterConvertFailure) { ASSERT_EQ(RunCrashChild(dbname_, "SecondCrashRecover", child_options), 43); PublishedSeqOnDisk rec; ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); - ASSERT_EQ(rec.generation & 1, after_open ? 0U : 1U); + // Recovery does not invalidate a valid published cursor: registered + // source files survive conversion failure and can be retried. + ASSERT_EQ(rec.generation & 1, 0U); ASSERT_OK(TryReopen(options)); EXPECT_EQ(Get("a"), "1"); EXPECT_EQ(Get("b"), "2"); @@ -889,7 +1771,7 @@ TEST_F(DBCsppCrashSafeTest, LogRefRecoveryIgnoresCounters) { ASSERT_EQ(RunCrashChild(dbname_, "LogRefRecoveryIgnoresCounters", config), 42); const auto leftovers = ListLeftovers(options, dbname_); ASSERT_EQ(leftovers.size(), 1U); - const int fd = ::open(leftovers[0].c_str(), O_RDONLY); + const int fd = ::open(leftovers[0].c_str(), O_RDWR); ASSERT_GE(fd, 0); // Header statistics are not a source of truth for WAL references. const size_t offset = config[0] == 'O' @@ -897,21 +1779,27 @@ TEST_F(DBCsppCrashSafeTest, LogRefRecoveryIgnoresCounters) { : offsetof(terark::DFA_MmapHeader, reserved) + 4 * sizeof(uint32_t); uint64_t wal[3]; // fileno, cnt, bytes const ssize_t n = ::pread(fd, wal, sizeof(wal), offset); - ::close(fd); ASSERT_EQ(n, static_cast(sizeof(wal))); + if (config[2] == 'S') { + // Simulate a crash before TLS statistics were flushed to the mapped header. + // A zero approximate count must not discard the real WAL references. + wal[1] = 0; + wal[2] = 0; + ASSERT_EQ(::pwrite(fd, wal, sizeof(wal), offset), + static_cast(sizeof(wal))); + } + ::close(fd); ASSERT_NE(wal[0], 0U); - ASSERT_EQ(wal[1], 0U); - ASSERT_EQ(wal[2], 0U); - uint64_t wal_size = 0; - ASSERT_OK(env_->GetFileSize(LogFileName(options.wal_dir, wal[0]), &wal_size)); for (int reopen = 0; reopen < 2; ++reopen) { ASSERT_OK(TryReopen(options)); ASSERT_GT(CountL0(db_), 0); ColumnFamilyMetaData cf_meta; db_->GetColumnFamilyMetaData(&cf_meta); ASSERT_EQ(cf_meta.blob_files.size(), 1U); - ASSERT_EQ(cf_meta.blob_files[0].total_blob_count, 1U); - ASSERT_EQ(cf_meta.blob_files[0].total_blob_bytes, wal_size); + ASSERT_EQ(cf_meta.blob_files[0].total_blob_count, + std::max(wal[1], 1)); + ASSERT_EQ(cf_meta.blob_files[0].total_blob_bytes, + std::max(wal[2], 1)); ASSERT_EQ(Get("0"), std::string(128, 'v')); ASSERT_EQ(Get("1"), std::string(128, 'v')); ASSERT_EQ(Get("inline"), "v"); @@ -920,6 +1808,73 @@ TEST_F(DBCsppCrashSafeTest, LogRefRecoveryIgnoresCounters) { } } +TEST_F(CrashChild, DISABLED_LogRefRecoveryMultipleWals) { + Options options = LogRefCrashOptions(dbname_, arg_); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + Options auxiliary = options; + auxiliary.memtable_factory = std::make_shared(); + ColumnFamilyHandle* handle = nullptr; + ASSERT_OK(child_db->CreateColumnFamily(auxiliary, "rotate", &handle)); + auto* impl = static_cast(child_db); + auto* cfd = impl->GetVersionSet()->GetColumnFamilySet()->GetColumnFamily( + handle->GetID()); + for (int i = 0; i < 3; ++i) { + ASSERT_OK(child_db->Put(WriteOptions(), std::to_string(i), + std::string(128, 'a' + i))); + if (i != 2) { + ASSERT_OK(impl->TEST_SwitchMemtable(cfd)); + } + } + // Only the auxiliary CF switches: all three WAL slots belong to one primary + // memtable, exercising the parallel mapping array rather than three memtables. + ::_exit(42); +} + +TEST_F(DBCsppCrashSafeTest, LogRefRecoveryMultipleWals) { + for (const char* config : {"CPS", "CSS", "OPS", "OSS"}) { + SCOPED_TRACE(config); + Close(); + Options options = LogRefCrashOptions(dbname_, config); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "LogRefRecoveryMultipleWals", config), 42); + const auto leftovers = ListLeftovers(options, dbname_); + ASSERT_EQ(leftovers.size(), 1U); + const int fd = ::open(leftovers.front().c_str(), O_RDONLY); + ASSERT_GE(fd, 0); + uint32_t num_wals = 0; + const size_t num_wals_offset = config[0] == 'O' + ? offsetof(terark::OSL_MmapHeader, reserved) + sizeof(uint32_t) + : offsetof(terark::DFA_MmapHeader, reserved) + 2 * sizeof(uint32_t); + ASSERT_EQ(::pread(fd, &num_wals, sizeof(num_wals), + num_wals_offset), + static_cast(sizeof(num_wals))); + ::close(fd); + ASSERT_EQ(num_wals, 3U); + for (int reopen = 0; reopen < 2; ++reopen) { + ASSERT_OK(TryReopenWithColumnFamilies({"default", "rotate"}, options)); + ASSERT_EQ(CountL0(db_), 1); + ColumnFamilyMetaData meta; + db_->GetColumnFamilyMetaData(&meta); + ASSERT_EQ(meta.blob_files.size(), 3U); + std::unique_ptr it(db_->NewIterator(ReadOptions())); + it->SeekToFirst(); + for (int i = 0; i < 3; ++i) { + const std::string value(128, 'a' + i); + ASSERT_EQ(Get(0, std::to_string(i)), value); + ASSERT_TRUE(it->Valid()); + ASSERT_EQ(it->key().ToString(), std::to_string(i)); + ASSERT_EQ(it->value().ToString(), value); + it->Next(); + } + ASSERT_FALSE(it->Valid()); + ASSERT_OK(it->status()); + it.reset(); + Close(); + } + } +} + TEST_F(CrashChild, DISABLED_ChangedFactoryWithOtherCfLeftover) { Options options = BaseCrashSafeOptions(dbname_, true, false); Options other = options; @@ -1083,32 +2038,49 @@ TEST_F(DBCsppCrashSafeTest, CloseConvertsLeftovers) { Destroy(options); ASSERT_EQ(RunCrashChild(dbname_, "CloseConvertsLeftovers", std::to_string(osl) + std::to_string(atomic)), 0); - ASSERT_TRUE(ListLeftovers(options, dbname_).empty()); + ASSERT_LE(ListLeftovers(options, dbname_).size(), 1U); ASSERT_OK(env_->FileExists(CrashSafePubSeqFileName(dbname_))); ASSERT_OK(TryReopen(options)); ASSERT_EQ(Get("k1"), "v1"); ASSERT_EQ(Get("k2"), "v2"); ASSERT_GE(CountL0(db_), 1); Close(); - ASSERT_TRUE(ListLeftovers(options, dbname_).empty()); + ASSERT_LE(ListLeftovers(options, dbname_).size(), 1U); } } } #endif -TEST_F(DBCsppCrashSafeTest, AvoidFlushDuringShutdownLeavesNoLeftoverThenWal) { +TEST_F(DBCsppCrashSafeTest, AvoidFlushDuringShutdownKeepsRegisteredMemTable) { Close(); - Options options = BaseCrashSafeOptions(dbname_, true, false); - options.avoid_flush_during_shutdown = true; - Destroy(options); - ASSERT_OK(TryReopen(options)); - ASSERT_OK(Put("k", "v")); - Close(); - ASSERT_TRUE(ListLeftovers(options, dbname_).empty()); - ASSERT_OK(env_->FileExists(CrashSafePubSeqFileName(dbname_))); - options.avoid_flush_during_shutdown = false; - ASSERT_OK(TryReopen(options)); - ASSERT_EQ(Get("k"), "v"); + for (bool osl : {false, true}) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + options.avoid_flush_during_shutdown = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "v")); + const auto registered = ListLeftovers(options, dbname_); + ASSERT_EQ(registered.size(), 1U); + Close(); + ASSERT_EQ(ListLeftovers(options, dbname_), registered); + for (const auto& path : registered) ASSERT_OK(env_->FileExists(path)); + ASSERT_OK(env_->FileExists(CrashSafePubSeqFileName(dbname_))); + options.avoid_flush_during_shutdown = false; + std::atomic converted{0}; + SyncPoint::GetInstance()->SetCallBack( + "MemTableRep::ConvertToSST:After", [&](void*) { ++converted; }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(options)); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_EQ(converted.load(), 1); + ASSERT_EQ(Get("k"), "v"); + ASSERT_EQ(CountL0(db_), 1); + Close(); + for (const auto& path : ListLeftovers(options, dbname_)) + ASSERT_OK(env_->FileExists(path)); + } } TEST_F(DBCsppCrashSafeTest, AvoidFlushCloseReopenDoesNotProbeWal) { @@ -1151,6 +2123,12 @@ TEST_F(DBCsppCrashSafeTest, FreshSidecarProbesOnlyOlderWal) { ASSERT_EQ(probed.size(), 1U); ASSERT_OK(Put("b", "2")); Close(); + // A normal close retains registered sources, enabling prefix conversion. + // Force full WAL replay here to exercise mixed old/new WAL format probing: + // only WALs older than the fresh sidecar's kind boundary need inspection. + const auto registered = ListLeftovers(on, dbname_); + ASSERT_FALSE(registered.empty()); + ASSERT_OK(env_->DeleteFile(registered.front())); probed.clear(); ASSERT_OK(TryReopen(on)); SyncPoint::GetInstance()->DisableProcessing(); @@ -1194,7 +2172,7 @@ TEST_F(DBCsppCrashSafeTest, ClassicWalToLogIndexWithoutSidecarIsNotSupported) { ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); ASSERT_EQ(rec.wal_offset_kind, 2U); ASSERT_EQ(rec.kind_since_wal, 0U); - ASSERT_EQ(rec.generation & 1, 1U); + ASSERT_EQ(rec.generation & 1, 0U); } // Failed Open must not force KindPrep with the unestablished log-index @@ -1369,7 +2347,7 @@ TEST_F(DBCsppCrashSafeTest, AtomicFlushCloseConverts) { ASSERT_OK(TryReopen(options)); ASSERT_OK(Put("k", "v")); Close(); - ASSERT_TRUE(ListLeftovers(options, dbname_).empty()); + ASSERT_LE(ListLeftovers(options, dbname_).size(), 1U); ASSERT_OK(env_->FileExists(CrashSafePubSeqFileName(dbname_))); ASSERT_OK(TryReopen(options)); ASSERT_EQ(Get("k"), "v"); @@ -2105,17 +3083,17 @@ TEST_F(DBCsppCrashSafeTest, LeftoverOnDbPathNotCfPaths0) { ASSERT_OK(TryReopen(options)); ASSERT_OK(Put("k", "0")); Close(); - ASSERT_TRUE(ListLeftovers(options, dbname_).empty()); + ASSERT_LE(ListLeftovers(options, dbname_).size(), 1U); #if !defined(OS_WIN) ASSERT_EQ(RunCrashChild(dbname_, "LeftoverOnDbPathNotCfPaths0"), 1); - auto leftovers_l0 = ListLeftovers(options, l0); + auto leftovers_l0 = ListLeftovers(options, dbname_); ASSERT_FALSE(leftovers_l0.empty()); ASSERT_OK(TryReopen(options)); ASSERT_EQ(Get("k"), "0"); ASSERT_EQ(Get("pad"), "p"); ASSERT_EQ(Get("x"), "1"); for (const auto& leftover_path : leftovers_l0) { - ASSERT_TRUE(env_->FileExists(leftover_path).IsNotFound()); + ASSERT_OK(env_->FileExists(leftover_path)); } #endif } @@ -2599,7 +3577,7 @@ TEST_F(DBCsppCrashSafeTest, LogIndexOnRecoverOffUsesWal) { ASSERT_EQ(Get("k"), "v"); } -TEST_F(DBCsppCrashSafeTest, ListLeftoversAdvancesFileNumber) { +TEST_F(DBCsppCrashSafeTest, ManifestRegistryIgnoresUnregisteredFiles) { for (bool osl : {false, true}) { SCOPED_TRACE(osl ? "OSL" : "CSPP"); Close(); @@ -2609,27 +3587,24 @@ TEST_F(DBCsppCrashSafeTest, ListLeftoversAdvancesFileNumber) { } Destroy(options); ASSERT_OK(env_->CreateDirIfMissing(dbname_)); - const std::string prefix = dbname_ + (osl ? "/OffsetSkipList-" : "/cspp-"); - const std::string high = prefix + "000100.memtab-0"; - const std::string low = prefix + "000010.memtab-0"; + const std::string high = MakeTableFileName(dbname_, 100); + const std::string low = MakeTableFileName(dbname_, 10); ASSERT_OK(WriteStringToFile(env_, "", high)); ASSERT_OK(WriteStringToFile(env_, "", low)); - ASSERT_EQ(ListLeftovers(options, dbname_).size(), 2U); + ASSERT_TRUE(ListLeftovers(options, dbname_).empty()); // Remove both so collision checks cannot hide a stale counter. ASSERT_OK(env_->DeleteFile(high)); ASSERT_OK(env_->DeleteFile(low)); ASSERT_OK(TryReopen(options)); - ASSERT_EQ(ListLeftovers(options, dbname_), - std::vector{prefix + "000101.memtab-0"}); + ASSERT_EQ(ListLeftovers(options, dbname_).size(), 1U); Close(); - ASSERT_TRUE(ListLeftovers(options, dbname_).empty()); - // Empty and lower-number scans must not move the counter backwards. + const auto before = ListLeftovers(options, dbname_); + // The manifest remains authoritative when unrelated files appear. ASSERT_OK(WriteStringToFile(env_, "", low)); - ASSERT_EQ(ListLeftovers(options, dbname_).size(), 1U); + ASSERT_EQ(ListLeftovers(options, dbname_), before); ASSERT_OK(env_->DeleteFile(low)); ASSERT_OK(TryReopen(options)); - ASSERT_EQ(ListLeftovers(options, dbname_), - std::vector{prefix + "000102.memtab-0"}); + ASSERT_EQ(ListLeftovers(options, dbname_).size(), 1U); Close(); Destroy(options); } @@ -2663,7 +3638,7 @@ TEST_F(DBCsppCrashSafeTest, Allow2pcAloneStillConverts) { ASSERT_OK(Put("k", "v")); ASSERT_OK(dbfull()->TEST_SwitchMemtable()); Close(); - ASSERT_TRUE(ListLeftovers(options, dbname_).empty()); + ASSERT_LE(ListLeftovers(options, dbname_).size(), 1U); ASSERT_OK(TryReopen(options)); ASSERT_EQ(Get("k"), "v"); } @@ -2730,21 +3705,29 @@ TEST_F(CrashChild, DISABLED_LeftoverNoMagicFallsBackToWal) { TEST_F(DBCsppCrashSafeTest, LeftoverNoMagicFallsBackToWal) { Close(); Options options = BaseCrashSafeOptions(dbname_, true, false); - Destroy(options); - ASSERT_EQ(RunCrashChild(dbname_, "LeftoverNoMagicFallsBackToWal"), 1); - auto leftovers = ListLeftovers(options, dbname_); - ASSERT_FALSE(leftovers.empty()); - const int fd = ::open(leftovers[0].c_str(), O_RDWR); - ASSERT_GE(fd, 0); - terark::DFA_MmapHeader hdr{}; - ASSERT_EQ(::pread(fd, &hdr, sizeof(hdr), 0), - static_cast(sizeof(hdr))); - std::memset(hdr.reserved, 0, sizeof(hdr.reserved)); - ASSERT_EQ(::pwrite(fd, &hdr, sizeof(hdr), 0), - static_cast(sizeof(hdr))); - ::close(fd); - ASSERT_OK(TryReopen(options)); - ASSERT_EQ(Get("bad"), "hdr"); + for (const std::string damage : {"empty", "short-header", "missing-magic"}) { + SCOPED_TRACE(damage); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "LeftoverNoMagicFallsBackToWal"), 1); + auto leftovers = ListLeftovers(options, dbname_); + ASSERT_FALSE(leftovers.empty()); + const int fd = ::open(leftovers[0].c_str(), O_RDWR); + ASSERT_GE(fd, 0); + terark::DFA_MmapHeader hdr{}; + if (damage == "missing-magic") { + ASSERT_EQ(::pread(fd, &hdr, sizeof(hdr), 0), + static_cast(sizeof(hdr))); + std::memset(hdr.reserved, 0, sizeof(hdr.reserved)); + ASSERT_EQ(::pwrite(fd, &hdr, sizeof(hdr), 0), + static_cast(sizeof(hdr))); + } else { + ASSERT_EQ(::ftruncate(fd, damage == "empty" ? 0 : sizeof(hdr) - 1), 0); + } + ::close(fd); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("bad"), "hdr"); + Close(); + } } TEST_F(CrashChild, DISABLED_DualLeftoverSecondConvertFails) { @@ -2805,78 +3788,130 @@ TEST_F(DBCsppCrashSafeTest, TruncateInjectFailureFallsBackToWal) { ASSERT_EQ(RunCrashChild(dbname_, "TruncateInjectFailureFallsBackToWal"), 1); SyncPoint::GetInstance()->EnableProcessing(); SyncPoint::GetInstance()->ClearAllCallBacks(); + bool truncate_called = false; SyncPoint::GetInstance()->SetCallBack( - "CrashSafeRecover::Truncate:InjectStatus", [](void* arg) { - *static_cast(arg) = Status::IOError("inject truncate"); + "MemTableRep::ConvertToSST:Truncate", [&truncate_called](void* arg) { + truncate_called = true; + *static_cast(arg) = IOStatus::IOError("inject truncate"); }); ASSERT_OK(TryReopen(options)); + ASSERT_TRUE(truncate_called); ASSERT_EQ(Get("tr"), "ok"); SyncPoint::GetInstance()->DisableProcessing(); SyncPoint::GetInstance()->ClearAllCallBacks(); } -TEST_F(CrashChild, DISABLED_LinkFileInjectFailureFallsBackToWal) { - Options options = BaseCrashSafeOptions(dbname_, true, true); +TEST_F(CrashChild, DISABLED_AfterConvertBeforeAddFileKeepsRegisteredFile) { + ASSERT_EQ(arg_.size(), 3U); + Options options = BaseCrashSafeOptions(dbname_, true, arg_[1] == '1'); + if (arg_[0] == '1') SetupOsl(&options, true); SyncPoint::GetInstance()->SetCallBack( "DBImpl::PersistPublishedSequence:AfterCommit", [](void*) { ::_exit(1); }); SyncPoint::GetInstance()->EnableProcessing(); DB* child_db = nullptr; ASSERT_OK(DB::Open(options, dbname_, &child_db)); - ASSERT_OK(child_db->Put(WriteOptions(), "lk", "ok")); + ASSERT_OK(child_db->Put(WriteOptions(), "or", std::string(128, 'v'))); ::_exit(0); } -TEST_F(DBCsppCrashSafeTest, LinkFileInjectFailureFallsBackToWal) { - Close(); - Options options = BaseCrashSafeOptions(dbname_, true, true); - Destroy(options); - ASSERT_EQ(RunCrashChild(dbname_, "LinkFileInjectFailureFallsBackToWal"), 1); - SyncPoint::GetInstance()->EnableProcessing(); - SyncPoint::GetInstance()->ClearAllCallBacks(); +TEST_F(CrashChild, DISABLED_AfterConvertBeforeAddFileKeepsRegisteredFileRecover) { + ASSERT_EQ(arg_.size(), 3U); + Options options = BaseCrashSafeOptions(dbname_, true, arg_[1] == '1'); + if (arg_[0] == '1') SetupOsl(&options, true); + const char* points[] = { + "MemTableRep::ConvertToSST:Truncate", + "CrashSafeRecover::AfterConvertBeforeAddFile", + "CrashSafeRecover::AfterConvertBeforeAddFile", + "DBImpl::RegisterMemTableFile:AfterLogAndApply"}; + ASSERT_LT(arg_[2] - '0', 4); SyncPoint::GetInstance()->SetCallBack( - "CrashSafeRecover::LinkFile:InjectStatus", [](void* arg) { - *static_cast(arg) = Status::IOError("inject link"); + points[arg_[2] - '0'], + [&](void*) { + if (arg_[2] == '1') { + // Model an incomplete SST tail, without claiming this callback runs + // in the middle of a write. The persisted trie remains intact. + const auto files = ListLeftovers(options, dbname_); + ASSERT_EQ(files.size(), 1U); + const int fd = ::open(files.front().c_str(), O_RDWR); + ASSERT_GE(fd, 0); + uint64_t structure_size = 0; + if (arg_[0] == '1') { + terark::OSL_MmapHeader hdr{}; + ASSERT_EQ(::pread(fd, &hdr, sizeof(hdr), 0), + static_cast(sizeof(hdr))); + structure_size = hdr.mem_used; + } else { + terark::DFA_MmapHeader hdr{}; + ASSERT_EQ(::pread(fd, &hdr, sizeof(hdr), 0), + static_cast(sizeof(hdr))); + structure_size = hdr.file_size; + } + uint64_t size = 0; + ASSERT_OK(options.env->GetFileSize(files.front(), &size)); + ASSERT_GT(size, structure_size); + ASSERT_EQ(::ftruncate(fd, size - 1), 0); + ::close(fd); + } + ::_exit(1); }); - ASSERT_OK(TryReopen(options)); - ASSERT_EQ(Get("lk"), "ok"); - SyncPoint::GetInstance()->DisableProcessing(); - SyncPoint::GetInstance()->ClearAllCallBacks(); -} - -TEST_F(CrashChild, DISABLED_AfterRenameBeforeAddFileFallsBackToWal) { - Options options = BaseCrashSafeOptions(dbname_, true, false); - SyncPoint::GetInstance()->SetCallBack( - "DBImpl::PersistPublishedSequence:AfterCommit", - [](void*) { ::_exit(1); }); - SyncPoint::GetInstance()->EnableProcessing(); - DB* child_db = nullptr; - ASSERT_OK(DB::Open(options, dbname_, &child_db)); - ASSERT_OK(child_db->Put(WriteOptions(), "or", "phan")); - ::_exit(0); -} - -TEST_F(CrashChild, DISABLED_AfterRenameBeforeAddFileFallsBackToWalRecover) { - Options options = BaseCrashSafeOptions(dbname_, true, false); - SyncPoint::GetInstance()->SetCallBack( - "CrashSafeRecover::AfterRenameBeforeAddFile", - [](void*) { ::_exit(1); }); SyncPoint::GetInstance()->EnableProcessing(); DB* recover_db = nullptr; DB::Open(options, dbname_, &recover_db); ::_exit(0); } -TEST_F(DBCsppCrashSafeTest, AfterRenameBeforeAddFileFallsBackToWal) { +TEST_F(DBCsppCrashSafeTest, AfterConvertBeforeAddFileKeepsRegisteredFile) { Close(); - Options options = BaseCrashSafeOptions(dbname_, true, false); - Destroy(options); - ASSERT_EQ(RunCrashChild(dbname_, "AfterRenameBeforeAddFileFallsBackToWal"), 1); - SyncPoint::GetInstance()->EnableProcessing(); - SyncPoint::GetInstance()->ClearAllCallBacks(); - ASSERT_EQ(RunCrashChild(dbname_, "AfterRenameBeforeAddFileFallsBackToWalRecover"), 1); - ASSERT_OK(TryReopen(options)); - ASSERT_EQ(Get("or"), "phan"); + for (bool osl : {false, true}) { + for (bool log_index : {false, true}) { + for (int window = 0; window < 4; ++window) { + const std::string config = std::to_string(osl) + + std::to_string(log_index) + std::to_string(window); + SCOPED_TRACE(config); + Options options = BaseCrashSafeOptions(dbname_, true, log_index); + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, + "AfterConvertBeforeAddFileKeepsRegisteredFile", config), 1); + const auto before = ListLeftovers(options, dbname_); + ASSERT_EQ(before.size(), 1U); + // Before commit, interrupt the same file twice to exercise footer + // replacement, not just a one-time conversion of the original source. + for (int crash = 0; crash < (window == 3 ? 1 : 2); ++crash) { + ASSERT_EQ(RunCrashChild(dbname_, + "AfterConvertBeforeAddFileKeepsRegisteredFileRecover", config), 1); + ASSERT_OK(env_->FileExists(before.front())); + if (window != 3) { + ASSERT_EQ(ListLeftovers(options, dbname_), before); + } + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.generation & 1, 0U); + } + int converted = 0; + SyncPoint::GetInstance()->SetCallBack( + "MemTableRep::ConvertToSST:After", [&](void*) { ++converted; }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(options)); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_EQ(converted, window == 3 ? 0 : 1); + ASSERT_EQ(Get("or"), std::string(128, 'v')); + std::vector files; + db_->GetLiveFilesMetaData(&files); + ASSERT_EQ(files.size(), 1U); + ASSERT_EQ(files.front().level, 0); + ASSERT_EQ(MakeTableFileName(dbname_, files.front().file_number), + before.front()); + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("or"), std::string(128, 'v')); + ASSERT_EQ(CountL0(db_), 1); + Close(); + } + } + } } TEST_F(DBCsppCrashSafeTest, CrashSafeOrLogIndexDisablesWalCompression) { @@ -3143,7 +4178,7 @@ TEST_F(DBCsppCrashSafeTest, WritePreparedFallsBackToWal) { delete txn_db; } -TEST_F(DBCsppCrashSafeTest, AfterRenameCloseSecondFlushInject) { +TEST_F(DBCsppCrashSafeTest, AfterConvertCloseSecondFlushInject) { Close(); Options options = BaseCrashSafeOptions(dbname_, true, false); Destroy(options); @@ -3245,6 +4280,80 @@ TEST_F(DBCsppCrashSafeTest, AtomicFlushDualCfLeftoverConvertsBoth) { ASSERT_GE(CountL0(db_, "one"), 1); } +TEST_F(CrashChild, DISABLED_AtomicFlushManifestWindow) { + ASSERT_EQ(arg_.size(), 2U); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.atomic_flush = true; + if (arg_[0] == '1') SetupOsl(&options, true); + DB* child_db = nullptr; + std::vector handles; + const std::vector cfs = { + {kDefaultColumnFamilyName, options}, {"one", options}}; + ASSERT_OK(DB::Open(options, dbname_, cfs, &handles, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), handles[0], "default-key", "one")); + ASSERT_OK(child_db->Put(WriteOptions(), handles[1], "other-key", "two")); + SyncPoint::GetInstance()->SetCallBack( + arg_[1] == '0' ? "FlushJob::BeforeManifest" : "FlushJob::AfterManifest", + [](void*) { ::_exit(42); }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(child_db->Flush(FlushOptions(), handles)); + ::_exit(1); +} + +TEST_F(DBCsppCrashSafeTest, AtomicFlushCrashAcrossManifestCommit) { + Close(); + for (bool osl : {false, true}) { + for (bool committed : {false, true}) { + SCOPED_TRACE(osl); + SCOPED_TRACE(committed); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.atomic_flush = true; + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_OK(TryReopen(options)); + CreateAndReopenWithCF({"one"}, options); + Close(); + const std::string arg = std::string(osl ? "1" : "0") + + (committed ? "1" : "0"); + ASSERT_EQ(RunCrashChild(dbname_, "AtomicFlushManifestWindow", arg), 42); + const auto registered = ListLeftovers(options, dbname_); + ASSERT_GE(registered.size(), 2U); + for (const auto& path : registered) ASSERT_OK(env_->FileExists(path)); + std::atomic converted{0}; + SyncPoint::GetInstance()->SetCallBack( + "MemTableRep::ConvertToSST:After", + [&](void*) { ++converted; }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopenWithColumnFamilies( + {kDefaultColumnFamilyName, "one"}, options)); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_EQ(converted.load(), committed ? 0 : 2); + ASSERT_EQ(Get(0, "default-key"), "one"); + ASSERT_EQ(Get(1, "other-key"), "two"); + ASSERT_EQ(CountL0(db_, kDefaultColumnFamilyName), 1); + ASSERT_EQ(CountL0(db_, "one"), 1); + std::vector files; + db_->GetLiveFilesMetaData(&files); + ASSERT_EQ(files.size(), 2U); + for (const auto& file : files) { + const std::string path = MakeTableFileName(dbname_, file.file_number); + ASSERT_OK(env_->FileExists(path)); + ASSERT_EQ(std::count(registered.begin(), registered.end(), path), + committed ? 0 : 1); + } + Close(); + ASSERT_OK(TryReopenWithColumnFamilies( + {kDefaultColumnFamilyName, "one"}, options)); + ASSERT_EQ(Get(0, "default-key"), "one"); + ASSERT_EQ(Get(1, "other-key"), "two"); + ASSERT_EQ(CountL0(db_, kDefaultColumnFamilyName), 1); + ASSERT_EQ(CountL0(db_, "one"), 1); + Close(); + } + } +} + TEST_F(CrashChild, DISABLED_DroppedCfLeftoverSkipped) { Options options = BaseCrashSafeOptions(dbname_, true, false); std::atomic pubs{0}; @@ -3278,13 +4387,10 @@ TEST_F(DBCsppCrashSafeTest, DroppedCfLeftoverSkipped) { ASSERT_EQ(Get(0, "keep"), "1"); ASSERT_EQ(Get(1, "drop"), "2"); ASSERT_OK(Flush(0)); - std::string left1; - for (const auto& p : ListLeftovers(options, dbname_)) { - if (p.find(".memtab-1") != std::string::npos) { - left1 = p; - break; - } - } + const auto registered = dbfull()->GetVersionSet()->GetMemTableFiles(); + ASSERT_EQ(registered.count(1), 1U); + ASSERT_FALSE(registered.at(1).empty()); + const std::string left1 = MakeTableFileName(dbname_, *registered.at(1).begin()); ASSERT_FALSE(left1.empty()); const std::string bak = left1 + ".bak"; CopyFile(left1, bak); @@ -3297,13 +4403,9 @@ TEST_F(DBCsppCrashSafeTest, DroppedCfLeftoverSkipped) { } ASSERT_OK(TryReopen(options)); ASSERT_EQ(Get("keep"), "1"); - bool dropped_left = false; - for (const auto& p : ListLeftovers(options, dbname_)) { - if (p.find(".memtab-1") != std::string::npos) { - dropped_left = true; - } - } - ASSERT_TRUE(dropped_left); + ASSERT_EQ(dbfull()->GetVersionSet()->GetMemTableFiles().count(1), 0U); + const auto leftovers = ListLeftovers(options, dbname_); + ASSERT_EQ(std::count(leftovers.begin(), leftovers.end(), left1), 0); } TEST_F(CrashChild, DISABLED_MultiChunkAfterCommitStillReadable) { diff --git a/db/db_impl/db_impl.cc b/db/db_impl/db_impl.cc index 4b139aab1..b0ddc2092 100644 --- a/db/db_impl/db_impl.cc +++ b/db/db_impl/db_impl.cc @@ -4075,6 +4075,11 @@ Status DBImpl::CreateColumnFamilyImpl(const ColumnFamilyOptions& cf_options, // LogAndApply will both write the creation in MANIFEST and create // ColumnFamilyData object + std::unique_ptr::iterator> pending_memtable; + if (cf_options.memtable_factory->SupportCrashSafe()) { + pending_memtable = std::make_unique::iterator>( + CaptureCurrentFileNumberInPendingOutputs()); + } { // write thread WriteThread::Writer w; write_thread_.EnterUnbatched(&w, &mutex_); @@ -4091,7 +4096,20 @@ Status DBImpl::CreateColumnFamilyImpl(const ColumnFamilyOptions& cf_options, assert(cfd != nullptr); std::map> dummy_created_dirs; s = cfd->AddDirectories(&dummy_created_dirs); + if (s.ok()) { + s = RegisterMemTableFile(cfd, cfd->mem()); + if (!s.ok()) { + // CF creation is already committed. Keep its in-memory state valid, + // but do not return a writable handle after registration fails. + InstallSuperVersionAndScheduleWork(cfd, &sv_context, + *cfd->GetLatestMutableCFOptions()); + cfd->set_initialized(); + error_handler_.SetBGError(s, BackgroundErrorReason::kManifestWrite) + .PermitUncheckedError(); + } + } } + ReleaseFileNumberFromPendingOutputs(pending_memtable); if (s.ok()) { auto* cfd = versions_->GetColumnFamilySet()->GetColumnFamily(column_family_name); diff --git a/db/db_impl/db_impl.h b/db/db_impl/db_impl.h index c62f559ee..71fb85804 100644 --- a/db/db_impl/db_impl.h +++ b/db/db_impl/db_impl.h @@ -1437,8 +1437,6 @@ class DBImpl : public DB { // such a file's absolute path to its parent directory. std::unordered_map files_to_delete_; bool is_new_db_ = false; - // Restore the saved WAL cursor only after recovery edits are installed. - bool restore_published_seq_ = false; // WAL tail cursor for this Recover. RecoverLogFiles reads it from here. bool crash_safe_wal_tail_replay_ = false; uint8_t crash_safe_wal_offset_kind_ = 0; @@ -1966,12 +1964,10 @@ class DBImpl : public DB { void StagePublishedWal(SequenceNumber seq, uint64_t wal_number, uint64_t wal_offset); void AccountPendingMemtableWrites(size_t n); + Status RegisterMemTableFile(ColumnFamilyData* cfd, MemTable* mem); bool CanConvertLeftoverForCrashSafeRecover( - const std::vector& leftover_snapshot, - SequenceNumber mmap_pubseq, uint64_t mmap_wal_number, - std::string* fail_reason); + SequenceNumber mmap_pubseq, std::string* fail_reason); Status ConvertLeftoverMemtables( - const std::vector& leftover_snapshot, SequenceNumber max_visible_seq, RecoveryContext* recovery_ctx); // The following two methods are used to flush a memtable to diff --git a/db/db_impl/db_impl_files.cc b/db/db_impl/db_impl_files.cc index 47901f4d2..9d2fed670 100644 --- a/db/db_impl/db_impl_files.cc +++ b/db/db_impl/db_impl_files.cc @@ -175,6 +175,15 @@ void DBImpl::FindObsoleteFiles(JobContext* job_context, bool force, if (doing_the_full_scan) { versions_->AddLiveFiles(&job_context->sst_live, &job_context->blob_live); + // These share the SST namespace, but are still mutable. Protect them from + // GC without exposing them as immutable SSTs to checkpoints and backups. + for (const auto& cf : versions_->GetMemTableFiles()) { + job_context->sst_live.insert(job_context->sst_live.end(), + cf.second.begin(), cf.second.end()); + } + for (auto* cfd : *versions_->GetColumnFamilySet()) { + cfd->AddMemTableCacheFileNumbers(&job_context->sst_live); + } InfoLogPrefix info_log_prefix(!immutable_db_options_.db_log_dir.empty(), dbname_); std::set paths; diff --git a/db/db_impl/db_impl_open.cc b/db/db_impl/db_impl_open.cc index 5665a55ae..3b50df376 100644 --- a/db/db_impl/db_impl_open.cc +++ b/db/db_impl/db_impl_open.cc @@ -543,20 +543,6 @@ bool PeekPublishedWalKind(const std::string& dbname, uint64_t* kind) { return ok; } -uint32_t ParseLeftoverCfId(const std::string& path) { - const auto pos = path.rfind(".memtab-"); - if (pos == std::string::npos) { - return std::numeric_limits::max(); - } - const char* p = path.c_str() + pos + 8; - char* end = nullptr; - const unsigned long id = std::strtoul(p, &end, 10); - if (end == p) { - return std::numeric_limits::max(); - } - return static_cast(id); -} - void DestroyKindPrepDb(DB* db, std::vector* handles) { for (auto* h : *handles) { delete h; @@ -745,13 +731,13 @@ void DBImpl::PersistPublishedSequence(SequenceNumber seq, uint64_t wal_number, } TEST_SYNC_POINT("DBImpl::PersistPublishedSequence:SameSeq"); } - // Recovery may leave an odd generation until the first new publication. + // A failed publication may leave an odd generation. const uint64_t g = pubseq_mmap_->generation | 1; #if defined(__AVX__) // One aligned 32-byte store. A process crash falls between instructions, // so the record is all old or all new. Publish an even generation only - // after this store, including when recovery left it odd. A host without it - // uses the odd/even bracket below after the file is moved. + // after this store, including when a prior publication left it odd. A host + // without it uses the odd/even bracket below after the file is moved. // PublishedSeqRecord next; // next.pubseq = seq; // next.wal_number = wal_number; @@ -820,15 +806,44 @@ void DBImpl::AccountPendingMemtableWrites(size_t n) { } } +Status DBImpl::RegisterMemTableFile(ColumnFamilyData* cfd, MemTable* mem) { + mutex_.AssertHeld(); + if (mem->GetBackingFileNumber() == 0 || mem->IsFileRegistered()) { + return Status::OK(); + } + VersionEdit edit; + edit.SetColumnFamily(cfd->GetID()); + edit.SetMemTableFileTracking(); + edit.AddMemTableFile(mem->GetBackingFileNumber()); + Status s; + TEST_SYNC_POINT_CALLBACK("DBImpl::RegisterMemTableFile:BeforeLogAndApply", &s); + if (s.ok()) { + s = versions_->LogAndApply(cfd, *cfd->GetLatestMutableCFOptions(), + ReadOptions(), &edit, &mutex_, + directories_.GetDbDir()); + } + // A dropped CF's edit can be discarded with OK status. + if (s.ok() && cfd->IsDropped()) { + s = Status::ColumnFamilyDropped(); + } + TEST_SYNC_POINT_CALLBACK("DBImpl::RegisterMemTableFile:AfterLogAndApply", &s); + if (s.ok()) { + mem->MarkFileRegistered(); + TEST_SYNC_POINT("DBImpl::RegisterMemTableFile:BeforeInstall"); + } + return s; +} + bool DBImpl::CanConvertLeftoverForCrashSafeRecover( - const std::vector& leftover_snapshot, - SequenceNumber mmap_pubseq, uint64_t mmap_wal_number, - std::string* fail_reason) { + SequenceNumber mmap_pubseq, std::string* fail_reason) { mutex_.AssertHeld(); auto fail = [&](const std::string& reason) { *fail_reason = reason; return false; }; + if (!versions_->HasMemTableFileTracking()) { + return fail("MANIFEST predates MemTable file tracking"); + } if (immutable_db_options_.wal_filter != nullptr) { return fail("wal_filter requires full WAL replay"); } @@ -838,6 +853,9 @@ bool DBImpl::CanConvertLeftoverForCrashSafeRecover( if (immutable_db_options_.best_efforts_recovery) { return fail("best_efforts_recovery"); } + if (mmap_pubseq == 0 && !versions_->GetMemTableFiles().empty()) { + return fail("CSPUBSEQ pubseq 0 with leftover present"); + } for (auto* cfd : *versions_->GetColumnFamilySet()) { if (cfd->IsDropped()) { continue; @@ -846,90 +864,14 @@ bool DBImpl::CanConvertLeftoverForCrashSafeRecover( if (fac == nullptr || !fac->SupportCrashSafe()) { return fail("CF " + cfd->GetName() + " factory !SupportCrashSafe"); } - if (cfd->mem() != nullptr && !cfd->mem()->SupportConvertToSST()) { - return fail("CF " + cfd->GetName() + " mem !SupportConvertToSST"); - } - if (!cfd->imm()->UnflushedMemtablesSupportConvertToSST()) { - return fail("CF " + cfd->GetName() + " imm !SupportConvertToSST"); - } - } - - // Other tiers may be slow and contain many files; memtables use cf_paths[0]. - std::vector paths; - for (const auto* cfd : *versions_->GetColumnFamilySet()) { - if (!cfd->IsDropped()) { - paths.push_back( - NormalizePath(cfd->ioptions()->cf_paths[0].path + - std::string(1, kFilePathSeparator))); - } - } - std::sort(paths.begin(), paths.end()); - paths.erase(std::unique(paths.begin(), paths.end()), paths.end()); - const uint64_t next_file_number = versions_->current_next_file_number(); - for (const auto& path : paths) { - std::vector files; - const Status ls = env_->GetChildren(path, &files); - if (!ls.ok()) { - continue; - } - for (const auto& fname : files) { - uint64_t number = 0; - FileType type; - if (!ParseFileName(fname, &number, &type)) { - continue; - } - // Only numbers still ahead of MANIFEST next_file (crash mid-Open - // before LogAndApply). A Convert that failed after this Open already - // advanced next_file is invisible here; ConvertLeftover deletes those. - if (type == kTableFile && number >= next_file_number) { - return fail("orphan SST " + path + fname); - } - } - } - - std::unordered_set leftover_cfs; - for (const auto& leftover_path : leftover_snapshot) { - const uint32_t cf_id = ParseLeftoverCfId(leftover_path); - ColumnFamilyData* cfd = - versions_->GetColumnFamilySet()->GetColumnFamily(cf_id); - if (cfd == nullptr) { - ROCKS_LOG_WARN(immutable_db_options_.info_log, - "Crash-safe leftover %s belongs to dropped CF %u, skip", - leftover_path.c_str(), cf_id); - continue; - } - auto* fac = cfd->ioptions()->memtable_factory.get(); - const Status ps = - fac->ProbeCrashSafeLeftover(leftover_path, - immutable_db_options_.GetWalDir()); - if (!ps.ok()) { - return fail("probe " + leftover_path + ": " + ps.ToString()); - } - leftover_cfs.insert(cf_id); - } - if (mmap_pubseq > versions_->LastSequence() && leftover_cfs.empty()) { - return fail("CSPUBSEQ > MANIFEST LastSequence and no leftover"); - } - if (!leftover_cfs.empty() && mmap_pubseq == 0) { - return fail("CSPUBSEQ pubseq 0 with leftover present"); - } - if (mmap_wal_number != 0) { - for (const auto* cfd : *versions_->GetColumnFamilySet()) { - // Only CFs persisted past this WAL can omit their leftover safely. - if (!cfd->IsDropped() && cfd->GetLogNumber() <= mmap_wal_number && - leftover_cfs.count(cfd->GetID()) == 0) { - return fail("CF " + cfd->GetName() + " has no leftover before WAL cursor"); - } - } } return true; } Status DBImpl::ConvertLeftoverMemtables( - const std::vector& leftover_snapshot, SequenceNumber max_visible_seq, RecoveryContext* recovery_ctx) { mutex_.AssertHeld(); - if (leftover_snapshot.empty()) { + if (versions_->GetMemTableFiles().empty()) { return Status::OK(); } @@ -939,19 +881,23 @@ Status DBImpl::ConvertLeftoverMemtables( std::vector blobs; }; std::vector converted; + // MANIFEST is the complete inventory, including empty, precreated tables. + // A directory scan cannot detect one missing file among several in a CF. + for (const auto& cf : versions_->GetMemTableFiles()) { + auto* cfd = versions_->GetColumnFamilySet()->GetColumnFamily(cf.first); + assert(cfd != nullptr && !cfd->IsDropped()); + for (uint64_t file_num : cf.second) { + auto& one = converted.emplace_back(); + one.cfd = cfd; + one.meta.fd = FileDescriptor(file_num, 0, 0); + } + } Status s; mutex_.Unlock(); - for (const auto& leftover_path : leftover_snapshot) { - const uint32_t cf_id = ParseLeftoverCfId(leftover_path); - ColumnFamilyData* cfd = - versions_->GetColumnFamilySet()->GetColumnFamily(cf_id); - if (cfd == nullptr) { - continue; - } - ConvertedLeftover one; - one.cfd = cfd; - const uint64_t file_num = versions_->NewFileNumber(); - one.meta.fd = FileDescriptor(file_num, 0, 0); + for (auto& one : converted) { + auto* cfd = one.cfd; + const uint64_t file_num = one.meta.fd.GetNumber(); + const auto leftover_path = TableFileName(cfd->ioptions()->cf_paths, file_num, 0); one.meta.fd.smallest_seqno = 0; one.meta.fd.largest_seqno = max_visible_seq; one.meta.epoch_number = cfd->NewEpochNumber(); @@ -968,31 +914,16 @@ Status DBImpl::ConvertLeftoverMemtables( tboptions.add_blob_file = [&one](BlobFileAddition b) { one.blobs.push_back(std::move(b)); }; - Status one_s = - cfd->ioptions()->memtable_factory->RecoverCrashSafeMemTableToSST( - leftover_path, &one.meta, tboptions); - if (!one_s.ok()) { - s = one_s; - // Inject / rename-then-fail still leaves an SST. Include it so the - // cleanup below can unlink it; leftover is already gone. - converted.push_back(std::move(one)); + s = cfd->ioptions()->memtable_factory->RecoverCrashSafeMemTableToSST( + leftover_path, &one.meta, tboptions); + if (!s.ok()) { break; } - converted.push_back(std::move(one)); } mutex_.Lock(); if (!s.ok()) { - // NewFileNumber already ran. After this Open finishes, next_file is - // past these numbers, so DeleteUnreferencedSstFiles (only >= next_file, - // and it runs before Convert) will not collect them. + // Do not retire registered sources until full WAL recovery commits. for (auto& one : converted) { - if (one.meta.fd.GetNumber() == 0) { - continue; - } - env_->DeleteFile(TableFileName(one.cfd->ioptions()->cf_paths, - one.meta.fd.GetNumber(), - one.meta.fd.GetPathId())) - .PermitUncheckedError(); for (const auto& blob : one.blobs) { env_->DeleteFile(BlobFileName(one.cfd->ioptions()->cf_paths.front().path, blob.GetBlobFileNumber())) @@ -1002,13 +933,13 @@ Status DBImpl::ConvertLeftoverMemtables( return s; } for (auto& one : converted) { - if (one.meta.fd.GetFileSize() == 0) { - continue; - } VersionEdit edit; edit.SetColumnFamily(one.cfd->GetID()); - one.meta.marked_for_compaction = true; - edit.AddFile(0, one.meta); + edit.DeleteMemTableFile(one.meta.fd.GetNumber()); + if (one.meta.fd.GetFileSize() != 0) { + one.meta.marked_for_compaction = true; + edit.AddFile(0, one.meta); + } for (const auto& blob : one.blobs) { edit.AddBlobFile(blob); } @@ -1022,24 +953,12 @@ Status DBImpl::Recover( bool error_if_wal_file_exists, bool error_if_data_exists_in_wals, uint64_t* recovered_seq, RecoveryContext* recovery_ctx) { mutex_.AssertHeld(); - - std::vector leftover_snapshot; - if (immutable_db_options_.memtable_crash_safe_recover && !read_only) { - for (const auto& desc : column_families) { - if (!desc.options.memtable_factory) { - continue; + if (read_only) { + for (const auto& cf : column_families) { + if (cf.options.memtable_factory->SupportCrashSafe()) { + return Status::InvalidArgument( + "FileMmap memtable is not supported in read-only mode", cf.name); } - const auto& paths = desc.options.cf_paths.empty() - ? immutable_db_options_.db_paths - : desc.options.cf_paths; - desc.options.memtable_factory->ListCrashSafeLeftovers( - paths[0].path, &leftover_snapshot); - } - if (!leftover_snapshot.empty()) { - std::sort(leftover_snapshot.begin(), leftover_snapshot.end()); - leftover_snapshot.erase( - std::unique(leftover_snapshot.begin(), leftover_snapshot.end()), - leftover_snapshot.end()); } } @@ -1289,7 +1208,6 @@ Status DBImpl::Recover( s = SetupDBId(read_only, recovery_ctx); ROCKS_LOG_INFO(immutable_db_options_.info_log, "DB ID: %s\n", db_id_.c_str()); bool crash_safe_convert = false; - bool recovered_wals_ok = false; PublishedSeqRecord mmap_rec; bool mmap_valid = false; if (s.ok() && !read_only && @@ -1298,7 +1216,7 @@ Status DBImpl::Recover( std::string fail_reason; if (mmap_valid) { crash_safe_convert = CanConvertLeftoverForCrashSafeRecover( - leftover_snapshot, mmap_rec.pubseq, mmap_rec.wal_number, &fail_reason); + mmap_rec.pubseq, &fail_reason); } else { fail_reason = "invalid CSPUBSEQ generation"; } @@ -1307,26 +1225,23 @@ Status DBImpl::Recover( "Crash-safe recover check failed (%s), fallback to full " "WAL RecoverLogFiles", fail_reason.c_str()); - for (const auto& leftover_path : leftover_snapshot) { - ROCKS_LOG_WARN(immutable_db_options_.info_log, - "Crash-safe leftover not converted: %s", - leftover_path.c_str()); - } } } if (s.ok() && !read_only) { - if (pubseq_mmap_ != nullptr) { - // Recovery can consume only part of the leftover set before failing. - // Keep the cursor invalid across a second crash, including one after - // orphan/failed-conversion cleanup has removed the failure evidence. - pubseq_mmap_->generation |= 1; - recovery_ctx->restore_published_seq_ = mmap_valid; - } s = DeleteUnreferencedSstFiles(recovery_ctx); } + if (s.ok()) { + // MANIFEST recovery restores the allocator; orphan discovery then advances + // it past files created before a previous Open could commit. + for (auto* cfd : *versions_->GetColumnFamilySet()) { + if (cfd->mem() == nullptr) { + cfd->CreateNewMemtable(*cfd->GetLatestMutableCFOptions(), + versions_->LastSequence(), !read_only); + } + } + } if (s.ok() && crash_safe_convert) { - const Status cs = ConvertLeftoverMemtables(leftover_snapshot, mmap_rec.pubseq, - recovery_ctx); + const Status cs = ConvertLeftoverMemtables(mmap_rec.pubseq, recovery_ctx); if (!cs.ok()) { ROCKS_LOG_WARN(immutable_db_options_.info_log, "Crash-safe leftover Convert failed (%s), fallback to " @@ -1468,7 +1383,6 @@ Status DBImpl::Recover( bool corrupted_wal_found = false; s = RecoverLogFiles(wals, &next_sequence, read_only, &corrupted_wal_found, recovery_ctx); - recovered_wals_ok = s.ok(); if (corrupted_wal_found && recovered_seq != nullptr) { *recovered_seq = next_sequence; } @@ -1482,13 +1396,17 @@ Status DBImpl::Recover( } } - if (s.ok() && recovered_wals_ok && !crash_safe_convert && - !leftover_snapshot.empty()) { - for (const auto& leftover_path : leftover_snapshot) { - ROCKS_LOG_INFO(immutable_db_options_.info_log, - "Crash-safe leftover deleted after WAL fallback: %s", - leftover_path.c_str()); - env_->DeleteFile(leftover_path).PermitUncheckedError(); + if (s.ok() && !read_only && !crash_safe_convert) { + // Retire the old inventory only together with the successful WAL recovery. + // Missing files are allowed here: their absence is what forced the replay. + for (const auto& cf : versions_->GetMemTableFiles()) { + auto* cfd = versions_->GetColumnFamilySet()->GetColumnFamily(cf.first); + VersionEdit edit; + edit.SetColumnFamily(cf.first); + for (uint64_t number : cf.second) { + edit.DeleteMemTableFile(number); + } + recovery_ctx->UpdateVersionEdits(cfd, edit); } } @@ -1654,9 +1572,41 @@ Status DBImpl::LogAndApplyForRecovery(const RecoveryContext& recovery_ctx) { mutex_.AssertHeld(); assert(versions_->descriptor_log_ == nullptr); const ReadOptions read_options(Env::IOActivity::kDBOpen); - Status s = versions_->LogAndApply( - recovery_ctx.cfds_, recovery_ctx.mutable_cf_opts_, read_options, - recovery_ctx.edit_lists_, &mutex_, directories_.GetDbDir()); + // Recovery retires the old inventory and installs all recovered CFs at once. + // A truncated MANIFEST tail must not install just part of that transition. + uint32_t remaining = 0; + bool tracks_memtable_files = false; + for (const auto& edits : recovery_ctx.edit_lists_) { + remaining += static_cast(edits.size()); + for (const auto* edit : edits) { + tracks_memtable_files |= edit->HasMemTableFileTracking() || + !edit->GetMemTableFileAdditions().empty() || + !edit->GetMemTableFileDeletions().empty(); + } + } + if (tracks_memtable_files && remaining > 1) { + for (const auto& edits : recovery_ctx.edit_lists_) { + for (auto* edit : edits) { + edit->MarkAtomicGroup(--remaining); + } + } + } + Status s; + TEST_SYNC_POINT_CALLBACK("DBImpl::RegisterMemTableFile:BeforeLogAndApply", &s); + if (s.ok()) { + s = versions_->LogAndApply( + recovery_ctx.cfds_, recovery_ctx.mutable_cf_opts_, read_options, + recovery_ctx.edit_lists_, &mutex_, directories_.GetDbDir()); + } + TEST_SYNC_POINT_CALLBACK("DBImpl::RegisterMemTableFile:AfterLogAndApply", &s); + if (s.ok()) { + for (auto* cfd : *versions_->GetColumnFamilySet()) { + if (cfd->mem()->GetBackingFileNumber() != 0) { + cfd->mem()->MarkFileRegistered(); + } + } + TEST_SYNC_POINT("DBImpl::RegisterMemTableFile:BeforeInstall"); + } if (s.ok() && !(recovery_ctx.files_to_delete_.empty())) { mutex_.Unlock(); for (const auto& stale_sst_file : recovery_ctx.files_to_delete_) { @@ -1670,9 +1620,6 @@ Status DBImpl::LogAndApplyForRecovery(const RecoveryContext& recovery_ctx) { } mutex_.Lock(); } - if (s.ok() && recovery_ctx.restore_published_seq_) { - pubseq_mmap_->generation += 1; - } return s; } @@ -2916,6 +2863,18 @@ Status DBImpl::Open(const DBOptions& db_options, const std::string& dbname, impl->pubseq_mmap_->kind_since_wal = impl->logfile_number_; } if (s.ok()) { + for (auto* cfd : *impl->versions_->GetColumnFamilySet()) { + const uint64_t number = cfd->mem()->GetBackingFileNumber(); + if (number != 0 || impl->immutable_db_options_.memtable_crash_safe_recover) { + VersionEdit edit; + edit.SetColumnFamily(cfd->GetID()); + edit.SetMemTableFileTracking(); + if (number != 0) { + edit.AddMemTableFile(number); + } + recovery_ctx.UpdateVersionEdits(cfd, edit); + } + } s = impl->LogAndApplyForRecovery(recovery_ctx); } diff --git a/db/db_impl/db_impl_secondary.cc b/db/db_impl/db_impl_secondary.cc index 8f4d2e641..bfc2fca60 100644 --- a/db/db_impl/db_impl_secondary.cc +++ b/db/db_impl/db_impl_secondary.cc @@ -36,6 +36,12 @@ Status DBImplSecondary::Recover( bool /*error_if_data_exists_in_wals*/, uint64_t*, RecoveryContext* /*recovery_ctx*/) { mutex_.AssertHeld(); + for (const auto& cf : column_families) { + if (cf.options.memtable_factory->SupportCrashSafe()) { + return Status::InvalidArgument( + "FileMmap memtable is not supported in secondary mode", cf.name); + } + } JobContext job_context(0); Status s; @@ -55,6 +61,10 @@ Status DBImplSecondary::Recover( max_total_in_memory_state_ = 0; for (auto cfd : *versions_->GetColumnFamilySet()) { auto* mutable_cf_options = cfd->GetLatestMutableCFOptions(); + if (cfd->mem() == nullptr) { + cfd->CreateNewMemtable(*mutable_cf_options, versions_->LastSequence(), + /*create_file=*/false); + } max_total_in_memory_state_ += mutable_cf_options->write_buffer_size * mutable_cf_options->max_write_buffer_number; } @@ -302,7 +312,8 @@ Status DBImplSecondary::RecoverLogFiles( const MutableCFOptions mutable_cf_options = *cfd->GetLatestMutableCFOptions(); MemTable* new_mem = - cfd->ConstructNewMemtable(mutable_cf_options, seq_of_batch); + cfd->ConstructNewMemtable(mutable_cf_options, seq_of_batch, + /*create_file=*/false); cfd->mem()->SetNextLogNumber(log_number); cfd->mem()->ConstructFragmentedRangeTombstones(); cfd->imm()->Add(cfd->mem(), &job_context->memtables_to_free); diff --git a/db/db_impl/db_impl_write.cc b/db/db_impl/db_impl_write.cc index 2211106a5..9c3d92820 100644 --- a/db/db_impl/db_impl_write.cc +++ b/db/db_impl/db_impl_write.cc @@ -2353,6 +2353,11 @@ Status DBImpl::SwitchMemtable(ColumnFamilyData* cfd, WriteContext* context) { int num_imm_unflushed = cfd->imm()->NumNotFlushed(); const auto preallocate_block_size = GetWalPreallocateBlockSize(mutable_cf_options.write_buffer_size); + std::unique_ptr::iterator> pending_memtable; + if (cfd->ioptions()->memtable_factory->SupportCrashSafe()) { + pending_memtable = std::make_unique::iterator>( + CaptureCurrentFileNumberInPendingOutputs()); + } mutex_.Unlock(); if (creating_new_log) { // TODO: Write buffer size passed in should be max of all CF's instead @@ -2385,6 +2390,10 @@ Status DBImpl::SwitchMemtable(ColumnFamilyData* cfd, WriteContext* context) { assert(log_recycle_files_.front() == recycle_log_number); log_recycle_files_.pop_front(); } + if (s.ok()) { + s = RegisterMemTableFile(cfd, new_mem); + } + ReleaseFileNumberFromPendingOutputs(pending_memtable); if (s.ok() && creating_new_log) { InstrumentedMutexLock l(&log_write_mutex_); assert(new_log != nullptr); @@ -2417,8 +2426,6 @@ Status DBImpl::SwitchMemtable(ColumnFamilyData* cfd, WriteContext* context) { } if (!s.ok()) { - // how do we fail if we're not creating new log? - assert(creating_new_log); delete new_mem; delete new_log; context->superversion_context.new_superversion.reset(); diff --git a/db/db_memtable_convert_test.cc b/db/db_memtable_convert_test.cc index d3d20c436..831a3473f 100644 --- a/db/db_memtable_convert_test.cc +++ b/db/db_memtable_convert_test.cc @@ -2,6 +2,7 @@ // Flush/Close conversion, independently of crash-safe recovery. #include +#include #include #include @@ -10,8 +11,14 @@ #include "db/db_impl/db_impl.h" #include "db/db_test_util.h" +#include "db/memtable.h" +#include "db/version_set.h" +#include "env/env_chroot.h" +#include "file/filename.h" +#include "memory/arena.h" #include "port/stack_trace.h" #include "test_util/sync_point.h" +#include "table/table_builder.h" namespace ROCKSDB_NAMESPACE { @@ -112,6 +119,150 @@ TEST_P(DBMemtableConvertTest, ManualFlushConverts) { Close(); } +TEST_P(DBMemtableConvertTest, FileMmapConversionKeepsFileNumberAndPath) { + if (std::string(std::get<1>(GetParam())) != "kFileMmap") return; + Options options = ConvertOptions(); + DestroyAndReopen(options); + ASSERT_OK(Put("same-file", "value")); + const auto registered = dbfull()->GetVersionSet()->GetMemTableFiles(); + ASSERT_EQ(registered.count(0), 1U); + ASSERT_EQ(registered.at(0).size(), 1U); + const uint64_t number = *registered.at(0).begin(); + const std::string path = MakeTableFileName(dbname_, number); + ASSERT_OK(env_->FileExists(path)); + ObserveConversion(); + ASSERT_OK(db_->Flush(FlushOptions())); + std::vector files; + db_->GetLiveFilesMetaData(&files); + ASSERT_EQ(files.size(), 1U); + ASSERT_EQ(files[0].file_number, number); + ASSERT_OK(env_->FileExists(path)); + const auto& after = dbfull()->GetVersionSet()->GetMemTableFiles(); + auto cf = after.find(0); + ASSERT_TRUE(cf == after.end() || cf->second.count(number) == 0); + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("same-file"), "value"); +} + +#if !defined(OS_WIN) +TEST_P(DBMemtableConvertTest, FileMmapExclusiveCreationAndChrootConversion) { + if (std::string(std::get<1>(GetParam())) != "kFileMmap") return; + Options options = ConvertOptions(); + DestroyAndReopen(options); + Close(); + const std::string logical_path = MakeTableFileName("", 900000); + const std::string physical_path = dbname_ + logical_path; + auto factory = PluginFactorySP::AcquirePlugin( + std::get<0>(GetParam()) ? "OffsetSkipList" : "CSPPMemTab", + {{"mem_cap", 16777216}, {"convert_to_sst", "kFileMmap"}, + {"chroot_dir", dbname_}}, repo_); + InternalKeyComparator icmp(options.comparator); + MemTable::KeyComparator cmp(icmp); + MutableCFOptions moptions(options); + moptions.write_buffer_size = 1 << 20; + Arena arena; + if (std::get<0>(GetParam())) { + const std::string sentinel = "existing SST must remain intact"; + ASSERT_OK(WriteStringToFile(env_, sentinel, physical_path)); + ASSERT_THROW({ + std::unique_ptr collision(factory->CreateMemTableRep( + logical_path, moptions, cmp, &arena, nullptr, nullptr, 0)); + }, std::runtime_error); + std::string after; + ASSERT_OK(ReadFileToString(env_, physical_path, &after)); + ASSERT_EQ(after, sentinel); + ASSERT_OK(env_->DeleteFile(physical_path)); + } + + std::unique_ptr chroot_env(NewChrootEnv(env_, dbname_)); + options.env = chroot_env.get(); + options.cf_paths = {{"", 0}}; + ImmutableOptions ioptions(options); + IntTblPropCollectorFactories collectors; + const std::string cf_name = "default"; + TableBuilderOptions tbo(ioptions, moptions, icmp, &collectors, + options.compression, options.compression_opts, 0, + cf_name, 0); + std::unique_ptr rep(factory->CreateMemTableRep( + logical_path, moptions, cmp, &arena, nullptr, nullptr, 0)); + ASSERT_TRUE(rep->IsFileMmap()); + rep->InitSetMemTableAsLogIndex(false); + rep->MarkFileRegistered(); + ASSERT_TRUE(rep->InsertKeyValue(PackSequenceAndType(1, kTypeValue), "k", "v")); + rep->MarkReadOnly(); + FileMetaData meta; + meta.fd = FileDescriptor(900001, 0, 0); + meta.num_entries = 1; + ASSERT_TRUE(rep->ConvertToSST(&meta, tbo).IsInvalidArgument()); + rep.reset(); + ASSERT_OK(env_->FileExists(physical_path)); + meta.fd = FileDescriptor(900000, 0, 0); + meta.fd.largest_seqno = kMaxSequenceNumber; + ASSERT_OK(factory->RecoverCrashSafeMemTableToSST(logical_path, &meta, tbo)); + ASSERT_GT(meta.fd.GetFileSize(), 0U); + ASSERT_OK(env_->FileExists(physical_path)); + ASSERT_OK(env_->DeleteFile(physical_path)); + + // Retry conversion on the same live rep after a completed footer exists. + rep.reset(factory->CreateMemTableRep( + logical_path, moptions, cmp, &arena, nullptr, nullptr, 0)); + rep->InitSetMemTableAsLogIndex(false); + rep->MarkFileRegistered(); + ASSERT_TRUE(rep->InsertKeyValue(PackSequenceAndType(1, kTypeValue), "k", "v")); + rep->MarkReadOnly(); + meta.fd = FileDescriptor(900000, 0, 0); + ASSERT_OK(rep->ConvertToSST(&meta, tbo)); + const uint64_t first_size = meta.fd.GetFileSize(); + ASSERT_GT(first_size, 0U); + ASSERT_OK(rep->ConvertToSST(&meta, tbo)); + ASSERT_EQ(meta.fd.GetFileSize(), first_size); + uint64_t actual_size = 0; + ASSERT_OK(env_->GetFileSize(physical_path, &actual_size)); + ASSERT_EQ(actual_size, first_size); + std::string value; + rep->GetPIK(ReadOptions(), ParsedInternalKey("k", 1, kTypeValue), &value, + [](void* arg, const MemTableRep::KeyValuePair& kv) { + *static_cast(arg) = kv.value.ToString(); + return false; + }); + ASSERT_EQ(value, "v"); + rep.reset(); + meta.fd = FileDescriptor(900000, 0, 0); + meta.fd.largest_seqno = kMaxSequenceNumber; + ASSERT_OK(factory->RecoverCrashSafeMemTableToSST(logical_path, &meta, tbo)); + ASSERT_OK(env_->DeleteFile(physical_path)); +} +#endif + +TEST_P(DBMemtableConvertTest, FileMmapCloseKeepsEveryFileNumber) { + if (std::string(std::get<1>(GetParam())) != "kFileMmap") return; + Options options = ConvertOptions(); + DestroyAndReopen(options); + ASSERT_OK(Put("head", "1")); + ASSERT_OK(dbfull()->TEST_SwitchMemtable()); + ASSERT_OK(Put("tail", "2")); + const auto registered = dbfull()->GetVersionSet()->GetMemTableFiles(); + ASSERT_EQ(registered.count(0), 1U); + ASSERT_EQ(registered.at(0).size(), 2U); + ObserveConversion(); + ASSERT_OK(db_->Close()); + Close(); + ASSERT_EQ(converts_.load(), 2); + for (uint64_t number : registered.at(0)) { + ASSERT_OK(env_->FileExists(MakeTableFileName(dbname_, number))); + } + ASSERT_OK(TryReopen(options)); + std::vector files; + db_->GetLiveFilesMetaData(&files); + ASSERT_EQ(files.size(), 2U); + for (const auto& file : files) { + ASSERT_EQ(registered.at(0).count(file.file_number), 1U); + } + ASSERT_EQ(Get("head"), "1"); + ASSERT_EQ(Get("tail"), "2"); +} + TEST_P(DBMemtableConvertTest, CloseConvertsAllMemtables) { CheckClose(false); } diff --git a/db/flush_job.cc b/db/flush_job.cc index 9370b1a32..cc24a440a 100644 --- a/db/flush_job.cc +++ b/db/flush_job.cc @@ -209,7 +209,12 @@ void FlushJob::PickMemTable() { edit_->SetColumnFamily(cfd_->GetID()); // path 0 for level 0 file. - meta_.fd = FileDescriptor(versions_->NewFileNumber(), 0, 0); + const uint64_t backing_file_number = + mems_.size() == 1 && m->SupportConvertToSST() + ? m->GetBackingFileNumber() : 0; + meta_.fd = FileDescriptor(backing_file_number != 0 + ? backing_file_number : versions_->NewFileNumber(), + 0, 0); meta_.epoch_number = cfd_->NewEpochNumber(); base_ = cfd_->current(); @@ -232,6 +237,13 @@ Status FlushJob::Run(LogsWithPrepTracker* prep_tracker, FileMetaData* file_meta, if (db_options_.memtable_as_log_index) { mempurge_threshold = 0; // not supported } + for (const auto* mem : mems_) { + if (mem->IsFileRegistered()) { + // File-backed inputs must retire their registration with an SST edit. + mempurge_threshold = 0; + break; + } + } AutoThreadOperationStageUpdater stage_run(ThreadStatus::STAGE_FLUSH_RUN); if (mems_.empty()) { @@ -988,8 +1000,10 @@ Status FlushJob::WriteLevel0Table() { cfd_->GetName().c_str(), job_context_->job_id, meta_.fd.GetNumber(), memtable->ApproximateMemoryUsage(), s.ToString().c_str()); - // Do not turn a failed close conversion into a full table rebuild. - if (flush_reason_ != FlushReason::kShutDown) { + // Rebuilding a file-backed table here would overwrite its registered + // source at the same path. Preserve it for crash recovery. + if (flush_reason_ != FlushReason::kShutDown && + memtable->GetBackingFileNumber() == 0) { goto UseBuildTable; } } else { @@ -1077,10 +1091,17 @@ Status FlushJob::WriteLevel0Table() { } base_->Unref(); - // Note that if file_size is zero, the file has been deleted and - // should not be added to the manifest. + // Zero-sized output does not become an SST in the manifest. const bool has_output = meta_.fd.GetFileSize() > 0; + if (s.ok()) { + for (const auto* mem : mems_) { + if (mem->IsFileRegistered()) { + edit_->DeleteMemTableFile(mem->GetBackingFileNumber()); + } + } + } + if (s.ok() && has_output) { TEST_SYNC_POINT("DBImpl::FlushJob:SSTFileCreated"); // if we have more than 1 background thread, then we cannot diff --git a/db/memtable.cc b/db/memtable.cc index f8e7545bd..5f18a7ba4 100644 --- a/db/memtable.cc +++ b/db/memtable.cc @@ -22,6 +22,7 @@ #include "db/range_tombstone_fragmenter.h" #include "db/read_callback.h" #include "db/wide/wide_column_serialization.h" +#include "file/filename.h" #include "logging/logging.h" #include "memory/arena.h" #include "memory/memory_usage.h" @@ -72,7 +73,8 @@ MemTable::MemTable(const InternalKeyComparator& cmp, const ImmutableOptions& ioptions, const MutableCFOptions& mutable_cf_options, WriteBufferManager* write_buffer_manager, - SequenceNumber latest_seq, uint32_t column_family_id) + SequenceNumber latest_seq, uint32_t column_family_id, + uint64_t backing_file_number) : comparator_(cmp), moptions_(ioptions, mutable_cf_options), refs_(0), @@ -86,7 +88,9 @@ MemTable::MemTable(const InternalKeyComparator& cmp, : nullptr, mutable_cf_options.memtable_huge_page_size), table_(ioptions.memtable_factory->CreateMemTableRep( - ioptions.cf_paths[0].path, // level0_dir + backing_file_number == 0 + ? std::string() + : TableFileName(ioptions.cf_paths, backing_file_number, 0), mutable_cf_options, comparator_, &arena_, mutable_cf_options.prefix_extractor.get(), ioptions.logger, column_family_id)), @@ -106,6 +110,7 @@ MemTable::MemTable(const InternalKeyComparator& cmp, flush_in_progress_(false), flush_completed_(false), file_number_(0), + backing_file_number_(table_->IsFileMmap() ? backing_file_number : 0), first_seqno_(0), earliest_seqno_(latest_seq), creation_seq_(latest_seq), diff --git a/db/memtable.h b/db/memtable.h index e1a69ac26..cf25794e6 100644 --- a/db/memtable.h +++ b/db/memtable.h @@ -154,7 +154,8 @@ class MemTable : public CacheAlignedNewDelete { const ImmutableOptions& ioptions, const MutableCFOptions& mutable_cf_options, WriteBufferManager* write_buffer_manager, - SequenceNumber earliest_seq, uint32_t column_family_id); + SequenceNumber earliest_seq, uint32_t column_family_id, + uint64_t backing_file_number = 0); // No copying allowed MemTable(const MemTable&) = delete; MemTable& operator=(const MemTable&) = delete; @@ -585,6 +586,12 @@ class MemTable : public CacheAlignedNewDelete { void SetFlushCompleted(bool completed) { flush_completed_ = completed; } uint64_t GetFileNumber() const { return file_number_; } + uint64_t GetBackingFileNumber() const { return backing_file_number_; } + bool IsFileRegistered() const { return file_registered_; } + void MarkFileRegistered() { + table_->MarkFileRegistered(); + file_registered_ = true; + } void SetFileNumber(uint64_t file_num) { file_number_ = file_num; } @@ -664,7 +671,9 @@ class MemTable : public CacheAlignedNewDelete { bool needs_user_key_cmp_in_get_; bool support_convert_to_sst_; bool reject_memtable_as_log_index_; + bool file_registered_ = false; uint64_t file_number_; // filled up after flush is complete + uint64_t backing_file_number_; // MANIFEST identity, separate from flush output // The updates to be applied to the transaction log when this // memtable is flushed to storage. diff --git a/db/memtable_list.cc b/db/memtable_list.cc index c02bc5f9e..27aec027d 100644 --- a/db/memtable_list.cc +++ b/db/memtable_list.cc @@ -621,10 +621,16 @@ Status MemTableList::TryInstallMemtableFlushResults( const auto manifest_write_cb = [this, cfd, batch_count, log_buffer, to_delete, mu](const Status& status) { + if (status.ok() && !cfd->IsDropped()) { + cfd->PublishRegisteredMemTableCache(); + TEST_SYNC_POINT("FlushJob::AfterManifest"); + } RemoveMemTablesOrRestoreFlags(status, cfd, batch_count, log_buffer, to_delete, mu); }; if (write_edits) { + cfd->AddPendingMemTableFileEdits(edit_list.front()); + TEST_SYNC_POINT("FlushJob::BeforeManifest"); // this can release and reacquire the mutex. s = vset->LogAndApply(cfd, mutable_cf_options, read_options, edit_list, mu, db_directory, /*new_descriptor_log=*/false, @@ -893,11 +899,13 @@ Status InstallMemtableAtomicFlushResults( autovector> edit_lists; uint32_t num_entries = 0; - for (const auto mems : mems_list) { + for (size_t k = 0; k < mems_list.size(); ++k) { + const auto mems = mems_list[k]; assert(mems != nullptr); autovector edits; assert(!mems->empty()); edits.emplace_back((*mems)[0]->GetEdits()); + cfds[k]->AddPendingMemTableFileEdits(edits.front()); ++num_entries; edit_lists.emplace_back(edits); } @@ -933,11 +941,16 @@ Status InstallMemtableAtomicFlushResults( assert(0 == num_entries); } + TEST_SYNC_POINT("FlushJob::BeforeManifest"); // this can release and reacquire the mutex. s = vset->LogAndApply(cfds, mutable_cf_options_list, read_options, edit_lists, mu, db_directory); for (size_t k = 0; k != cfds.size(); ++k) { + if (s.ok() && !cfds[k]->IsDropped()) { + cfds[k]->PublishRegisteredMemTableCache(); + TEST_SYNC_POINT("FlushJob::AfterManifest"); + } auto* imm = (imm_lists == nullptr) ? cfds[k]->imm() : imm_lists->at(k); imm->InstallNewVersion(); } diff --git a/db/version_edit_handler.cc b/db/version_edit_handler.cc index fb32f4db4..8e2defc11 100644 --- a/db/version_edit_handler.cc +++ b/db/version_edit_handler.cc @@ -496,7 +496,8 @@ ColumnFamilyData* VersionEditHandler::CreateCfAndInit( const ColumnFamilyOptions& cf_options, const VersionEdit& edit) { uint32_t cf_id = edit.GetColumnFamily(); ColumnFamilyData* cfd = - version_set_->CreateColumnFamily(cf_options, read_options_, &edit); + version_set_->CreateColumnFamily(cf_options, read_options_, &edit, + /*create_memtable=*/!cf_options.memtable_factory->SupportCrashSafe()); assert(cfd != nullptr); cfd->set_initialized(); assert(builders_.find(cf_id) == builders_.end()); diff --git a/db/version_set.cc b/db/version_set.cc index ad13602ab..089877d6c 100644 --- a/db/version_set.cc +++ b/db/version_set.cc @@ -7646,7 +7646,7 @@ void VersionSet::ApplyMemTableFileEdit(const VersionEdit& edit) { ColumnFamilyData* VersionSet::CreateColumnFamily( const ColumnFamilyOptions& cf_options, const ReadOptions& read_options, - const VersionEdit* edit) { + const VersionEdit* edit, bool create_memtable) { assert(edit->IsColumnFamilyAdd()); MutableCFOptions dummy_cf_options; @@ -7671,8 +7671,10 @@ ColumnFamilyData* VersionSet::CreateColumnFamily( AppendVersion(new_cfd, v); // GetLatestMutableCFOptions() is safe here without mutex since the // cfd is not available to client - new_cfd->CreateNewMemtable(*new_cfd->GetLatestMutableCFOptions(), + if (create_memtable) { + new_cfd->CreateNewMemtable(*new_cfd->GetLatestMutableCFOptions(), LastSequence()); + } new_cfd->SetLogNumber(edit->GetLogNumber()); return new_cfd; } diff --git a/db/version_set.h b/db/version_set.h index 987258d1c..8a49959a2 100644 --- a/db/version_set.h +++ b/db/version_set.h @@ -1678,7 +1678,8 @@ class VersionSet { ColumnFamilyData* CreateColumnFamily(const ColumnFamilyOptions& cf_options, const ReadOptions& read_options, - const VersionEdit* edit); + const VersionEdit* edit, + bool create_memtable = true); Status VerifyFileMetadata(const ReadOptions& read_options, ColumnFamilyData* cfd, const std::string& fpath, diff --git a/include/rocksdb/memtablerep.h b/include/rocksdb/memtablerep.h index 9f2ae0de3..965e864a5 100644 --- a/include/rocksdb/memtablerep.h +++ b/include/rocksdb/memtablerep.h @@ -212,6 +212,10 @@ class MemTableRep : public CacheAlignedNewDelete { // of time. Otherwise, RocksDB may be blocked. virtual void MarkFlushed() {} + // Registered backing files are retained for the DB's MANIFEST lifecycle. + virtual void MarkFileRegistered() {} + virtual bool IsFileMmap() const { return false; } + struct KeyValuePair { Slice ukey; uint64_t tag; @@ -360,8 +364,10 @@ class MemTableRepFactory : public Customizable { uint32_t /* column_family_id */) { return CreateMemTableRep(key_cmp, allocator, slice_transform, logger); } + // The DB supplies the final SST path for a file-backed memtable. An empty + // path requests an anonymous in-memory representation. virtual MemTableRep* CreateMemTableRep( - const std::string& /*level0_dir*/, + const std::string& /*memtable_file_path*/, const MutableCFOptions&, const MemTableRep::KeyComparator& key_cmp, Allocator* allocator, const SliceTransform* slice_transform, Logger* logger, @@ -387,19 +393,6 @@ class MemTableRepFactory : public Customizable { // Default: false virtual bool SupportCrashSafe() const { return false; } - // Append leftover crash-safe mmap paths under cf_dir (plus factory chroot). - // Default: no leftovers. - virtual void ListCrashSafeLeftovers(const std::string& /*cf_dir*/, - std::vector* /*leftovers*/) { - } - - // Read-only leftover probe for Recover check. Must not truncate/rename. - // wal_dir is used to verify each WAL fileno can still be opened. - virtual Status ProbeCrashSafeLeftover(const std::string& /*path*/, - const std::string& /*wal_dir*/) const { - return Status::OK(); - } - // Load leftover, cap visible entries, truncate, ConvertToSST. The caller // supplies the visibility bound in meta->fd.largest_seqno; preserve it for // subsequent table readers. From 6faf82c6987e89489e6a7fed19946a000c6d096b Mon Sep 17 00:00:00 2001 From: leipeng Date: Sun, 4 Oct 2026 10:22:41 +0800 Subject: [PATCH 22/58] Revert "Recover registered MemTables with stable DB file numbers" This reverts commit cd79ca7a1fd6915c395b9403726a36fd0f07272a. --- db/column_family.cc | 82 +- db/column_family.h | 13 +- db/db_cspp_crash_safe_test.cc | 1384 ++++--------------------------- db/db_impl/db_impl.cc | 18 - db/db_impl/db_impl.h | 8 +- db/db_impl/db_impl_files.cc | 9 - db/db_impl/db_impl_open.cc | 317 ++++--- db/db_impl/db_impl_secondary.cc | 13 +- db/db_impl/db_impl_write.cc | 11 +- db/db_memtable_convert_test.cc | 151 ---- db/flush_job.cc | 31 +- db/memtable.cc | 9 +- db/memtable.h | 11 +- db/memtable_list.cc | 15 +- db/version_edit_handler.cc | 3 +- db/version_set.cc | 6 +- db/version_set.h | 3 +- include/rocksdb/memtablerep.h | 21 +- 18 files changed, 371 insertions(+), 1734 deletions(-) diff --git a/db/column_family.cc b/db/column_family.cc index c6a828f45..2ea7cb857 100644 --- a/db/column_family.cc +++ b/db/column_family.cc @@ -37,7 +37,6 @@ #include "rocksdb/convenience.h" #include "rocksdb/table.h" #include "table/merging_iterator.h" -#include "test_util/sync_point.h" #include "util/autovector.h" #include "util/cast_util.h" #include "util/compression.h" @@ -1165,14 +1164,7 @@ uint64_t ColumnFamilyData::GetLiveSstFilesSize() const { void ColumnFamilyData::PrepareNewMemtableInBackground( const MutableCFOptions& mutable_cf_options) { -#if defined(ROCKSDB_UNIT_TEST) - bool cache_enabled = false; -#else - bool cache_enabled = true; -#endif - TEST_SYNC_POINT_CALLBACK("ColumnFamilyData::MemTableCache:Enabled", - &cache_enabled); - if (!cache_enabled) return; + #if !defined(ROCKSDB_UNIT_TEST) { std::lock_guard lk(precreated_memtable_mutex_); if (precreated_memtable_list_.full()) { @@ -1181,11 +1173,8 @@ void ColumnFamilyData::PrepareNewMemtableInBackground( } } auto beg = ioptions_.clock->NowNanos(); - // dummy_versions_ remains alive for the lifetime of this CF, unlike current_. - uint64_t number = ioptions_.memtable_factory->SupportCrashSafe() - ? dummy_versions_->version_set()->NewFileNumber() : 0; auto tab = new MemTable(internal_comparator_, ioptions_, mutable_cf_options, - write_buffer_manager_, 0/*earliest_seq*/, id_, number); + write_buffer_manager_, 0/*earliest_seq*/, id_); auto end = ioptions_.clock->NowNanos(); RecordInHistogram(ioptions_.stats, MEMTAB_CONSTRUCT_NANOS, end - beg); { @@ -1202,67 +1191,21 @@ void ColumnFamilyData::PrepareNewMemtableInBackground( "precreated_memtable_list_ is full, discard the newly created memtab"); delete tab; } -} - -void ColumnFamilyData::AddPendingMemTableFileEdits(VersionEdit* edit) { - std::lock_guard lk(precreated_memtable_mutex_); - for (size_t i = 0; i < precreated_memtable_list_.size(); ++i) { - const auto& mem = *(precreated_memtable_list_.begin() + ptrdiff_t(i)); - if (mem->GetBackingFileNumber() != 0 && !mem->IsFileRegistered()) { - edit->AddMemTableFile(mem->GetBackingFileNumber()); - edit->SetMemTableFileTracking(); - } - } -} - -void ColumnFamilyData::PublishRegisteredMemTableCache() { - TEST_SYNC_POINT("FlushJob::MemTableCache:BeforePublish"); - { - std::lock_guard lk(precreated_memtable_mutex_); - const auto& files = dummy_versions_->version_set()->GetMemTableFiles(); - auto iter = files.find(id_); - if (iter != files.end()) { - for (size_t i = 0; i < precreated_memtable_list_.size(); ++i) { - const auto& mem = *(precreated_memtable_list_.begin() + ptrdiff_t(i)); - if (iter->second.count(mem->GetBackingFileNumber())) { - mem->MarkFileRegistered(); - } - } - } - } - TEST_SYNC_POINT("FlushJob::MemTableCache:AfterPublish"); -} - -void ColumnFamilyData::AddMemTableCacheFileNumbers(std::vector* live) { - std::lock_guard lk(precreated_memtable_mutex_); - for (size_t i = 0; i < precreated_memtable_list_.size(); ++i) { - const auto& mem = *(precreated_memtable_list_.begin() + ptrdiff_t(i)); - if (mem->GetBackingFileNumber() != 0) { - live->push_back(mem->GetBackingFileNumber()); - } - } + #endif } MemTable* ColumnFamilyData::ConstructNewMemtable( - const MutableCFOptions& mutable_cf_options, SequenceNumber earliest_seq, - bool create_file) { + const MutableCFOptions& mutable_cf_options, SequenceNumber earliest_seq) { MemTable* tab = nullptr; -#if defined(ROCKSDB_UNIT_TEST) - bool cache_enabled = false; -#else - bool cache_enabled = true; -#endif - TEST_SYNC_POINT_CALLBACK("ColumnFamilyData::MemTableCache:Enabled", - &cache_enabled); - if (cache_enabled && create_file) { + #if !defined(ROCKSDB_UNIT_TEST) + { std::lock_guard lk(precreated_memtable_mutex_); - if (!precreated_memtable_list_.empty() && - (precreated_memtable_list_.front()->GetBackingFileNumber() == 0 || - precreated_memtable_list_.front()->IsFileRegistered())) { + if (!precreated_memtable_list_.empty()) { tab = precreated_memtable_list_.front().release(); precreated_memtable_list_.pop_front(); } } + #endif if (tab) { tab->SetCreationSeq(earliest_seq); tab->SetEarliestSequenceNumber(earliest_seq); @@ -1270,10 +1213,8 @@ MemTable* ColumnFamilyData::ConstructNewMemtable( #if !defined(ROCKSDB_UNIT_TEST) auto beg = ioptions_.clock->NowNanos(); #endif - uint64_t number = create_file && ioptions_.memtable_factory->SupportCrashSafe() - ? dummy_versions_->version_set()->NewFileNumber() : 0; tab = new MemTable(internal_comparator_, ioptions_, mutable_cf_options, - write_buffer_manager_, earliest_seq, id_, number); + write_buffer_manager_, earliest_seq, id_); #if !defined(ROCKSDB_UNIT_TEST) auto end = ioptions_.clock->NowNanos(); RecordInHistogram(ioptions_.stats, MEMTAB_CONSTRUCT_NANOS, end - beg); @@ -1283,12 +1224,11 @@ MemTable* ColumnFamilyData::ConstructNewMemtable( } void ColumnFamilyData::CreateNewMemtable( - const MutableCFOptions& mutable_cf_options, SequenceNumber earliest_seq, - bool create_file) { + const MutableCFOptions& mutable_cf_options, SequenceNumber earliest_seq) { if (mem_ != nullptr) { delete mem_->Unref(); } - SetMemtable(ConstructNewMemtable(mutable_cf_options, earliest_seq, create_file)); + SetMemtable(ConstructNewMemtable(mutable_cf_options, earliest_seq)); mem_->Ref(); } diff --git a/db/column_family.h b/db/column_family.h index c8e923f91..367b94160 100644 --- a/db/column_family.h +++ b/db/column_family.h @@ -37,7 +37,6 @@ namespace ROCKSDB_NAMESPACE { class Version; class VersionSet; -class VersionEdit; class VersionStorageInfo; class MemTable; class MemTableListVersion; @@ -374,18 +373,12 @@ class ColumnFamilyData { uint64_t OldestLogToKeep(); void PrepareNewMemtableInBackground(const MutableCFOptions&); - // DB mutex must be held when registering or publishing cached files. - void AddPendingMemTableFileEdits(VersionEdit* edit); - void AddMemTableCacheFileNumbers(std::vector* live); - void PublishRegisteredMemTableCache(); // See Memtable constructor for explanation of earliest_seq param. MemTable* ConstructNewMemtable(const MutableCFOptions& mutable_cf_options, - SequenceNumber earliest_seq, - bool create_file = true); + SequenceNumber earliest_seq); void CreateNewMemtable(const MutableCFOptions& mutable_cf_options, - SequenceNumber earliest_seq, - bool create_file = true); + SequenceNumber earliest_seq); TableCache* table_cache() const { return table_cache_.get(); } BlobSource* blob_source() const { return blob_source_.get(); } @@ -619,9 +612,11 @@ class ColumnFamilyData { WriteBufferManager* write_buffer_manager_; + #if !defined(ROCKSDB_UNIT_TEST) // precreated_memtable_list_.size() is normally 1 terark::fixed_circular_queue, 4> precreated_memtable_list_; std::mutex precreated_memtable_mutex_; + #endif MemTable* mem_; MemTableList imm_; diff --git a/db/db_cspp_crash_safe_test.cc b/db/db_cspp_crash_safe_test.cc index f724b95fc..aa0268ea4 100644 --- a/db/db_cspp_crash_safe_test.cc +++ b/db/db_cspp_crash_safe_test.cc @@ -10,8 +10,6 @@ #include #include #include -#include -#include #include #include #include @@ -29,7 +27,6 @@ #include "db/log_reader.h" #include "db/log_writer.h" #include "db/pre_release_callback.h" -#include "db/version_set.h" #include "file/filename.h" #include "file/file_util.h" #include "file/sequence_file_reader.h" @@ -38,7 +35,6 @@ #include "port/stack_trace.h" #include "rocksdb/io_status.h" #include "rocksdb/statistics.h" -#include "rocksdb/utilities/checkpoint.h" #include "rocksdb/utilities/transaction_db.h" #include "rocksdb/wal_filter.h" #include "table/get_context.h" @@ -111,74 +107,8 @@ Options BaseCrashSafeOptions(const std::string& dbname, bool recover, std::vector ListLeftovers(const Options& options, const std::string& dir) { std::vector leftovers; - Env* env = options.env ? options.env : Env::Default(); - std::string current; - Status s = env->FileExists(CurrentFileName(dir)); - if (s.IsNotFound()) return leftovers; - EXPECT_OK(s); - if (!s.ok()) return leftovers; - s = ReadFileToString(env, CurrentFileName(dir), ¤t); - if (!s.ok()) { - ADD_FAILURE() << s.ToString(); - return leftovers; - } - EXPECT_FALSE(current.empty()); - if (current.empty()) return leftovers; - if (current.back() == '\n') current.pop_back(); - const std::string manifest = dir + "/" + current; - std::unique_ptr file; - s = env->GetFileSystem()->NewSequentialFile(manifest, FileOptions(), &file, - nullptr); - EXPECT_OK(s); - if (!s.ok()) return leftovers; - struct Reporter : log::Reader::Reporter { - void Corruption(size_t, const Status& status) override { - ADD_FAILURE() << status.ToString(); - } - } reporter; - auto input = std::make_unique(std::move(file), manifest); - log::Reader reader(nullptr, std::move(input), &reporter, true, 0); - std::map> registered; - auto apply = [&](const VersionEdit& edit) { - const uint32_t cf = edit.GetColumnFamily(); - if (edit.IsColumnFamilyDrop()) { - registered.erase(cf); - return; - } - for (uint64_t number : edit.GetMemTableFileDeletions()) { - registered[cf].erase(number); - } - for (uint64_t number : edit.GetMemTableFileAdditions()) { - registered[cf].insert(number); - } - }; - AtomicGroupReadBuffer group; - Slice record; - std::string scratch; - while (reader.ReadRecord(&record, &scratch)) { - VersionEdit edit; - s = edit.DecodeFrom(record); - EXPECT_OK(s); - if (!s.ok()) break; - s = group.AddEdit(&edit); - EXPECT_OK(s); - if (!s.ok()) break; - if (!edit.IsInAtomicGroup()) { - apply(edit); - } else if (group.IsFull()) { - for (const auto& member : group.replay_buffer()) apply(member); - group.Clear(); - } - } - const std::string path = !options.cf_paths.empty() - ? options.cf_paths[0].path - : !options.db_paths.empty() - ? options.db_paths[0].path - : dir; - for (const auto& cf : registered) { - for (uint64_t number : cf.second) { - leftovers.push_back(MakeTableFileName(path, number)); - } + if (options.memtable_factory) { + options.memtable_factory->ListCrashSafeLeftovers(dir, &leftovers); } return leftovers; } @@ -300,695 +230,6 @@ class DBCsppCrashSafeTest : public DBTestBase { : DBTestBase("db_cspp_crash_safe_test", /*env_do_fsync=*/false) {} }; -#if !defined(OS_WIN) -TEST_F(CrashChild, DISABLED_RegisteredMemTables) { - Options options = BaseCrashSafeOptions(dbname_, true, false); - if (arg_ == "osl") SetupOsl(&options, true); - DB* child_db = nullptr; - ASSERT_OK(DB::Open(options, dbname_, &child_db)); - ASSERT_OK(child_db->Put(WriteOptions(), "first", "1")); - ASSERT_OK(static_cast(child_db)->TEST_SwitchMemtable()); - ASSERT_OK(child_db->Put(WriteOptions(), "second", "2")); - ASSERT_OK(static_cast(child_db)->TEST_SwitchMemtable()); - // The registered empty active file must also exist for prefix recovery. - ::_exit(42); -} - -TEST_F(DBCsppCrashSafeTest, MissingRegisteredMemTableUsesFullWal) { - Close(); - for (bool osl : {false, true}) { - for (int missing : {0, 1, 2, 3}) { - SCOPED_TRACE(osl); - SCOPED_TRACE(missing); - Options options = BaseCrashSafeOptions(dbname_, true, false); - if (osl) SetupOsl(&options, true); - Destroy(options); - ASSERT_EQ(RunCrashChild(dbname_, "RegisteredMemTables", - osl ? "osl" : "cspp"), 42); - const auto registered = ListLeftovers(options, dbname_); - ASSERT_EQ(registered.size(), 3U); - if (missing == 3) { - for (const auto& path : registered) ASSERT_OK(env_->DeleteFile(path)); - } else { - ASSERT_OK(env_->DeleteFile(registered[missing])); - } - std::atomic converted{0}; - SyncPoint::GetInstance()->SetCallBack( - "MemTableRep::ConvertToSST:After", - [&](void*) { ++converted; }); - SyncPoint::GetInstance()->EnableProcessing(); - ASSERT_OK(TryReopen(options)); - SyncPoint::GetInstance()->DisableProcessing(); - SyncPoint::GetInstance()->ClearAllCallBacks(); - // Recovery may convert an intact prefix before discovering the missing - // registered file, but must discard that prefix and replay the full WAL. - ASSERT_EQ(converted.load(), missing == 3 ? 0 : missing); - ASSERT_EQ(Get("first"), "1"); - ASSERT_EQ(Get("second"), "2"); - ASSERT_EQ(NumTableFilesAtLevel(0), 0); - ASSERT_OK(Flush()); - ASSERT_EQ(NumTableFilesAtLevel(0), 1); - Close(); - ASSERT_OK(TryReopen(options)); - ASSERT_EQ(Get("first"), "1"); - ASSERT_EQ(Get("second"), "2"); - ASSERT_EQ(NumTableFilesAtLevel(0), 1); - Close(); - } - } -} - -TEST_F(DBCsppCrashSafeTest, FailedRegistrationCannotAcceptWrites) { - Close(); - for (bool osl : {false, true}) { - Options options = BaseCrashSafeOptions(dbname_, true, false); - if (osl) SetupOsl(&options, true); - Destroy(options); - ASSERT_OK(TryReopen(options)); - ASSERT_OK(Put("before", "safe")); - std::atomic failed{0}; - SyncPoint::GetInstance()->SetCallBack( - "DBImpl::RegisterMemTableFile:BeforeLogAndApply", [&](void* p) { - ++failed; - *static_cast(p) = Status::IOError("register injection"); - }); - SyncPoint::GetInstance()->EnableProcessing(); - ASSERT_NOK(dbfull()->TEST_SwitchMemtable()); - ASSERT_GT(failed.load(), 0); - ASSERT_NOK(Put("unregistered", "must-not-commit")); - SyncPoint::GetInstance()->DisableProcessing(); - SyncPoint::GetInstance()->ClearAllCallBacks(); - ASSERT_EQ(Get("before"), "safe"); - Close(); - ASSERT_OK(TryReopen(options)); - ASSERT_EQ(Get("before"), "safe"); - ASSERT_EQ(Get("unregistered"), "NOT_FOUND"); - Close(); - } -} - -TEST_F(DBCsppCrashSafeTest, FailedInitialRegistrationClearsDbPointer) { - Close(); - for (bool osl : {false, true}) { - Options options = BaseCrashSafeOptions(dbname_, true, false); - if (osl) SetupOsl(&options, true); - Destroy(options); - std::atomic failed{0}; - SyncPoint::GetInstance()->SetCallBack( - "DBImpl::RegisterMemTableFile:BeforeLogAndApply", [&](void* p) { - ++failed; - *static_cast(p) = Status::IOError("initial register injection"); - }); - SyncPoint::GetInstance()->EnableProcessing(); - DB* opened = nullptr; - const Status status = DB::Open(options, dbname_, &opened); - SyncPoint::GetInstance()->DisableProcessing(); - SyncPoint::GetInstance()->ClearAllCallBacks(); - ASSERT_NOK(status); - ASSERT_GT(failed.load(), 0); - ASSERT_EQ(opened, nullptr); - ASSERT_OK(TryReopen(options)); - ASSERT_OK(Put("after-failure", "safe")); - Close(); - } -} - -TEST_F(CrashChild, DISABLED_RegistrationCommitWindow) { - ASSERT_EQ(arg_.size(), 3U); - Options options = BaseCrashSafeOptions(dbname_, true, false); - if (arg_[0] == '1') SetupOsl(&options, true); - const char* point = arg_[1] == '0' - ? "DBImpl::RegisterMemTableFile:AfterLogAndApply" - : "DBImpl::RegisterMemTableFile:BeforeInstall"; - const auto arm = [&] { - SyncPoint::GetInstance()->SetCallBack(point, [](void*) { ::_exit(42); }); - SyncPoint::GetInstance()->EnableProcessing(); - }; - if (arg_[2] == '0') arm(); - DB* child_db = nullptr; - ASSERT_OK(DB::Open(options, dbname_, &child_db)); - ASSERT_OK(child_db->Put(WriteOptions(), "before-switch", "preserved")); - if (arg_[2] == '1') arm(); - ASSERT_OK(static_cast(child_db)->TEST_SwitchMemtable()); - ::_exit(1); -} - -TEST_F(DBCsppCrashSafeTest, RegistrationCommitCrashKeepsManifestInventory) { - Close(); - for (bool osl : {false, true}) { - for (bool marked : {false, true}) { - for (bool switching : {false, true}) { - SCOPED_TRACE(osl); - SCOPED_TRACE(marked); - SCOPED_TRACE(switching); - Options options = BaseCrashSafeOptions(dbname_, true, false); - if (osl) SetupOsl(&options, true); - Destroy(options); - const std::string arg = std::to_string(osl) + std::to_string(marked) + - std::to_string(switching); - ASSERT_EQ(RunCrashChild(dbname_, "RegistrationCommitWindow", arg), 42); - const auto registered = ListLeftovers(options, dbname_); - ASSERT_EQ(registered.size(), switching ? 2U : 1U); - for (const auto& path : registered) ASSERT_OK(env_->FileExists(path)); - ASSERT_OK(TryReopen(options)); - ASSERT_TRUE(dbfull()->GetVersionSet()->HasMemTableFileTracking()); - ASSERT_EQ(Get("before-switch"), switching ? "preserved" : "NOT_FOUND"); - ASSERT_OK(Put("after-crash", "committed")); - Close(); - for (const auto& path : ListLeftovers(options, dbname_)) - ASSERT_OK(env_->FileExists(path)); - ASSERT_OK(TryReopen(options)); - ASSERT_EQ(Get("before-switch"), switching ? "preserved" : "NOT_FOUND"); - ASSERT_EQ(Get("after-crash"), "committed"); - Close(); - } - } - } -} - -TEST_F(DBCsppCrashSafeTest, FailedNewColumnFamilyRegistrationRemainsReopenable) { - Close(); - for (bool osl : {false, true}) { - Options options = BaseCrashSafeOptions(dbname_, true, false); - if (osl) SetupOsl(&options, true); - Destroy(options); - ASSERT_OK(TryReopen(options)); - ASSERT_OK(Put("existing", "preserved")); - std::atomic failed{0}; - SyncPoint::GetInstance()->SetCallBack( - "DBImpl::RegisterMemTableFile:BeforeLogAndApply", [&](void* p) { - ++failed; - *static_cast(p) = Status::IOError("new CF register injection"); - }); - SyncPoint::GetInstance()->EnableProcessing(); - ColumnFamilyHandle* handle = nullptr; - const Status created = db_->CreateColumnFamily(options, "failed", &handle); - SyncPoint::GetInstance()->DisableProcessing(); - SyncPoint::GetInstance()->ClearAllCallBacks(); - ASSERT_NOK(created); - ASSERT_GT(failed.load(), 0); - ASSERT_EQ(handle, nullptr); - ASSERT_EQ(Get("existing"), "preserved"); - // Closing and reopening also exercises manifest snapshots over this CF. - Close(); - ASSERT_OK(TryReopenWithColumnFamilies({kDefaultColumnFamilyName, "failed"}, - options)); - ASSERT_EQ(Get(0, "existing"), "preserved"); - ASSERT_EQ(Get(1, "existing"), "NOT_FOUND"); - std::unique_ptr iterator(db_->NewIterator(ReadOptions(), handles_[1])); - iterator->SeekToFirst(); - ASSERT_FALSE(iterator->Valid()); - ASSERT_OK(iterator->status()); - iterator.reset(); - Close(); - } -} - -TEST_F(DBCsppCrashSafeTest, FileMmapRejectsReadOnlyAndSecondary) { - Close(); - for (bool osl : {false, true}) { - Options options = BaseCrashSafeOptions(dbname_, true, false); - if (osl) SetupOsl(&options, true); - Destroy(options); - ASSERT_EQ(RunCrashChild(dbname_, "WalFilterFallsBackToWal", - osl ? "10" : "00"), 0); - std::vector before; - ASSERT_OK(env_->GetChildren(dbname_, &before)); - std::sort(before.begin(), before.end()); - for (bool secondary : {false, true}) { - SCOPED_TRACE(osl); - SCOPED_TRACE(secondary); - DB* rejected = nullptr; - const Status s = secondary - ? DB::OpenAsSecondary(options, dbname_, dbname_ + "_secondary", - &rejected) - : DB::OpenForReadOnly(options, dbname_, &rejected); - ASSERT_TRUE(s.IsInvalidArgument()) << s.ToString(); - ASSERT_EQ(rejected, nullptr); - std::vector after; - ASSERT_OK(env_->GetChildren(dbname_, &after)); - std::sort(after.begin(), after.end()); - ASSERT_EQ(after, before); - } - } -} - -TEST_F(DBCsppCrashSafeTest, ReadWriteWalRecoveryFailureDoesNotAbort) { - class CorruptSecondRecord final : public WalFilter { - public: - int calls = 0; - const char* Name() const override { return "CorruptSecondRecord"; } - WalProcessingOption LogRecordFound(unsigned long long, const std::string&, - const WriteBatch&, WriteBatch*, - bool*) override { - return ++calls == 1 ? WalProcessingOption::kContinueProcessing - : WalProcessingOption::kCorruptedRecord; - } - }; - Close(); - for (bool osl : {false, true}) { - Options options = BaseCrashSafeOptions(dbname_, true, false); - if (osl) SetupOsl(&options, true); - Destroy(options); - ASSERT_EQ(RunCrashChild(dbname_, "WalFilterFallsBackToWal", - osl ? "10" : "00"), 0); - auto list_sst = [&] { - std::vector children; - EXPECT_OK(env_->GetChildren(dbname_, &children)); - std::vector files; - for (const auto& child : children) { - if (child.size() >= 4 && child.compare(child.size() - 4, 4, ".sst") == 0) - files.push_back(child); - } - std::sort(files.begin(), files.end()); - return files; - }; - const auto before = list_sst(); - ASSERT_FALSE(before.empty()); - CorruptSecondRecord filter; - options.memtable_crash_safe_recover = false; - options.wal_filter = &filter; - options.wal_recovery_mode = WALRecoveryMode::kAbsoluteConsistency; - DB* failed = nullptr; - const Status status = DB::Open(options, dbname_, &failed); - ASSERT_TRUE(status.IsCorruption()) << status.ToString(); - ASSERT_EQ(failed, nullptr); - ASSERT_EQ(filter.calls, 2); - ASSERT_EQ(list_sst(), before); - options.wal_filter = nullptr; - options.memtable_crash_safe_recover = true; - ASSERT_OK(TryReopen(options)); - ASSERT_EQ(Get("a"), std::string(128, 'a')); - ASSERT_EQ(Get("b"), std::string(128, 'b')); - Close(); - } -} - -TEST_F(DBCsppCrashSafeTest, ManifestRolloverPreservesMemTableRegistry) { - Close(); - for (bool osl : {false, true}) { - Options options = BaseCrashSafeOptions(dbname_, true, false); - options.max_manifest_file_size = 1; - if (osl) SetupOsl(&options, true); - Destroy(options); - ASSERT_OK(TryReopen(options)); - ASSERT_OK(Put("first", "1")); - std::string before; - ASSERT_OK(ReadFileToString(env_, CurrentFileName(dbname_), &before)); - ASSERT_OK(dbfull()->TEST_SwitchMemtable()); - ASSERT_OK(Put("second", "2")); - std::string after; - ASSERT_OK(ReadFileToString(env_, CurrentFileName(dbname_), &after)); - ASSERT_NE(before, after); - const auto registered = dbfull()->GetVersionSet()->GetMemTableFiles(); - ASSERT_EQ(registered.count(0), 1U); - ASSERT_EQ(registered.at(0).size(), 2U); - const auto disk = ListLeftovers(options, dbname_); - ASSERT_EQ(disk.size(), 2U); - for (uint64_t number : registered.at(0)) { - const auto path = MakeTableFileName(dbname_, number); - ASSERT_EQ(std::count(disk.begin(), disk.end(), path), 1); - ASSERT_OK(env_->FileExists(path)); - } - Close(); - ASSERT_OK(TryReopen(options)); - ASSERT_EQ(Get("first"), "1"); - ASSERT_EQ(Get("second"), "2"); - Close(); - } -} - -TEST_F(CrashChild, DISABLED_LegacyManifestWal) { - Options options = BaseCrashSafeOptions(dbname_, false, false); - options.memtable_factory = std::make_shared(); - options.table_factory = Options().table_factory; - DB* child_db = nullptr; - ASSERT_OK(DB::Open(options, dbname_, &child_db)); - WriteOptions write; - write.sync = true; - ASSERT_OK(child_db->Put(write, "legacy-first", "one")); - ASSERT_OK(child_db->Put(write, "legacy-second", "two")); - ::_exit(42); -} - -TEST_F(DBCsppCrashSafeTest, LegacyManifestWithoutTrackingUsesFullWal) { - Close(); - for (bool osl : {false, true}) { - Options options = BaseCrashSafeOptions(dbname_, true, false); - if (osl) SetupOsl(&options, true); - Destroy(options); - ASSERT_EQ(RunCrashChild(dbname_, "LegacyManifestWal"), 42); - ASSERT_TRUE(ListLeftovers(options, dbname_).empty()); - std::vector children; - ASSERT_OK(env_->GetChildren(dbname_, &children)); - uint64_t wal_number = 0; - for (const auto& child : children) { - uint64_t number; - FileType type; - if (ParseFileName(child, &number, &type) && type == kWalFile) - wal_number = std::max(wal_number, number); - } - ASSERT_NE(wal_number, 0U); - uint64_t wal_size = 0; - ASSERT_OK(env_->GetFileSize(LogFileName(dbname_, wal_number), &wal_size)); - ASSERT_GT(wal_size, 0U); - // Valid classic sidecar deliberately points past both keys. An inventory - // without tracking cannot justify skipping this WAL prefix. - PublishedSeqOnDisk record; - record.magic = 0x5145534255505343ULL; - record.version = 1; - record.header_size = sizeof(record); - record.wal_offset_kind = 1; - record.kind_since_wal = static_cast(wal_number); - record.generation = 2; - record.pubseq = 2; - record.wal_number = wal_number; - record.wal_offset = wal_size; - std::string sidecar(4096, '\0'); - std::memcpy(&sidecar[0], &record, sizeof(record)); - ASSERT_OK(WriteStringToFile(env_, sidecar, CrashSafePubSeqFileName(dbname_))); - std::atomic converted{0}; - std::atomic reads{0}; - SyncPoint::GetInstance()->SetCallBack( - "MemTableRep::ConvertToSST:After", [&](void*) { ++converted; }); - SyncPoint::GetInstance()->SetCallBack( - "DBImpl::RecoverLogFiles:BeforeReadWal", [&](void*) { ++reads; }); - SyncPoint::GetInstance()->EnableProcessing(); - ASSERT_OK(TryReopen(options)); - SyncPoint::GetInstance()->DisableProcessing(); - SyncPoint::GetInstance()->ClearAllCallBacks(); - ASSERT_EQ(converted.load(), 0); - ASSERT_GT(reads.load(), 0); - ASSERT_EQ(Get("legacy-first"), "one"); - ASSERT_EQ(Get("legacy-second"), "two"); - ASSERT_TRUE(dbfull()->GetVersionSet()->HasMemTableFileTracking()); - Close(); - ASSERT_OK(TryReopen(options)); - ASSERT_EQ(Get("legacy-first"), "one"); - ASSERT_EQ(Get("legacy-second"), "two"); - Close(); - } -} - -TEST_F(CrashChild, DISABLED_FlushManifestWindow) { - ASSERT_EQ(arg_.size(), 2U); - Options options = BaseCrashSafeOptions(dbname_, true, false); - if (arg_[0] == '1') SetupOsl(&options, true); - DB* child_db = nullptr; - ASSERT_OK(DB::Open(options, dbname_, &child_db)); - ASSERT_OK(child_db->Put(WriteOptions(), "converted", "durable")); - SyncPoint::GetInstance()->SetCallBack( - arg_[1] == '0' ? "FlushJob::BeforeManifest" - : "FlushJob::AfterManifest", - [](void*) { ::_exit(42); }); - SyncPoint::GetInstance()->EnableProcessing(); - ASSERT_OK(child_db->Flush(FlushOptions())); - ::_exit(1); -} - -TEST_F(DBCsppCrashSafeTest, ConversionCrashAcrossManifestCommit) { - Close(); - for (bool osl : {false, true}) { - for (bool committed : {false, true}) { - SCOPED_TRACE(osl); - SCOPED_TRACE(committed); - Options options = BaseCrashSafeOptions(dbname_, true, false); - if (osl) SetupOsl(&options, true); - Destroy(options); - const std::string arg = std::string(osl ? "1" : "0") + - (committed ? "1" : "0"); - ASSERT_EQ(RunCrashChild(dbname_, "FlushManifestWindow", arg), 42); - const auto registered = ListLeftovers(options, dbname_); - ASSERT_FALSE(registered.empty()); - for (const auto& path : registered) ASSERT_OK(env_->FileExists(path)); - std::atomic converts{0}; - SyncPoint::GetInstance()->SetCallBack( - "MemTableRep::ConvertToSST:After", - [&](void*) { ++converts; }); - SyncPoint::GetInstance()->EnableProcessing(); - ASSERT_OK(TryReopen(options)); - SyncPoint::GetInstance()->DisableProcessing(); - SyncPoint::GetInstance()->ClearAllCallBacks(); - ASSERT_EQ(converts.load(), committed ? 0 : 1); - ASSERT_EQ(Get("converted"), "durable"); - ASSERT_EQ(CountL0(db_), 1); - std::vector files; - db_->GetLiveFilesMetaData(&files); - ASSERT_EQ(files.size(), 1U); - if (!committed) { - const std::string path = MakeTableFileName(dbname_, files[0].file_number); - ASSERT_EQ(std::count(registered.begin(), registered.end(), path), 1); - ASSERT_OK(env_->FileExists(path)); - } - Close(); - ASSERT_OK(TryReopen(options)); - ASSERT_EQ(Get("converted"), "durable"); - ASSERT_EQ(CountL0(db_), 1); - Close(); - } - } -} - -TEST_F(DBCsppCrashSafeTest, GarbageCollectionKeepsActiveAndCachedMemTables) { - Close(); - for (bool osl : {false, true}) { - Options options = BaseCrashSafeOptions(dbname_, true, false); - if (osl) SetupOsl(&options, true); - Destroy(options); - SyncPoint::GetInstance()->SetCallBack( - "ColumnFamilyData::MemTableCache:Enabled", - [](void* p) { *static_cast(p) = true; }); - std::atomic published{0}; - SyncPoint::GetInstance()->SetCallBack( - "FlushJob::MemTableCache:BeforePublish", [&](void*) { - const auto& files = dbfull()->GetVersionSet()->GetMemTableFiles(); - std::vector expected; - for (const auto& cf : files) { - for (uint64_t number : cf.second) { - expected.push_back(MakeTableFileName(dbname_, number)); - } - } - ASSERT_EQ(ListLeftovers(options, dbname_), expected); - for (const auto& path : expected) ASSERT_OK(env_->FileExists(path)); - }); - SyncPoint::GetInstance()->SetCallBack( - "FlushJob::MemTableCache:AfterPublish", [&](void*) { ++published; }); - SyncPoint::GetInstance()->EnableProcessing(); - ASSERT_OK(TryReopen(options)); - ASSERT_OK(Put("first", "1")); - ASSERT_OK(Flush()); - ASSERT_GT(published.load(), 0); - ASSERT_OK(Put("active", "2")); - const auto registered = dbfull()->GetVersionSet()->GetMemTableFiles(); - ASSERT_EQ(registered.count(0), 1U); - ASSERT_GE(registered.at(0).size(), 2U); - ASSERT_OK(db_->DisableFileDeletions()); - ASSERT_OK(db_->EnableFileDeletions(true)); - for (uint64_t number : registered.at(0)) { - ASSERT_OK(env_->FileExists(MakeTableFileName(dbname_, number))); - } - ASSERT_EQ(Get("first"), "1"); - ASSERT_EQ(Get("active"), "2"); - Close(); - const auto kept = ListLeftovers(options, dbname_); - ASSERT_FALSE(kept.empty()); - for (const auto& path : kept) ASSERT_OK(env_->FileExists(path)); - SyncPoint::GetInstance()->DisableProcessing(); - SyncPoint::GetInstance()->ClearAllCallBacks(); - ASSERT_OK(TryReopen(options)); - ASSERT_EQ(Get("first"), "1"); - ASSERT_EQ(Get("active"), "2"); - Close(); - } -} - -TEST_F(DBCsppCrashSafeTest, EmptyFlushRetiresSourceFileWithoutFullScan) { - Close(); - for (bool osl : {false, true}) { - Options options = BaseCrashSafeOptions(dbname_, true, false); - options.delete_obsolete_files_period_micros = UINT64_MAX; - if (osl) SetupOsl(&options, true); - Destroy(options); - ASSERT_OK(TryReopen(options)); - const auto source = ListLeftovers(options, dbname_); - ASSERT_EQ(source.size(), 1U); - ASSERT_OK(env_->FileExists(source[0])); - ASSERT_OK(dbfull()->TEST_SwitchMemtable()); - ASSERT_OK(Flush()); - ASSERT_OK(dbfull()->TEST_WaitForBackgroundWork()); - ASSERT_OK(dbfull()->TEST_WaitForPurge()); - std::vector files; - db_->GetLiveFilesMetaData(&files); - ASSERT_TRUE(files.empty()); - ASSERT_TRUE(env_->FileExists(source[0]).IsNotFound()); - const auto active = ListLeftovers(options, dbname_); - ASSERT_EQ(active.size(), 1U); - for (const auto& path : active) ASSERT_OK(env_->FileExists(path)); - ASSERT_OK(Put("converted", "preserved")); - ASSERT_OK(Flush()); - ASSERT_OK(dbfull()->TEST_WaitForBackgroundWork()); - ASSERT_OK(dbfull()->TEST_WaitForPurge()); - db_->GetLiveFilesMetaData(&files); - ASSERT_EQ(files.size(), 1U); - ASSERT_EQ(MakeTableFileName(dbname_, files[0].file_number), active[0]); - ASSERT_OK(env_->FileExists(active[0])); - for (const auto& path : ListLeftovers(options, dbname_)) - ASSERT_OK(env_->FileExists(path)); - ASSERT_EQ(Get("converted"), "preserved"); - Close(); - ASSERT_OK(TryReopen(options)); - ASSERT_EQ(Get("converted"), "preserved"); - Close(); - } -} - -TEST_F(DBCsppCrashSafeTest, DroppedCfRetiresSourceFilesWithoutFullScan) { - Close(); - for (bool osl : {false, true}) { - Options options = BaseCrashSafeOptions(dbname_, true, false); - options.delete_obsolete_files_period_micros = UINT64_MAX; - if (osl) SetupOsl(&options, true); - Destroy(options); - ASSERT_OK(TryReopen(options)); - CreateAndReopenWithCF({"retire"}, options); - ASSERT_OK(Put(0, "keep", "one")); - ASSERT_OK(Put(1, "drop", "two")); - const auto registered = dbfull()->GetVersionSet()->GetMemTableFiles(); - ASSERT_EQ(registered.count(1), 1U); - ASSERT_EQ(registered.at(1).size(), 1U); - const std::string source = MakeTableFileName(dbname_, *registered.at(1).begin()); - ASSERT_OK(env_->FileExists(source)); - ASSERT_OK(db_->DropColumnFamily(handles_[1])); - ASSERT_OK(db_->DestroyColumnFamilyHandle(handles_[1])); - handles_.pop_back(); - // An ordinary flush provides normal obsolete-file GC, without a scan. - ASSERT_OK(Flush(0)); - ASSERT_OK(dbfull()->TEST_WaitForBackgroundWork()); - ASSERT_OK(dbfull()->TEST_WaitForPurge()); - ASSERT_TRUE(env_->FileExists(source).IsNotFound()); - ASSERT_EQ(dbfull()->GetVersionSet()->GetMemTableFiles().count(1), 0U); - for (const auto& path : ListLeftovers(options, dbname_)) - ASSERT_OK(env_->FileExists(path)); - std::vector files; - db_->GetLiveFilesMetaData(&files); - ASSERT_EQ(files.size(), 1U); - ASSERT_OK(env_->FileExists(MakeTableFileName(dbname_, files[0].file_number))); - ASSERT_EQ(Get(0, "keep"), "one"); - Close(); - ASSERT_OK(TryReopen(options)); - ASSERT_EQ(Get("keep"), "one"); - Close(); - } -} - -TEST_F(DBCsppCrashSafeTest, CheckpointDoesNotHardLinkWritableMemTable) { - Close(); - for (bool osl : {false, true}) { - Options options = BaseCrashSafeOptions(dbname_, true, false); - if (osl) SetupOsl(&options, true); - Destroy(options); - ASSERT_OK(TryReopen(options)); - ASSERT_OK(Put("checkpoint", "original")); - const auto registered = dbfull()->GetVersionSet()->GetMemTableFiles(); - ASSERT_EQ(registered.count(0), 1U); - ASSERT_EQ(registered.at(0).size(), 1U); - const uint64_t number = *registered.at(0).begin(); - const std::string checkpoint_dir = dbname_ + ".checkpoint"; - Options copy_options = options; - copy_options.wal_dir = checkpoint_dir; - ASSERT_OK(DestroyDB(checkpoint_dir, copy_options)); - Checkpoint* raw = nullptr; - ASSERT_OK(Checkpoint::Create(db_, &raw)); - std::unique_ptr checkpoint(raw); - ASSERT_OK(checkpoint->CreateCheckpoint(checkpoint_dir, UINT64_MAX)); - // This memtable is still mutable, and hence must not be a live SST. - const auto& after = dbfull()->GetVersionSet()->GetMemTableFiles(); - auto cf = after.find(0); - if (cf != after.end() && cf->second.count(number)) { - ASSERT_TRUE(env_->FileExists(MakeTableFileName(checkpoint_dir, number)) - .IsNotFound()); - } - ASSERT_OK(Put("checkpoint", "source-changed")); - DB* copy_raw = nullptr; - ASSERT_OK(DB::Open(copy_options, checkpoint_dir, ©_raw)); - std::unique_ptr copy(copy_raw); - std::string value; - ASSERT_OK(copy->Get(ReadOptions(), "checkpoint", &value)); - ASSERT_EQ(value, "original"); - copy.reset(); - checkpoint.reset(); - ASSERT_OK(DestroyDB(checkpoint_dir, copy_options)); - Close(); - } -} - -TEST_F(DBCsppCrashSafeTest, FailedCacheRegistrationSurvivesGcAndReopen) { - Close(); - for (bool osl : {false, true}) { - Options options = BaseCrashSafeOptions(dbname_, true, false); - options.max_bgerror_resume_count = 0; - if (osl) SetupOsl(&options, true); - Destroy(options); - SyncPoint::GetInstance()->SetCallBack( - "ColumnFamilyData::MemTableCache:Enabled", - [](void* p) { *static_cast(p) = true; }); - SyncPoint::GetInstance()->EnableProcessing(); - ASSERT_OK(TryReopen(options)); - ASSERT_OK(Put("retry", "preserved")); - std::vector pending; - std::atomic pending_commit{false}; - SyncPoint::GetInstance()->SetCallBack("FlushJob::BeforeManifest", [&](void*) { - std::vector children; - ASSERT_OK(env_->GetChildren(dbname_, &children)); - for (const auto& child : children) { - if (child.size() >= 4 && child.compare(child.size() - 4, 4, ".sst") == 0) - pending.push_back(dbname_ + "/" + child); - } - pending_commit.store(true); - }); - std::atomic injected{false}; - std::atomic published{0}; - SyncPoint::GetInstance()->SetCallBack( - "VersionSet::ProcessManifestWrites:AfterSyncManifest", [&](void* p) { - if (pending_commit.load() && !injected.exchange(true)) - *static_cast(p) = IOStatus::IOError("cache register injection"); - }); - SyncPoint::GetInstance()->SetCallBack( - "FlushJob::MemTableCache:AfterPublish", [&](void*) { ++published; }); - ASSERT_NOK(Flush()); - ASSERT_TRUE(injected.load()); - ASSERT_EQ(published.load(), 0); - ASSERT_GE(pending.size(), 3U); // converting input, active, pending cache - ASSERT_OK(db_->DisableFileDeletions()); - ASSERT_OK(db_->EnableFileDeletions(true)); - for (const auto& path : pending) ASSERT_OK(env_->FileExists(path)); - SyncPoint::GetInstance()->ClearCallBack("FlushJob::BeforeManifest"); - // Plain MANIFEST IOError is fatal under the existing error policy. - // Resume preserves that error; reopening is the supported recovery path. - const Status resumed = db_->Resume(); - ASSERT_TRUE(resumed.IsIOError()); - ASSERT_EQ(Get("retry"), "preserved"); - Close(); - SyncPoint::GetInstance()->ClearCallBack( - "VersionSet::ProcessManifestWrites:AfterSyncManifest"); - ASSERT_OK(TryReopen(options)); - ASSERT_EQ(Get("retry"), "preserved"); - published.store(0); - ASSERT_OK(Put("after-reopen", "committed")); - ASSERT_OK(Flush()); - ASSERT_GT(published.load(), 0); - ASSERT_EQ(Get("retry"), "preserved"); - Close(); - SyncPoint::GetInstance()->DisableProcessing(); - SyncPoint::GetInstance()->ClearAllCallBacks(); - ASSERT_OK(TryReopen(options)); - ASSERT_EQ(Get("retry"), "preserved"); - ASSERT_EQ(Get("after-reopen"), "committed"); - Close(); - } -} -#endif - TEST_F(DBCsppCrashSafeTest, CrashSafeRequiresFileMmap) { for (const char* cls : {"CSPPMemTab", "OffsetSkipList"}) { for (const char* mode : {"kDontConvert", "kDumpMem", "kFileMmap"}) { @@ -1114,13 +355,15 @@ TEST_F(DBCsppCrashSafeTest, RecoveredTableUsesFileSequenceBound) { MutableCFOptions moptions(options); WriteBufferManager wb(options.db_write_buffer_size); std::unique_ptr mem(new MemTable( - icmp, ioptions, moptions, &wb, kMaxSequenceNumber, 0, 1)); + icmp, ioptions, moptions, &wb, kMaxSequenceNumber, 0)); ASSERT_OK(mem->Add(1, kTypeValue, "key", "old", nullptr)); ASSERT_OK(mem->Add(3, kTypeValue, "key", "new", nullptr)); ASSERT_OK(mem->Add(4, kTypeValue, "ghost", "unpublished", nullptr)); mem->MarkImmutable(); - const std::string leftover = MakeTableFileName(dbname_, 2); - CopyFile(MakeTableFileName(dbname_, 1), leftover); + const auto leftovers = ListLeftovers(options, dbname_); + ASSERT_EQ(leftovers.size(), 1U); + const std::string leftover = leftovers[0] + ".recovery"; + CopyFile(leftovers[0], leftover); mem.reset(); IntTblPropCollectorFactories collectors; @@ -1128,7 +371,7 @@ TEST_F(DBCsppCrashSafeTest, RecoveredTableUsesFileSequenceBound) { options.compression, options.compression_opts, 0, "default", 0); FileMetaData meta; - meta.fd = FileDescriptor(2, 0, 0); + meta.fd = FileDescriptor(1, 0, 0); meta.fd.smallest_seqno = 0; // The published bound need not be the sequence of any physical entry. meta.fd.largest_seqno = 2; @@ -1138,7 +381,7 @@ TEST_F(DBCsppCrashSafeTest, RecoveredTableUsesFileSequenceBound) { ASSERT_EQ(meta.fd.smallest_seqno, 0U); ASSERT_EQ(meta.fd.largest_seqno, 2U); - const std::string fname = TableFileName(options.cf_paths, 2, 0); + const std::string fname = TableFileName(options.cf_paths, 1, 0); for (SequenceNumber limit : {meta.fd.largest_seqno, SequenceNumber(4), SequenceNumber(0)}) { SCOPED_TRACE(limit); @@ -1195,7 +438,7 @@ TEST_F(DBCsppCrashSafeTest, RecoveredTableIteratorUsesFileSequenceBound) { MutableCFOptions moptions(options); WriteBufferManager wb(options.db_write_buffer_size); auto mem = std::make_unique( - icmp, ioptions, moptions, &wb, kMaxSequenceNumber, 0, 1); + icmp, ioptions, moptions, &wb, kMaxSequenceNumber, 0); std::vector physical; for (const auto& entry : {std::make_pair("b", 1), {"d", 2}, {"b", 3}, {"d", 5}, {"c", 6}, {"b", 7}, {"a", 8}, @@ -1210,20 +453,22 @@ TEST_F(DBCsppCrashSafeTest, RecoveredTableIteratorUsesFileSequenceBound) { }; std::sort(physical.begin(), physical.end(), less); mem->MarkImmutable(); - const std::string leftover = MakeTableFileName(dbname_, 2); - CopyFile(MakeTableFileName(dbname_, 1), leftover); + const auto leftovers = ListLeftovers(options, dbname_); + ASSERT_EQ(leftovers.size(), 1U); + const std::string leftover = leftovers[0] + ".iterator"; + CopyFile(leftovers[0], leftover); mem.reset(); IntTblPropCollectorFactories collectors; TableBuilderOptions tbo(ioptions, moptions, icmp, &collectors, options.compression, options.compression_opts, 0, "default", 0); FileMetaData meta; - meta.fd = FileDescriptor(2, 0, 0); + meta.fd = FileDescriptor(1, 0, 0); meta.fd.smallest_seqno = 0; meta.fd.largest_seqno = 4; ASSERT_OK(options.memtable_factory->RecoverCrashSafeMemTableToSST( leftover, &meta, tbo)); - const std::string fname = TableFileName(options.cf_paths, 2, 0); + const std::string fname = TableFileName(options.cf_paths, 1, 0); for (SequenceNumber limit : {SequenceNumber(0), SequenceNumber(4), SequenceNumber(9), kMaxSequenceNumber}) { SCOPED_TRACE(limit); @@ -1368,7 +613,7 @@ TEST_F(DBCsppCrashSafeTest, ConvertedTableVisibilityFilter) { MutableCFOptions moptions(options); WriteBufferManager wb(options.db_write_buffer_size); auto mem = std::make_unique( - icmp, ioptions, moptions, &wb, kMaxSequenceNumber, 0, 1); + icmp, ioptions, moptions, &wb, kMaxSequenceNumber, 0); ASSERT_OK(mem->Add(1, kTypeValue, "a", "1", nullptr)); ASSERT_OK(mem->Add(2, kTypeValue, "b", "2", nullptr)); ASSERT_OK(mem->Add(3, kTypeValue, "b", "3", nullptr)); @@ -1429,133 +674,6 @@ TEST_F(DBCsppCrashSafeTest, ConvertedTableVisibilityFilter) { } } -TEST_F(DBCsppCrashSafeTest, CsppSelfMmapUnmapsWholeFile) { - Close(); - Options options = BaseCrashSafeOptions(dbname_, true, false); - Destroy(options); - ASSERT_OK(env_->CreateDirIfMissing(dbname_)); - options.cf_paths = {{dbname_, 0}}; - InternalKeyComparator icmp(options.comparator); - ImmutableOptions ioptions(options); - MutableCFOptions moptions(options); - WriteBufferManager wb(options.db_write_buffer_size); - std::unique_ptr mem(new MemTable( - icmp, ioptions, moptions, &wb, kMaxSequenceNumber, 0, 1)); - ASSERT_OK(mem->Add(1, kTypeValue, "key", "value", nullptr)); - const std::string path = MakeTableFileName(dbname_, 2); - CopyFile(MakeTableFileName(dbname_, 1), path); - mem.reset(); - const int fd = ::open(path.c_str(), O_RDWR); - ASSERT_GE(fd, 0); - terark::DFA_MmapHeader hdr{}; - ASSERT_EQ(::pread(fd, &hdr, sizeof(hdr), 0), - static_cast(sizeof(hdr))); - ASSERT_EQ(::ftruncate(fd, hdr.file_size + 2 * ::sysconf(_SC_PAGESIZE)), 0); - const size_t physical_size = hdr.file_size + 2 * ::sysconf(_SC_PAGESIZE); - const auto original = hdr; - std::vector buffer((physical_size + 7) / 8); - ASSERT_EQ(::pread(fd, buffer.data(), physical_size, 0), - static_cast(physical_size)); - auto check_unmapped = [&] { - std::ifstream maps("/proc/self/maps"); - ASSERT_TRUE(maps.good()); - for (std::string line; std::getline(maps, line);) { - EXPECT_EQ(line.find(path), std::string::npos) << "leaked mapping: " << line; - } - }; - for (int entry = 0; entry < 4; ++entry) { - // self(path), self(fd), load(path), load(fd). - for (int damage = 0; damage < 7; ++damage) { - SCOPED_TRACE(entry); - SCOPED_TRACE(damage); - if (damage == 6 && entry < 2) continue; // load-only format check. - hdr = original; - size_t length = physical_size; - if (damage == 1) hdr.num_blocks = 0; // finish_load_mmap failure. - if (damage == 2) length = 0; - if (damage == 3) length = sizeof(hdr) - 1; - if (damage == 4) hdr.file_size = sizeof(hdr) - 1; - if (damage == 5) hdr.file_size = physical_size + 1; - if (damage == 6) hdr.magic[0] = '!'; - ASSERT_EQ(::ftruncate(fd, physical_size), 0); - ASSERT_EQ(::pwrite(fd, buffer.data(), physical_size, 0), - static_cast(physical_size)); - ASSERT_EQ(::pwrite(fd, &hdr, sizeof(hdr), 0), - static_cast(sizeof(hdr))); - ASSERT_EQ(::ftruncate(fd, length), 0); - auto open = [&] { - if (entry < 2) { - terark::MainPatricia trie(0, 16 << 20, - terark::Patricia::NoWriteReadOnly); - if (entry == 0) trie.self_mmap(path); - else trie.self_mmap(fd, false); - EXPECT_EQ(trie.get_mmap().size(), original.file_size); - } else { - std::unique_ptr trie(entry == 2 - ? terark::BaseDFA::load_mmap(path, false) - : terark::BaseDFA::load_mmap(fd)); - EXPECT_EQ(trie->get_mmap().size(), original.file_size); - } - }; - if (damage == 0) { - ASSERT_NO_THROW(open()); - } else { - try { - open(); - FAIL() << "expected invalid_argument"; - } catch (const std::invalid_argument& ex) { - if ((entry == 0 || entry == 2) && damage >= 2 && damage <= 5) { - EXPECT_NE(std::string(ex.what()).find(path), std::string::npos); - } - } - } - ASSERT_NE(::fcntl(fd, F_GETFD), -1); // Caller retains its descriptor. - check_unmapped(); - } - } - for (bool load : {false, true}) { - for (int damage = 0; damage < 5; ++damage) { - SCOPED_TRACE(load); - SCOPED_TRACE(damage); - auto* header = reinterpret_cast(buffer.data()); - *header = original; - const void* data = buffer.data(); - size_t length = physical_size; - if (damage == 1) { data = nullptr; length = 0; } - if (damage == 2) length = sizeof(original) - 1; - if (damage == 3) header->file_size = sizeof(original) - 1; - if (damage == 4) header->file_size = physical_size + 1; - buffer.back() = 0x87654321; - auto borrow = [&] { - if (load) { - std::unique_ptr trie( - terark::BaseDFA::load_mmap_user_mem(data, length)); - ASSERT_NE(trie, nullptr); - EXPECT_EQ(trie->get_mmap().size(), original.file_size); - } else { - terark::MainPatricia trie(0, 16 << 20, - terark::Patricia::NoWriteReadOnly); - trie.self_mmap_user_mem(data, length); - EXPECT_EQ(trie.get_mmap().size(), original.file_size); - } - }; - if (damage == 0) { - ASSERT_NO_THROW(borrow()); - } else { - ASSERT_THROW(borrow(), std::invalid_argument); - } - // Borrowers neither free the buffer nor alter its logical or extra tail. - ASSERT_EQ(header->file_size, damage == 3 ? sizeof(original) - 1 - : damage == 4 ? physical_size + 1 - : original.file_size); - ASSERT_EQ(buffer.back(), 0x87654321U); - buffer.back() = 0x12345678; - ASSERT_EQ(buffer.back(), 0x12345678U); - } - } - ::close(fd); -} - TEST_F(DBCsppCrashSafeTest, OslRecoveryUnmapsWholeFile) { Close(); Options options = BaseCrashSafeOptions(dbname_, true, false); @@ -1568,24 +686,26 @@ TEST_F(DBCsppCrashSafeTest, OslRecoveryUnmapsWholeFile) { MutableCFOptions moptions(options); WriteBufferManager wb(options.db_write_buffer_size); std::unique_ptr mem(new MemTable( - icmp, ioptions, moptions, &wb, kMaxSequenceNumber, 0, 1)); + icmp, ioptions, moptions, &wb, kMaxSequenceNumber, 0)); ASSERT_OK(mem->Add(1, kTypeValue, "key", "value", nullptr)); - const std::string leftover = MakeTableFileName(dbname_, 2); - CopyFile(MakeTableFileName(dbname_, 1), leftover); + const auto leftovers = ListLeftovers(options, dbname_); + ASSERT_EQ(leftovers.size(), 1U); + const std::string leftover = leftovers[0] + ".review-leak"; + CopyFile(leftovers[0], leftover); mem.reset(); IntTblPropCollectorFactories collectors; TableBuilderOptions tbo(ioptions, moptions, icmp, &collectors, options.compression, options.compression_opts, 0, "default", 0); FileMetaData meta; - meta.fd = FileDescriptor(2, 0, 0); + meta.fd = FileDescriptor(1, 0, 0); meta.fd.smallest_seqno = 0; meta.fd.largest_seqno = 1; ASSERT_OK(options.memtable_factory->RecoverCrashSafeMemTableToSST( leftover, &meta, tbo)); std::ifstream maps("/proc/self/maps"); ASSERT_TRUE(maps.good()); - const auto fname = TableFileName(options.cf_paths, 2, 0); + const auto fname = TableFileName(options.cf_paths, 1, 0); for (std::string line; std::getline(maps, line);) { EXPECT_EQ(line.find(fname), std::string::npos) << "leaked mapping: " << line; } @@ -1644,9 +764,7 @@ TEST_F(DBCsppCrashSafeTest, SecondCrashAfterConvertFailure) { ASSERT_EQ(RunCrashChild(dbname_, "SecondCrashRecover", child_options), 43); PublishedSeqOnDisk rec; ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); - // Recovery does not invalidate a valid published cursor: registered - // source files survive conversion failure and can be retried. - ASSERT_EQ(rec.generation & 1, 0U); + ASSERT_EQ(rec.generation & 1, after_open ? 0U : 1U); ASSERT_OK(TryReopen(options)); EXPECT_EQ(Get("a"), "1"); EXPECT_EQ(Get("b"), "2"); @@ -1771,7 +889,7 @@ TEST_F(DBCsppCrashSafeTest, LogRefRecoveryIgnoresCounters) { ASSERT_EQ(RunCrashChild(dbname_, "LogRefRecoveryIgnoresCounters", config), 42); const auto leftovers = ListLeftovers(options, dbname_); ASSERT_EQ(leftovers.size(), 1U); - const int fd = ::open(leftovers[0].c_str(), O_RDWR); + const int fd = ::open(leftovers[0].c_str(), O_RDONLY); ASSERT_GE(fd, 0); // Header statistics are not a source of truth for WAL references. const size_t offset = config[0] == 'O' @@ -1779,27 +897,21 @@ TEST_F(DBCsppCrashSafeTest, LogRefRecoveryIgnoresCounters) { : offsetof(terark::DFA_MmapHeader, reserved) + 4 * sizeof(uint32_t); uint64_t wal[3]; // fileno, cnt, bytes const ssize_t n = ::pread(fd, wal, sizeof(wal), offset); - ASSERT_EQ(n, static_cast(sizeof(wal))); - if (config[2] == 'S') { - // Simulate a crash before TLS statistics were flushed to the mapped header. - // A zero approximate count must not discard the real WAL references. - wal[1] = 0; - wal[2] = 0; - ASSERT_EQ(::pwrite(fd, wal, sizeof(wal), offset), - static_cast(sizeof(wal))); - } ::close(fd); + ASSERT_EQ(n, static_cast(sizeof(wal))); ASSERT_NE(wal[0], 0U); + ASSERT_EQ(wal[1], 0U); + ASSERT_EQ(wal[2], 0U); + uint64_t wal_size = 0; + ASSERT_OK(env_->GetFileSize(LogFileName(options.wal_dir, wal[0]), &wal_size)); for (int reopen = 0; reopen < 2; ++reopen) { ASSERT_OK(TryReopen(options)); ASSERT_GT(CountL0(db_), 0); ColumnFamilyMetaData cf_meta; db_->GetColumnFamilyMetaData(&cf_meta); ASSERT_EQ(cf_meta.blob_files.size(), 1U); - ASSERT_EQ(cf_meta.blob_files[0].total_blob_count, - std::max(wal[1], 1)); - ASSERT_EQ(cf_meta.blob_files[0].total_blob_bytes, - std::max(wal[2], 1)); + ASSERT_EQ(cf_meta.blob_files[0].total_blob_count, 1U); + ASSERT_EQ(cf_meta.blob_files[0].total_blob_bytes, wal_size); ASSERT_EQ(Get("0"), std::string(128, 'v')); ASSERT_EQ(Get("1"), std::string(128, 'v')); ASSERT_EQ(Get("inline"), "v"); @@ -1808,73 +920,6 @@ TEST_F(DBCsppCrashSafeTest, LogRefRecoveryIgnoresCounters) { } } -TEST_F(CrashChild, DISABLED_LogRefRecoveryMultipleWals) { - Options options = LogRefCrashOptions(dbname_, arg_); - DB* child_db = nullptr; - ASSERT_OK(DB::Open(options, dbname_, &child_db)); - Options auxiliary = options; - auxiliary.memtable_factory = std::make_shared(); - ColumnFamilyHandle* handle = nullptr; - ASSERT_OK(child_db->CreateColumnFamily(auxiliary, "rotate", &handle)); - auto* impl = static_cast(child_db); - auto* cfd = impl->GetVersionSet()->GetColumnFamilySet()->GetColumnFamily( - handle->GetID()); - for (int i = 0; i < 3; ++i) { - ASSERT_OK(child_db->Put(WriteOptions(), std::to_string(i), - std::string(128, 'a' + i))); - if (i != 2) { - ASSERT_OK(impl->TEST_SwitchMemtable(cfd)); - } - } - // Only the auxiliary CF switches: all three WAL slots belong to one primary - // memtable, exercising the parallel mapping array rather than three memtables. - ::_exit(42); -} - -TEST_F(DBCsppCrashSafeTest, LogRefRecoveryMultipleWals) { - for (const char* config : {"CPS", "CSS", "OPS", "OSS"}) { - SCOPED_TRACE(config); - Close(); - Options options = LogRefCrashOptions(dbname_, config); - Destroy(options); - ASSERT_EQ(RunCrashChild(dbname_, "LogRefRecoveryMultipleWals", config), 42); - const auto leftovers = ListLeftovers(options, dbname_); - ASSERT_EQ(leftovers.size(), 1U); - const int fd = ::open(leftovers.front().c_str(), O_RDONLY); - ASSERT_GE(fd, 0); - uint32_t num_wals = 0; - const size_t num_wals_offset = config[0] == 'O' - ? offsetof(terark::OSL_MmapHeader, reserved) + sizeof(uint32_t) - : offsetof(terark::DFA_MmapHeader, reserved) + 2 * sizeof(uint32_t); - ASSERT_EQ(::pread(fd, &num_wals, sizeof(num_wals), - num_wals_offset), - static_cast(sizeof(num_wals))); - ::close(fd); - ASSERT_EQ(num_wals, 3U); - for (int reopen = 0; reopen < 2; ++reopen) { - ASSERT_OK(TryReopenWithColumnFamilies({"default", "rotate"}, options)); - ASSERT_EQ(CountL0(db_), 1); - ColumnFamilyMetaData meta; - db_->GetColumnFamilyMetaData(&meta); - ASSERT_EQ(meta.blob_files.size(), 3U); - std::unique_ptr it(db_->NewIterator(ReadOptions())); - it->SeekToFirst(); - for (int i = 0; i < 3; ++i) { - const std::string value(128, 'a' + i); - ASSERT_EQ(Get(0, std::to_string(i)), value); - ASSERT_TRUE(it->Valid()); - ASSERT_EQ(it->key().ToString(), std::to_string(i)); - ASSERT_EQ(it->value().ToString(), value); - it->Next(); - } - ASSERT_FALSE(it->Valid()); - ASSERT_OK(it->status()); - it.reset(); - Close(); - } - } -} - TEST_F(CrashChild, DISABLED_ChangedFactoryWithOtherCfLeftover) { Options options = BaseCrashSafeOptions(dbname_, true, false); Options other = options; @@ -2038,49 +1083,32 @@ TEST_F(DBCsppCrashSafeTest, CloseConvertsLeftovers) { Destroy(options); ASSERT_EQ(RunCrashChild(dbname_, "CloseConvertsLeftovers", std::to_string(osl) + std::to_string(atomic)), 0); - ASSERT_LE(ListLeftovers(options, dbname_).size(), 1U); + ASSERT_TRUE(ListLeftovers(options, dbname_).empty()); ASSERT_OK(env_->FileExists(CrashSafePubSeqFileName(dbname_))); ASSERT_OK(TryReopen(options)); ASSERT_EQ(Get("k1"), "v1"); ASSERT_EQ(Get("k2"), "v2"); ASSERT_GE(CountL0(db_), 1); Close(); - ASSERT_LE(ListLeftovers(options, dbname_).size(), 1U); + ASSERT_TRUE(ListLeftovers(options, dbname_).empty()); } } } #endif -TEST_F(DBCsppCrashSafeTest, AvoidFlushDuringShutdownKeepsRegisteredMemTable) { +TEST_F(DBCsppCrashSafeTest, AvoidFlushDuringShutdownLeavesNoLeftoverThenWal) { Close(); - for (bool osl : {false, true}) { - Options options = BaseCrashSafeOptions(dbname_, true, false); - if (osl) SetupOsl(&options, true); - options.avoid_flush_during_shutdown = true; - Destroy(options); - ASSERT_OK(TryReopen(options)); - ASSERT_OK(Put("k", "v")); - const auto registered = ListLeftovers(options, dbname_); - ASSERT_EQ(registered.size(), 1U); - Close(); - ASSERT_EQ(ListLeftovers(options, dbname_), registered); - for (const auto& path : registered) ASSERT_OK(env_->FileExists(path)); - ASSERT_OK(env_->FileExists(CrashSafePubSeqFileName(dbname_))); - options.avoid_flush_during_shutdown = false; - std::atomic converted{0}; - SyncPoint::GetInstance()->SetCallBack( - "MemTableRep::ConvertToSST:After", [&](void*) { ++converted; }); - SyncPoint::GetInstance()->EnableProcessing(); - ASSERT_OK(TryReopen(options)); - SyncPoint::GetInstance()->DisableProcessing(); - SyncPoint::GetInstance()->ClearAllCallBacks(); - ASSERT_EQ(converted.load(), 1); - ASSERT_EQ(Get("k"), "v"); - ASSERT_EQ(CountL0(db_), 1); - Close(); - for (const auto& path : ListLeftovers(options, dbname_)) - ASSERT_OK(env_->FileExists(path)); - } + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.avoid_flush_during_shutdown = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "v")); + Close(); + ASSERT_TRUE(ListLeftovers(options, dbname_).empty()); + ASSERT_OK(env_->FileExists(CrashSafePubSeqFileName(dbname_))); + options.avoid_flush_during_shutdown = false; + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("k"), "v"); } TEST_F(DBCsppCrashSafeTest, AvoidFlushCloseReopenDoesNotProbeWal) { @@ -2123,12 +1151,6 @@ TEST_F(DBCsppCrashSafeTest, FreshSidecarProbesOnlyOlderWal) { ASSERT_EQ(probed.size(), 1U); ASSERT_OK(Put("b", "2")); Close(); - // A normal close retains registered sources, enabling prefix conversion. - // Force full WAL replay here to exercise mixed old/new WAL format probing: - // only WALs older than the fresh sidecar's kind boundary need inspection. - const auto registered = ListLeftovers(on, dbname_); - ASSERT_FALSE(registered.empty()); - ASSERT_OK(env_->DeleteFile(registered.front())); probed.clear(); ASSERT_OK(TryReopen(on)); SyncPoint::GetInstance()->DisableProcessing(); @@ -2172,7 +1194,7 @@ TEST_F(DBCsppCrashSafeTest, ClassicWalToLogIndexWithoutSidecarIsNotSupported) { ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); ASSERT_EQ(rec.wal_offset_kind, 2U); ASSERT_EQ(rec.kind_since_wal, 0U); - ASSERT_EQ(rec.generation & 1, 0U); + ASSERT_EQ(rec.generation & 1, 1U); } // Failed Open must not force KindPrep with the unestablished log-index @@ -2347,7 +1369,7 @@ TEST_F(DBCsppCrashSafeTest, AtomicFlushCloseConverts) { ASSERT_OK(TryReopen(options)); ASSERT_OK(Put("k", "v")); Close(); - ASSERT_LE(ListLeftovers(options, dbname_).size(), 1U); + ASSERT_TRUE(ListLeftovers(options, dbname_).empty()); ASSERT_OK(env_->FileExists(CrashSafePubSeqFileName(dbname_))); ASSERT_OK(TryReopen(options)); ASSERT_EQ(Get("k"), "v"); @@ -3083,17 +2105,17 @@ TEST_F(DBCsppCrashSafeTest, LeftoverOnDbPathNotCfPaths0) { ASSERT_OK(TryReopen(options)); ASSERT_OK(Put("k", "0")); Close(); - ASSERT_LE(ListLeftovers(options, dbname_).size(), 1U); + ASSERT_TRUE(ListLeftovers(options, dbname_).empty()); #if !defined(OS_WIN) ASSERT_EQ(RunCrashChild(dbname_, "LeftoverOnDbPathNotCfPaths0"), 1); - auto leftovers_l0 = ListLeftovers(options, dbname_); + auto leftovers_l0 = ListLeftovers(options, l0); ASSERT_FALSE(leftovers_l0.empty()); ASSERT_OK(TryReopen(options)); ASSERT_EQ(Get("k"), "0"); ASSERT_EQ(Get("pad"), "p"); ASSERT_EQ(Get("x"), "1"); for (const auto& leftover_path : leftovers_l0) { - ASSERT_OK(env_->FileExists(leftover_path)); + ASSERT_TRUE(env_->FileExists(leftover_path).IsNotFound()); } #endif } @@ -3577,7 +2599,7 @@ TEST_F(DBCsppCrashSafeTest, LogIndexOnRecoverOffUsesWal) { ASSERT_EQ(Get("k"), "v"); } -TEST_F(DBCsppCrashSafeTest, ManifestRegistryIgnoresUnregisteredFiles) { +TEST_F(DBCsppCrashSafeTest, ListLeftoversAdvancesFileNumber) { for (bool osl : {false, true}) { SCOPED_TRACE(osl ? "OSL" : "CSPP"); Close(); @@ -3587,24 +2609,27 @@ TEST_F(DBCsppCrashSafeTest, ManifestRegistryIgnoresUnregisteredFiles) { } Destroy(options); ASSERT_OK(env_->CreateDirIfMissing(dbname_)); - const std::string high = MakeTableFileName(dbname_, 100); - const std::string low = MakeTableFileName(dbname_, 10); + const std::string prefix = dbname_ + (osl ? "/OffsetSkipList-" : "/cspp-"); + const std::string high = prefix + "000100.memtab-0"; + const std::string low = prefix + "000010.memtab-0"; ASSERT_OK(WriteStringToFile(env_, "", high)); ASSERT_OK(WriteStringToFile(env_, "", low)); - ASSERT_TRUE(ListLeftovers(options, dbname_).empty()); + ASSERT_EQ(ListLeftovers(options, dbname_).size(), 2U); // Remove both so collision checks cannot hide a stale counter. ASSERT_OK(env_->DeleteFile(high)); ASSERT_OK(env_->DeleteFile(low)); ASSERT_OK(TryReopen(options)); - ASSERT_EQ(ListLeftovers(options, dbname_).size(), 1U); + ASSERT_EQ(ListLeftovers(options, dbname_), + std::vector{prefix + "000101.memtab-0"}); Close(); - const auto before = ListLeftovers(options, dbname_); - // The manifest remains authoritative when unrelated files appear. + ASSERT_TRUE(ListLeftovers(options, dbname_).empty()); + // Empty and lower-number scans must not move the counter backwards. ASSERT_OK(WriteStringToFile(env_, "", low)); - ASSERT_EQ(ListLeftovers(options, dbname_), before); + ASSERT_EQ(ListLeftovers(options, dbname_).size(), 1U); ASSERT_OK(env_->DeleteFile(low)); ASSERT_OK(TryReopen(options)); - ASSERT_EQ(ListLeftovers(options, dbname_).size(), 1U); + ASSERT_EQ(ListLeftovers(options, dbname_), + std::vector{prefix + "000102.memtab-0"}); Close(); Destroy(options); } @@ -3638,7 +2663,7 @@ TEST_F(DBCsppCrashSafeTest, Allow2pcAloneStillConverts) { ASSERT_OK(Put("k", "v")); ASSERT_OK(dbfull()->TEST_SwitchMemtable()); Close(); - ASSERT_LE(ListLeftovers(options, dbname_).size(), 1U); + ASSERT_TRUE(ListLeftovers(options, dbname_).empty()); ASSERT_OK(TryReopen(options)); ASSERT_EQ(Get("k"), "v"); } @@ -3705,29 +2730,21 @@ TEST_F(CrashChild, DISABLED_LeftoverNoMagicFallsBackToWal) { TEST_F(DBCsppCrashSafeTest, LeftoverNoMagicFallsBackToWal) { Close(); Options options = BaseCrashSafeOptions(dbname_, true, false); - for (const std::string damage : {"empty", "short-header", "missing-magic"}) { - SCOPED_TRACE(damage); - Destroy(options); - ASSERT_EQ(RunCrashChild(dbname_, "LeftoverNoMagicFallsBackToWal"), 1); - auto leftovers = ListLeftovers(options, dbname_); - ASSERT_FALSE(leftovers.empty()); - const int fd = ::open(leftovers[0].c_str(), O_RDWR); - ASSERT_GE(fd, 0); - terark::DFA_MmapHeader hdr{}; - if (damage == "missing-magic") { - ASSERT_EQ(::pread(fd, &hdr, sizeof(hdr), 0), - static_cast(sizeof(hdr))); - std::memset(hdr.reserved, 0, sizeof(hdr.reserved)); - ASSERT_EQ(::pwrite(fd, &hdr, sizeof(hdr), 0), - static_cast(sizeof(hdr))); - } else { - ASSERT_EQ(::ftruncate(fd, damage == "empty" ? 0 : sizeof(hdr) - 1), 0); - } - ::close(fd); - ASSERT_OK(TryReopen(options)); - ASSERT_EQ(Get("bad"), "hdr"); - Close(); - } + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "LeftoverNoMagicFallsBackToWal"), 1); + auto leftovers = ListLeftovers(options, dbname_); + ASSERT_FALSE(leftovers.empty()); + const int fd = ::open(leftovers[0].c_str(), O_RDWR); + ASSERT_GE(fd, 0); + terark::DFA_MmapHeader hdr{}; + ASSERT_EQ(::pread(fd, &hdr, sizeof(hdr), 0), + static_cast(sizeof(hdr))); + std::memset(hdr.reserved, 0, sizeof(hdr.reserved)); + ASSERT_EQ(::pwrite(fd, &hdr, sizeof(hdr), 0), + static_cast(sizeof(hdr))); + ::close(fd); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("bad"), "hdr"); } TEST_F(CrashChild, DISABLED_DualLeftoverSecondConvertFails) { @@ -3788,130 +2805,78 @@ TEST_F(DBCsppCrashSafeTest, TruncateInjectFailureFallsBackToWal) { ASSERT_EQ(RunCrashChild(dbname_, "TruncateInjectFailureFallsBackToWal"), 1); SyncPoint::GetInstance()->EnableProcessing(); SyncPoint::GetInstance()->ClearAllCallBacks(); - bool truncate_called = false; SyncPoint::GetInstance()->SetCallBack( - "MemTableRep::ConvertToSST:Truncate", [&truncate_called](void* arg) { - truncate_called = true; - *static_cast(arg) = IOStatus::IOError("inject truncate"); + "CrashSafeRecover::Truncate:InjectStatus", [](void* arg) { + *static_cast(arg) = Status::IOError("inject truncate"); }); ASSERT_OK(TryReopen(options)); - ASSERT_TRUE(truncate_called); ASSERT_EQ(Get("tr"), "ok"); SyncPoint::GetInstance()->DisableProcessing(); SyncPoint::GetInstance()->ClearAllCallBacks(); } -TEST_F(CrashChild, DISABLED_AfterConvertBeforeAddFileKeepsRegisteredFile) { - ASSERT_EQ(arg_.size(), 3U); - Options options = BaseCrashSafeOptions(dbname_, true, arg_[1] == '1'); - if (arg_[0] == '1') SetupOsl(&options, true); +TEST_F(CrashChild, DISABLED_LinkFileInjectFailureFallsBackToWal) { + Options options = BaseCrashSafeOptions(dbname_, true, true); SyncPoint::GetInstance()->SetCallBack( "DBImpl::PersistPublishedSequence:AfterCommit", [](void*) { ::_exit(1); }); SyncPoint::GetInstance()->EnableProcessing(); DB* child_db = nullptr; ASSERT_OK(DB::Open(options, dbname_, &child_db)); - ASSERT_OK(child_db->Put(WriteOptions(), "or", std::string(128, 'v'))); + ASSERT_OK(child_db->Put(WriteOptions(), "lk", "ok")); ::_exit(0); } -TEST_F(CrashChild, DISABLED_AfterConvertBeforeAddFileKeepsRegisteredFileRecover) { - ASSERT_EQ(arg_.size(), 3U); - Options options = BaseCrashSafeOptions(dbname_, true, arg_[1] == '1'); - if (arg_[0] == '1') SetupOsl(&options, true); - const char* points[] = { - "MemTableRep::ConvertToSST:Truncate", - "CrashSafeRecover::AfterConvertBeforeAddFile", - "CrashSafeRecover::AfterConvertBeforeAddFile", - "DBImpl::RegisterMemTableFile:AfterLogAndApply"}; - ASSERT_LT(arg_[2] - '0', 4); +TEST_F(DBCsppCrashSafeTest, LinkFileInjectFailureFallsBackToWal) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, true); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "LinkFileInjectFailureFallsBackToWal"), 1); + SyncPoint::GetInstance()->EnableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); SyncPoint::GetInstance()->SetCallBack( - points[arg_[2] - '0'], - [&](void*) { - if (arg_[2] == '1') { - // Model an incomplete SST tail, without claiming this callback runs - // in the middle of a write. The persisted trie remains intact. - const auto files = ListLeftovers(options, dbname_); - ASSERT_EQ(files.size(), 1U); - const int fd = ::open(files.front().c_str(), O_RDWR); - ASSERT_GE(fd, 0); - uint64_t structure_size = 0; - if (arg_[0] == '1') { - terark::OSL_MmapHeader hdr{}; - ASSERT_EQ(::pread(fd, &hdr, sizeof(hdr), 0), - static_cast(sizeof(hdr))); - structure_size = hdr.mem_used; - } else { - terark::DFA_MmapHeader hdr{}; - ASSERT_EQ(::pread(fd, &hdr, sizeof(hdr), 0), - static_cast(sizeof(hdr))); - structure_size = hdr.file_size; - } - uint64_t size = 0; - ASSERT_OK(options.env->GetFileSize(files.front(), &size)); - ASSERT_GT(size, structure_size); - ASSERT_EQ(::ftruncate(fd, size - 1), 0); - ::close(fd); - } - ::_exit(1); + "CrashSafeRecover::LinkFile:InjectStatus", [](void* arg) { + *static_cast(arg) = Status::IOError("inject link"); }); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("lk"), "ok"); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); +} + +TEST_F(CrashChild, DISABLED_AfterRenameBeforeAddFileFallsBackToWal) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", + [](void*) { ::_exit(1); }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "or", "phan")); + ::_exit(0); +} + +TEST_F(CrashChild, DISABLED_AfterRenameBeforeAddFileFallsBackToWalRecover) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + SyncPoint::GetInstance()->SetCallBack( + "CrashSafeRecover::AfterRenameBeforeAddFile", + [](void*) { ::_exit(1); }); SyncPoint::GetInstance()->EnableProcessing(); DB* recover_db = nullptr; DB::Open(options, dbname_, &recover_db); ::_exit(0); } -TEST_F(DBCsppCrashSafeTest, AfterConvertBeforeAddFileKeepsRegisteredFile) { +TEST_F(DBCsppCrashSafeTest, AfterRenameBeforeAddFileFallsBackToWal) { Close(); - for (bool osl : {false, true}) { - for (bool log_index : {false, true}) { - for (int window = 0; window < 4; ++window) { - const std::string config = std::to_string(osl) + - std::to_string(log_index) + std::to_string(window); - SCOPED_TRACE(config); - Options options = BaseCrashSafeOptions(dbname_, true, log_index); - if (osl) SetupOsl(&options, true); - Destroy(options); - ASSERT_EQ(RunCrashChild(dbname_, - "AfterConvertBeforeAddFileKeepsRegisteredFile", config), 1); - const auto before = ListLeftovers(options, dbname_); - ASSERT_EQ(before.size(), 1U); - // Before commit, interrupt the same file twice to exercise footer - // replacement, not just a one-time conversion of the original source. - for (int crash = 0; crash < (window == 3 ? 1 : 2); ++crash) { - ASSERT_EQ(RunCrashChild(dbname_, - "AfterConvertBeforeAddFileKeepsRegisteredFileRecover", config), 1); - ASSERT_OK(env_->FileExists(before.front())); - if (window != 3) { - ASSERT_EQ(ListLeftovers(options, dbname_), before); - } - PublishedSeqOnDisk rec; - ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); - ASSERT_EQ(rec.generation & 1, 0U); - } - int converted = 0; - SyncPoint::GetInstance()->SetCallBack( - "MemTableRep::ConvertToSST:After", [&](void*) { ++converted; }); - SyncPoint::GetInstance()->EnableProcessing(); - ASSERT_OK(TryReopen(options)); - SyncPoint::GetInstance()->DisableProcessing(); - SyncPoint::GetInstance()->ClearAllCallBacks(); - ASSERT_EQ(converted, window == 3 ? 0 : 1); - ASSERT_EQ(Get("or"), std::string(128, 'v')); - std::vector files; - db_->GetLiveFilesMetaData(&files); - ASSERT_EQ(files.size(), 1U); - ASSERT_EQ(files.front().level, 0); - ASSERT_EQ(MakeTableFileName(dbname_, files.front().file_number), - before.front()); - Close(); - ASSERT_OK(TryReopen(options)); - ASSERT_EQ(Get("or"), std::string(128, 'v')); - ASSERT_EQ(CountL0(db_), 1); - Close(); - } - } - } + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "AfterRenameBeforeAddFileFallsBackToWal"), 1); + SyncPoint::GetInstance()->EnableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_EQ(RunCrashChild(dbname_, "AfterRenameBeforeAddFileFallsBackToWalRecover"), 1); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("or"), "phan"); } TEST_F(DBCsppCrashSafeTest, CrashSafeOrLogIndexDisablesWalCompression) { @@ -4178,7 +3143,7 @@ TEST_F(DBCsppCrashSafeTest, WritePreparedFallsBackToWal) { delete txn_db; } -TEST_F(DBCsppCrashSafeTest, AfterConvertCloseSecondFlushInject) { +TEST_F(DBCsppCrashSafeTest, AfterRenameCloseSecondFlushInject) { Close(); Options options = BaseCrashSafeOptions(dbname_, true, false); Destroy(options); @@ -4280,80 +3245,6 @@ TEST_F(DBCsppCrashSafeTest, AtomicFlushDualCfLeftoverConvertsBoth) { ASSERT_GE(CountL0(db_, "one"), 1); } -TEST_F(CrashChild, DISABLED_AtomicFlushManifestWindow) { - ASSERT_EQ(arg_.size(), 2U); - Options options = BaseCrashSafeOptions(dbname_, true, false); - options.atomic_flush = true; - if (arg_[0] == '1') SetupOsl(&options, true); - DB* child_db = nullptr; - std::vector handles; - const std::vector cfs = { - {kDefaultColumnFamilyName, options}, {"one", options}}; - ASSERT_OK(DB::Open(options, dbname_, cfs, &handles, &child_db)); - ASSERT_OK(child_db->Put(WriteOptions(), handles[0], "default-key", "one")); - ASSERT_OK(child_db->Put(WriteOptions(), handles[1], "other-key", "two")); - SyncPoint::GetInstance()->SetCallBack( - arg_[1] == '0' ? "FlushJob::BeforeManifest" : "FlushJob::AfterManifest", - [](void*) { ::_exit(42); }); - SyncPoint::GetInstance()->EnableProcessing(); - ASSERT_OK(child_db->Flush(FlushOptions(), handles)); - ::_exit(1); -} - -TEST_F(DBCsppCrashSafeTest, AtomicFlushCrashAcrossManifestCommit) { - Close(); - for (bool osl : {false, true}) { - for (bool committed : {false, true}) { - SCOPED_TRACE(osl); - SCOPED_TRACE(committed); - Options options = BaseCrashSafeOptions(dbname_, true, false); - options.atomic_flush = true; - if (osl) SetupOsl(&options, true); - Destroy(options); - ASSERT_OK(TryReopen(options)); - CreateAndReopenWithCF({"one"}, options); - Close(); - const std::string arg = std::string(osl ? "1" : "0") + - (committed ? "1" : "0"); - ASSERT_EQ(RunCrashChild(dbname_, "AtomicFlushManifestWindow", arg), 42); - const auto registered = ListLeftovers(options, dbname_); - ASSERT_GE(registered.size(), 2U); - for (const auto& path : registered) ASSERT_OK(env_->FileExists(path)); - std::atomic converted{0}; - SyncPoint::GetInstance()->SetCallBack( - "MemTableRep::ConvertToSST:After", - [&](void*) { ++converted; }); - SyncPoint::GetInstance()->EnableProcessing(); - ASSERT_OK(TryReopenWithColumnFamilies( - {kDefaultColumnFamilyName, "one"}, options)); - SyncPoint::GetInstance()->DisableProcessing(); - SyncPoint::GetInstance()->ClearAllCallBacks(); - ASSERT_EQ(converted.load(), committed ? 0 : 2); - ASSERT_EQ(Get(0, "default-key"), "one"); - ASSERT_EQ(Get(1, "other-key"), "two"); - ASSERT_EQ(CountL0(db_, kDefaultColumnFamilyName), 1); - ASSERT_EQ(CountL0(db_, "one"), 1); - std::vector files; - db_->GetLiveFilesMetaData(&files); - ASSERT_EQ(files.size(), 2U); - for (const auto& file : files) { - const std::string path = MakeTableFileName(dbname_, file.file_number); - ASSERT_OK(env_->FileExists(path)); - ASSERT_EQ(std::count(registered.begin(), registered.end(), path), - committed ? 0 : 1); - } - Close(); - ASSERT_OK(TryReopenWithColumnFamilies( - {kDefaultColumnFamilyName, "one"}, options)); - ASSERT_EQ(Get(0, "default-key"), "one"); - ASSERT_EQ(Get(1, "other-key"), "two"); - ASSERT_EQ(CountL0(db_, kDefaultColumnFamilyName), 1); - ASSERT_EQ(CountL0(db_, "one"), 1); - Close(); - } - } -} - TEST_F(CrashChild, DISABLED_DroppedCfLeftoverSkipped) { Options options = BaseCrashSafeOptions(dbname_, true, false); std::atomic pubs{0}; @@ -4387,10 +3278,13 @@ TEST_F(DBCsppCrashSafeTest, DroppedCfLeftoverSkipped) { ASSERT_EQ(Get(0, "keep"), "1"); ASSERT_EQ(Get(1, "drop"), "2"); ASSERT_OK(Flush(0)); - const auto registered = dbfull()->GetVersionSet()->GetMemTableFiles(); - ASSERT_EQ(registered.count(1), 1U); - ASSERT_FALSE(registered.at(1).empty()); - const std::string left1 = MakeTableFileName(dbname_, *registered.at(1).begin()); + std::string left1; + for (const auto& p : ListLeftovers(options, dbname_)) { + if (p.find(".memtab-1") != std::string::npos) { + left1 = p; + break; + } + } ASSERT_FALSE(left1.empty()); const std::string bak = left1 + ".bak"; CopyFile(left1, bak); @@ -4403,9 +3297,13 @@ TEST_F(DBCsppCrashSafeTest, DroppedCfLeftoverSkipped) { } ASSERT_OK(TryReopen(options)); ASSERT_EQ(Get("keep"), "1"); - ASSERT_EQ(dbfull()->GetVersionSet()->GetMemTableFiles().count(1), 0U); - const auto leftovers = ListLeftovers(options, dbname_); - ASSERT_EQ(std::count(leftovers.begin(), leftovers.end(), left1), 0); + bool dropped_left = false; + for (const auto& p : ListLeftovers(options, dbname_)) { + if (p.find(".memtab-1") != std::string::npos) { + dropped_left = true; + } + } + ASSERT_TRUE(dropped_left); } TEST_F(CrashChild, DISABLED_MultiChunkAfterCommitStillReadable) { diff --git a/db/db_impl/db_impl.cc b/db/db_impl/db_impl.cc index b0ddc2092..4b139aab1 100644 --- a/db/db_impl/db_impl.cc +++ b/db/db_impl/db_impl.cc @@ -4075,11 +4075,6 @@ Status DBImpl::CreateColumnFamilyImpl(const ColumnFamilyOptions& cf_options, // LogAndApply will both write the creation in MANIFEST and create // ColumnFamilyData object - std::unique_ptr::iterator> pending_memtable; - if (cf_options.memtable_factory->SupportCrashSafe()) { - pending_memtable = std::make_unique::iterator>( - CaptureCurrentFileNumberInPendingOutputs()); - } { // write thread WriteThread::Writer w; write_thread_.EnterUnbatched(&w, &mutex_); @@ -4096,20 +4091,7 @@ Status DBImpl::CreateColumnFamilyImpl(const ColumnFamilyOptions& cf_options, assert(cfd != nullptr); std::map> dummy_created_dirs; s = cfd->AddDirectories(&dummy_created_dirs); - if (s.ok()) { - s = RegisterMemTableFile(cfd, cfd->mem()); - if (!s.ok()) { - // CF creation is already committed. Keep its in-memory state valid, - // but do not return a writable handle after registration fails. - InstallSuperVersionAndScheduleWork(cfd, &sv_context, - *cfd->GetLatestMutableCFOptions()); - cfd->set_initialized(); - error_handler_.SetBGError(s, BackgroundErrorReason::kManifestWrite) - .PermitUncheckedError(); - } - } } - ReleaseFileNumberFromPendingOutputs(pending_memtable); if (s.ok()) { auto* cfd = versions_->GetColumnFamilySet()->GetColumnFamily(column_family_name); diff --git a/db/db_impl/db_impl.h b/db/db_impl/db_impl.h index 71fb85804..c62f559ee 100644 --- a/db/db_impl/db_impl.h +++ b/db/db_impl/db_impl.h @@ -1437,6 +1437,8 @@ class DBImpl : public DB { // such a file's absolute path to its parent directory. std::unordered_map files_to_delete_; bool is_new_db_ = false; + // Restore the saved WAL cursor only after recovery edits are installed. + bool restore_published_seq_ = false; // WAL tail cursor for this Recover. RecoverLogFiles reads it from here. bool crash_safe_wal_tail_replay_ = false; uint8_t crash_safe_wal_offset_kind_ = 0; @@ -1964,10 +1966,12 @@ class DBImpl : public DB { void StagePublishedWal(SequenceNumber seq, uint64_t wal_number, uint64_t wal_offset); void AccountPendingMemtableWrites(size_t n); - Status RegisterMemTableFile(ColumnFamilyData* cfd, MemTable* mem); bool CanConvertLeftoverForCrashSafeRecover( - SequenceNumber mmap_pubseq, std::string* fail_reason); + const std::vector& leftover_snapshot, + SequenceNumber mmap_pubseq, uint64_t mmap_wal_number, + std::string* fail_reason); Status ConvertLeftoverMemtables( + const std::vector& leftover_snapshot, SequenceNumber max_visible_seq, RecoveryContext* recovery_ctx); // The following two methods are used to flush a memtable to diff --git a/db/db_impl/db_impl_files.cc b/db/db_impl/db_impl_files.cc index 9d2fed670..47901f4d2 100644 --- a/db/db_impl/db_impl_files.cc +++ b/db/db_impl/db_impl_files.cc @@ -175,15 +175,6 @@ void DBImpl::FindObsoleteFiles(JobContext* job_context, bool force, if (doing_the_full_scan) { versions_->AddLiveFiles(&job_context->sst_live, &job_context->blob_live); - // These share the SST namespace, but are still mutable. Protect them from - // GC without exposing them as immutable SSTs to checkpoints and backups. - for (const auto& cf : versions_->GetMemTableFiles()) { - job_context->sst_live.insert(job_context->sst_live.end(), - cf.second.begin(), cf.second.end()); - } - for (auto* cfd : *versions_->GetColumnFamilySet()) { - cfd->AddMemTableCacheFileNumbers(&job_context->sst_live); - } InfoLogPrefix info_log_prefix(!immutable_db_options_.db_log_dir.empty(), dbname_); std::set paths; diff --git a/db/db_impl/db_impl_open.cc b/db/db_impl/db_impl_open.cc index 3b50df376..5665a55ae 100644 --- a/db/db_impl/db_impl_open.cc +++ b/db/db_impl/db_impl_open.cc @@ -543,6 +543,20 @@ bool PeekPublishedWalKind(const std::string& dbname, uint64_t* kind) { return ok; } +uint32_t ParseLeftoverCfId(const std::string& path) { + const auto pos = path.rfind(".memtab-"); + if (pos == std::string::npos) { + return std::numeric_limits::max(); + } + const char* p = path.c_str() + pos + 8; + char* end = nullptr; + const unsigned long id = std::strtoul(p, &end, 10); + if (end == p) { + return std::numeric_limits::max(); + } + return static_cast(id); +} + void DestroyKindPrepDb(DB* db, std::vector* handles) { for (auto* h : *handles) { delete h; @@ -731,13 +745,13 @@ void DBImpl::PersistPublishedSequence(SequenceNumber seq, uint64_t wal_number, } TEST_SYNC_POINT("DBImpl::PersistPublishedSequence:SameSeq"); } - // A failed publication may leave an odd generation. + // Recovery may leave an odd generation until the first new publication. const uint64_t g = pubseq_mmap_->generation | 1; #if defined(__AVX__) // One aligned 32-byte store. A process crash falls between instructions, // so the record is all old or all new. Publish an even generation only - // after this store, including when a prior publication left it odd. A host - // without it uses the odd/even bracket below after the file is moved. + // after this store, including when recovery left it odd. A host without it + // uses the odd/even bracket below after the file is moved. // PublishedSeqRecord next; // next.pubseq = seq; // next.wal_number = wal_number; @@ -806,44 +820,15 @@ void DBImpl::AccountPendingMemtableWrites(size_t n) { } } -Status DBImpl::RegisterMemTableFile(ColumnFamilyData* cfd, MemTable* mem) { - mutex_.AssertHeld(); - if (mem->GetBackingFileNumber() == 0 || mem->IsFileRegistered()) { - return Status::OK(); - } - VersionEdit edit; - edit.SetColumnFamily(cfd->GetID()); - edit.SetMemTableFileTracking(); - edit.AddMemTableFile(mem->GetBackingFileNumber()); - Status s; - TEST_SYNC_POINT_CALLBACK("DBImpl::RegisterMemTableFile:BeforeLogAndApply", &s); - if (s.ok()) { - s = versions_->LogAndApply(cfd, *cfd->GetLatestMutableCFOptions(), - ReadOptions(), &edit, &mutex_, - directories_.GetDbDir()); - } - // A dropped CF's edit can be discarded with OK status. - if (s.ok() && cfd->IsDropped()) { - s = Status::ColumnFamilyDropped(); - } - TEST_SYNC_POINT_CALLBACK("DBImpl::RegisterMemTableFile:AfterLogAndApply", &s); - if (s.ok()) { - mem->MarkFileRegistered(); - TEST_SYNC_POINT("DBImpl::RegisterMemTableFile:BeforeInstall"); - } - return s; -} - bool DBImpl::CanConvertLeftoverForCrashSafeRecover( - SequenceNumber mmap_pubseq, std::string* fail_reason) { + const std::vector& leftover_snapshot, + SequenceNumber mmap_pubseq, uint64_t mmap_wal_number, + std::string* fail_reason) { mutex_.AssertHeld(); auto fail = [&](const std::string& reason) { *fail_reason = reason; return false; }; - if (!versions_->HasMemTableFileTracking()) { - return fail("MANIFEST predates MemTable file tracking"); - } if (immutable_db_options_.wal_filter != nullptr) { return fail("wal_filter requires full WAL replay"); } @@ -853,9 +838,6 @@ bool DBImpl::CanConvertLeftoverForCrashSafeRecover( if (immutable_db_options_.best_efforts_recovery) { return fail("best_efforts_recovery"); } - if (mmap_pubseq == 0 && !versions_->GetMemTableFiles().empty()) { - return fail("CSPUBSEQ pubseq 0 with leftover present"); - } for (auto* cfd : *versions_->GetColumnFamilySet()) { if (cfd->IsDropped()) { continue; @@ -864,14 +846,90 @@ bool DBImpl::CanConvertLeftoverForCrashSafeRecover( if (fac == nullptr || !fac->SupportCrashSafe()) { return fail("CF " + cfd->GetName() + " factory !SupportCrashSafe"); } + if (cfd->mem() != nullptr && !cfd->mem()->SupportConvertToSST()) { + return fail("CF " + cfd->GetName() + " mem !SupportConvertToSST"); + } + if (!cfd->imm()->UnflushedMemtablesSupportConvertToSST()) { + return fail("CF " + cfd->GetName() + " imm !SupportConvertToSST"); + } + } + + // Other tiers may be slow and contain many files; memtables use cf_paths[0]. + std::vector paths; + for (const auto* cfd : *versions_->GetColumnFamilySet()) { + if (!cfd->IsDropped()) { + paths.push_back( + NormalizePath(cfd->ioptions()->cf_paths[0].path + + std::string(1, kFilePathSeparator))); + } + } + std::sort(paths.begin(), paths.end()); + paths.erase(std::unique(paths.begin(), paths.end()), paths.end()); + const uint64_t next_file_number = versions_->current_next_file_number(); + for (const auto& path : paths) { + std::vector files; + const Status ls = env_->GetChildren(path, &files); + if (!ls.ok()) { + continue; + } + for (const auto& fname : files) { + uint64_t number = 0; + FileType type; + if (!ParseFileName(fname, &number, &type)) { + continue; + } + // Only numbers still ahead of MANIFEST next_file (crash mid-Open + // before LogAndApply). A Convert that failed after this Open already + // advanced next_file is invisible here; ConvertLeftover deletes those. + if (type == kTableFile && number >= next_file_number) { + return fail("orphan SST " + path + fname); + } + } + } + + std::unordered_set leftover_cfs; + for (const auto& leftover_path : leftover_snapshot) { + const uint32_t cf_id = ParseLeftoverCfId(leftover_path); + ColumnFamilyData* cfd = + versions_->GetColumnFamilySet()->GetColumnFamily(cf_id); + if (cfd == nullptr) { + ROCKS_LOG_WARN(immutable_db_options_.info_log, + "Crash-safe leftover %s belongs to dropped CF %u, skip", + leftover_path.c_str(), cf_id); + continue; + } + auto* fac = cfd->ioptions()->memtable_factory.get(); + const Status ps = + fac->ProbeCrashSafeLeftover(leftover_path, + immutable_db_options_.GetWalDir()); + if (!ps.ok()) { + return fail("probe " + leftover_path + ": " + ps.ToString()); + } + leftover_cfs.insert(cf_id); + } + if (mmap_pubseq > versions_->LastSequence() && leftover_cfs.empty()) { + return fail("CSPUBSEQ > MANIFEST LastSequence and no leftover"); + } + if (!leftover_cfs.empty() && mmap_pubseq == 0) { + return fail("CSPUBSEQ pubseq 0 with leftover present"); + } + if (mmap_wal_number != 0) { + for (const auto* cfd : *versions_->GetColumnFamilySet()) { + // Only CFs persisted past this WAL can omit their leftover safely. + if (!cfd->IsDropped() && cfd->GetLogNumber() <= mmap_wal_number && + leftover_cfs.count(cfd->GetID()) == 0) { + return fail("CF " + cfd->GetName() + " has no leftover before WAL cursor"); + } + } } return true; } Status DBImpl::ConvertLeftoverMemtables( + const std::vector& leftover_snapshot, SequenceNumber max_visible_seq, RecoveryContext* recovery_ctx) { mutex_.AssertHeld(); - if (versions_->GetMemTableFiles().empty()) { + if (leftover_snapshot.empty()) { return Status::OK(); } @@ -881,23 +939,19 @@ Status DBImpl::ConvertLeftoverMemtables( std::vector blobs; }; std::vector converted; - // MANIFEST is the complete inventory, including empty, precreated tables. - // A directory scan cannot detect one missing file among several in a CF. - for (const auto& cf : versions_->GetMemTableFiles()) { - auto* cfd = versions_->GetColumnFamilySet()->GetColumnFamily(cf.first); - assert(cfd != nullptr && !cfd->IsDropped()); - for (uint64_t file_num : cf.second) { - auto& one = converted.emplace_back(); - one.cfd = cfd; - one.meta.fd = FileDescriptor(file_num, 0, 0); - } - } Status s; mutex_.Unlock(); - for (auto& one : converted) { - auto* cfd = one.cfd; - const uint64_t file_num = one.meta.fd.GetNumber(); - const auto leftover_path = TableFileName(cfd->ioptions()->cf_paths, file_num, 0); + for (const auto& leftover_path : leftover_snapshot) { + const uint32_t cf_id = ParseLeftoverCfId(leftover_path); + ColumnFamilyData* cfd = + versions_->GetColumnFamilySet()->GetColumnFamily(cf_id); + if (cfd == nullptr) { + continue; + } + ConvertedLeftover one; + one.cfd = cfd; + const uint64_t file_num = versions_->NewFileNumber(); + one.meta.fd = FileDescriptor(file_num, 0, 0); one.meta.fd.smallest_seqno = 0; one.meta.fd.largest_seqno = max_visible_seq; one.meta.epoch_number = cfd->NewEpochNumber(); @@ -914,16 +968,31 @@ Status DBImpl::ConvertLeftoverMemtables( tboptions.add_blob_file = [&one](BlobFileAddition b) { one.blobs.push_back(std::move(b)); }; - s = cfd->ioptions()->memtable_factory->RecoverCrashSafeMemTableToSST( - leftover_path, &one.meta, tboptions); - if (!s.ok()) { + Status one_s = + cfd->ioptions()->memtable_factory->RecoverCrashSafeMemTableToSST( + leftover_path, &one.meta, tboptions); + if (!one_s.ok()) { + s = one_s; + // Inject / rename-then-fail still leaves an SST. Include it so the + // cleanup below can unlink it; leftover is already gone. + converted.push_back(std::move(one)); break; } + converted.push_back(std::move(one)); } mutex_.Lock(); if (!s.ok()) { - // Do not retire registered sources until full WAL recovery commits. + // NewFileNumber already ran. After this Open finishes, next_file is + // past these numbers, so DeleteUnreferencedSstFiles (only >= next_file, + // and it runs before Convert) will not collect them. for (auto& one : converted) { + if (one.meta.fd.GetNumber() == 0) { + continue; + } + env_->DeleteFile(TableFileName(one.cfd->ioptions()->cf_paths, + one.meta.fd.GetNumber(), + one.meta.fd.GetPathId())) + .PermitUncheckedError(); for (const auto& blob : one.blobs) { env_->DeleteFile(BlobFileName(one.cfd->ioptions()->cf_paths.front().path, blob.GetBlobFileNumber())) @@ -933,13 +1002,13 @@ Status DBImpl::ConvertLeftoverMemtables( return s; } for (auto& one : converted) { + if (one.meta.fd.GetFileSize() == 0) { + continue; + } VersionEdit edit; edit.SetColumnFamily(one.cfd->GetID()); - edit.DeleteMemTableFile(one.meta.fd.GetNumber()); - if (one.meta.fd.GetFileSize() != 0) { - one.meta.marked_for_compaction = true; - edit.AddFile(0, one.meta); - } + one.meta.marked_for_compaction = true; + edit.AddFile(0, one.meta); for (const auto& blob : one.blobs) { edit.AddBlobFile(blob); } @@ -953,12 +1022,24 @@ Status DBImpl::Recover( bool error_if_wal_file_exists, bool error_if_data_exists_in_wals, uint64_t* recovered_seq, RecoveryContext* recovery_ctx) { mutex_.AssertHeld(); - if (read_only) { - for (const auto& cf : column_families) { - if (cf.options.memtable_factory->SupportCrashSafe()) { - return Status::InvalidArgument( - "FileMmap memtable is not supported in read-only mode", cf.name); + + std::vector leftover_snapshot; + if (immutable_db_options_.memtable_crash_safe_recover && !read_only) { + for (const auto& desc : column_families) { + if (!desc.options.memtable_factory) { + continue; } + const auto& paths = desc.options.cf_paths.empty() + ? immutable_db_options_.db_paths + : desc.options.cf_paths; + desc.options.memtable_factory->ListCrashSafeLeftovers( + paths[0].path, &leftover_snapshot); + } + if (!leftover_snapshot.empty()) { + std::sort(leftover_snapshot.begin(), leftover_snapshot.end()); + leftover_snapshot.erase( + std::unique(leftover_snapshot.begin(), leftover_snapshot.end()), + leftover_snapshot.end()); } } @@ -1208,6 +1289,7 @@ Status DBImpl::Recover( s = SetupDBId(read_only, recovery_ctx); ROCKS_LOG_INFO(immutable_db_options_.info_log, "DB ID: %s\n", db_id_.c_str()); bool crash_safe_convert = false; + bool recovered_wals_ok = false; PublishedSeqRecord mmap_rec; bool mmap_valid = false; if (s.ok() && !read_only && @@ -1216,7 +1298,7 @@ Status DBImpl::Recover( std::string fail_reason; if (mmap_valid) { crash_safe_convert = CanConvertLeftoverForCrashSafeRecover( - mmap_rec.pubseq, &fail_reason); + leftover_snapshot, mmap_rec.pubseq, mmap_rec.wal_number, &fail_reason); } else { fail_reason = "invalid CSPUBSEQ generation"; } @@ -1225,23 +1307,26 @@ Status DBImpl::Recover( "Crash-safe recover check failed (%s), fallback to full " "WAL RecoverLogFiles", fail_reason.c_str()); + for (const auto& leftover_path : leftover_snapshot) { + ROCKS_LOG_WARN(immutable_db_options_.info_log, + "Crash-safe leftover not converted: %s", + leftover_path.c_str()); + } } } if (s.ok() && !read_only) { - s = DeleteUnreferencedSstFiles(recovery_ctx); - } - if (s.ok()) { - // MANIFEST recovery restores the allocator; orphan discovery then advances - // it past files created before a previous Open could commit. - for (auto* cfd : *versions_->GetColumnFamilySet()) { - if (cfd->mem() == nullptr) { - cfd->CreateNewMemtable(*cfd->GetLatestMutableCFOptions(), - versions_->LastSequence(), !read_only); - } + if (pubseq_mmap_ != nullptr) { + // Recovery can consume only part of the leftover set before failing. + // Keep the cursor invalid across a second crash, including one after + // orphan/failed-conversion cleanup has removed the failure evidence. + pubseq_mmap_->generation |= 1; + recovery_ctx->restore_published_seq_ = mmap_valid; } + s = DeleteUnreferencedSstFiles(recovery_ctx); } if (s.ok() && crash_safe_convert) { - const Status cs = ConvertLeftoverMemtables(mmap_rec.pubseq, recovery_ctx); + const Status cs = ConvertLeftoverMemtables(leftover_snapshot, mmap_rec.pubseq, + recovery_ctx); if (!cs.ok()) { ROCKS_LOG_WARN(immutable_db_options_.info_log, "Crash-safe leftover Convert failed (%s), fallback to " @@ -1383,6 +1468,7 @@ Status DBImpl::Recover( bool corrupted_wal_found = false; s = RecoverLogFiles(wals, &next_sequence, read_only, &corrupted_wal_found, recovery_ctx); + recovered_wals_ok = s.ok(); if (corrupted_wal_found && recovered_seq != nullptr) { *recovered_seq = next_sequence; } @@ -1396,17 +1482,13 @@ Status DBImpl::Recover( } } - if (s.ok() && !read_only && !crash_safe_convert) { - // Retire the old inventory only together with the successful WAL recovery. - // Missing files are allowed here: their absence is what forced the replay. - for (const auto& cf : versions_->GetMemTableFiles()) { - auto* cfd = versions_->GetColumnFamilySet()->GetColumnFamily(cf.first); - VersionEdit edit; - edit.SetColumnFamily(cf.first); - for (uint64_t number : cf.second) { - edit.DeleteMemTableFile(number); - } - recovery_ctx->UpdateVersionEdits(cfd, edit); + if (s.ok() && recovered_wals_ok && !crash_safe_convert && + !leftover_snapshot.empty()) { + for (const auto& leftover_path : leftover_snapshot) { + ROCKS_LOG_INFO(immutable_db_options_.info_log, + "Crash-safe leftover deleted after WAL fallback: %s", + leftover_path.c_str()); + env_->DeleteFile(leftover_path).PermitUncheckedError(); } } @@ -1572,41 +1654,9 @@ Status DBImpl::LogAndApplyForRecovery(const RecoveryContext& recovery_ctx) { mutex_.AssertHeld(); assert(versions_->descriptor_log_ == nullptr); const ReadOptions read_options(Env::IOActivity::kDBOpen); - // Recovery retires the old inventory and installs all recovered CFs at once. - // A truncated MANIFEST tail must not install just part of that transition. - uint32_t remaining = 0; - bool tracks_memtable_files = false; - for (const auto& edits : recovery_ctx.edit_lists_) { - remaining += static_cast(edits.size()); - for (const auto* edit : edits) { - tracks_memtable_files |= edit->HasMemTableFileTracking() || - !edit->GetMemTableFileAdditions().empty() || - !edit->GetMemTableFileDeletions().empty(); - } - } - if (tracks_memtable_files && remaining > 1) { - for (const auto& edits : recovery_ctx.edit_lists_) { - for (auto* edit : edits) { - edit->MarkAtomicGroup(--remaining); - } - } - } - Status s; - TEST_SYNC_POINT_CALLBACK("DBImpl::RegisterMemTableFile:BeforeLogAndApply", &s); - if (s.ok()) { - s = versions_->LogAndApply( - recovery_ctx.cfds_, recovery_ctx.mutable_cf_opts_, read_options, - recovery_ctx.edit_lists_, &mutex_, directories_.GetDbDir()); - } - TEST_SYNC_POINT_CALLBACK("DBImpl::RegisterMemTableFile:AfterLogAndApply", &s); - if (s.ok()) { - for (auto* cfd : *versions_->GetColumnFamilySet()) { - if (cfd->mem()->GetBackingFileNumber() != 0) { - cfd->mem()->MarkFileRegistered(); - } - } - TEST_SYNC_POINT("DBImpl::RegisterMemTableFile:BeforeInstall"); - } + Status s = versions_->LogAndApply( + recovery_ctx.cfds_, recovery_ctx.mutable_cf_opts_, read_options, + recovery_ctx.edit_lists_, &mutex_, directories_.GetDbDir()); if (s.ok() && !(recovery_ctx.files_to_delete_.empty())) { mutex_.Unlock(); for (const auto& stale_sst_file : recovery_ctx.files_to_delete_) { @@ -1620,6 +1670,9 @@ Status DBImpl::LogAndApplyForRecovery(const RecoveryContext& recovery_ctx) { } mutex_.Lock(); } + if (s.ok() && recovery_ctx.restore_published_seq_) { + pubseq_mmap_->generation += 1; + } return s; } @@ -2863,18 +2916,6 @@ Status DBImpl::Open(const DBOptions& db_options, const std::string& dbname, impl->pubseq_mmap_->kind_since_wal = impl->logfile_number_; } if (s.ok()) { - for (auto* cfd : *impl->versions_->GetColumnFamilySet()) { - const uint64_t number = cfd->mem()->GetBackingFileNumber(); - if (number != 0 || impl->immutable_db_options_.memtable_crash_safe_recover) { - VersionEdit edit; - edit.SetColumnFamily(cfd->GetID()); - edit.SetMemTableFileTracking(); - if (number != 0) { - edit.AddMemTableFile(number); - } - recovery_ctx.UpdateVersionEdits(cfd, edit); - } - } s = impl->LogAndApplyForRecovery(recovery_ctx); } diff --git a/db/db_impl/db_impl_secondary.cc b/db/db_impl/db_impl_secondary.cc index bfc2fca60..8f4d2e641 100644 --- a/db/db_impl/db_impl_secondary.cc +++ b/db/db_impl/db_impl_secondary.cc @@ -36,12 +36,6 @@ Status DBImplSecondary::Recover( bool /*error_if_data_exists_in_wals*/, uint64_t*, RecoveryContext* /*recovery_ctx*/) { mutex_.AssertHeld(); - for (const auto& cf : column_families) { - if (cf.options.memtable_factory->SupportCrashSafe()) { - return Status::InvalidArgument( - "FileMmap memtable is not supported in secondary mode", cf.name); - } - } JobContext job_context(0); Status s; @@ -61,10 +55,6 @@ Status DBImplSecondary::Recover( max_total_in_memory_state_ = 0; for (auto cfd : *versions_->GetColumnFamilySet()) { auto* mutable_cf_options = cfd->GetLatestMutableCFOptions(); - if (cfd->mem() == nullptr) { - cfd->CreateNewMemtable(*mutable_cf_options, versions_->LastSequence(), - /*create_file=*/false); - } max_total_in_memory_state_ += mutable_cf_options->write_buffer_size * mutable_cf_options->max_write_buffer_number; } @@ -312,8 +302,7 @@ Status DBImplSecondary::RecoverLogFiles( const MutableCFOptions mutable_cf_options = *cfd->GetLatestMutableCFOptions(); MemTable* new_mem = - cfd->ConstructNewMemtable(mutable_cf_options, seq_of_batch, - /*create_file=*/false); + cfd->ConstructNewMemtable(mutable_cf_options, seq_of_batch); cfd->mem()->SetNextLogNumber(log_number); cfd->mem()->ConstructFragmentedRangeTombstones(); cfd->imm()->Add(cfd->mem(), &job_context->memtables_to_free); diff --git a/db/db_impl/db_impl_write.cc b/db/db_impl/db_impl_write.cc index 9c3d92820..2211106a5 100644 --- a/db/db_impl/db_impl_write.cc +++ b/db/db_impl/db_impl_write.cc @@ -2353,11 +2353,6 @@ Status DBImpl::SwitchMemtable(ColumnFamilyData* cfd, WriteContext* context) { int num_imm_unflushed = cfd->imm()->NumNotFlushed(); const auto preallocate_block_size = GetWalPreallocateBlockSize(mutable_cf_options.write_buffer_size); - std::unique_ptr::iterator> pending_memtable; - if (cfd->ioptions()->memtable_factory->SupportCrashSafe()) { - pending_memtable = std::make_unique::iterator>( - CaptureCurrentFileNumberInPendingOutputs()); - } mutex_.Unlock(); if (creating_new_log) { // TODO: Write buffer size passed in should be max of all CF's instead @@ -2390,10 +2385,6 @@ Status DBImpl::SwitchMemtable(ColumnFamilyData* cfd, WriteContext* context) { assert(log_recycle_files_.front() == recycle_log_number); log_recycle_files_.pop_front(); } - if (s.ok()) { - s = RegisterMemTableFile(cfd, new_mem); - } - ReleaseFileNumberFromPendingOutputs(pending_memtable); if (s.ok() && creating_new_log) { InstrumentedMutexLock l(&log_write_mutex_); assert(new_log != nullptr); @@ -2426,6 +2417,8 @@ Status DBImpl::SwitchMemtable(ColumnFamilyData* cfd, WriteContext* context) { } if (!s.ok()) { + // how do we fail if we're not creating new log? + assert(creating_new_log); delete new_mem; delete new_log; context->superversion_context.new_superversion.reset(); diff --git a/db/db_memtable_convert_test.cc b/db/db_memtable_convert_test.cc index 831a3473f..d3d20c436 100644 --- a/db/db_memtable_convert_test.cc +++ b/db/db_memtable_convert_test.cc @@ -2,7 +2,6 @@ // Flush/Close conversion, independently of crash-safe recovery. #include -#include #include #include @@ -11,14 +10,8 @@ #include "db/db_impl/db_impl.h" #include "db/db_test_util.h" -#include "db/memtable.h" -#include "db/version_set.h" -#include "env/env_chroot.h" -#include "file/filename.h" -#include "memory/arena.h" #include "port/stack_trace.h" #include "test_util/sync_point.h" -#include "table/table_builder.h" namespace ROCKSDB_NAMESPACE { @@ -119,150 +112,6 @@ TEST_P(DBMemtableConvertTest, ManualFlushConverts) { Close(); } -TEST_P(DBMemtableConvertTest, FileMmapConversionKeepsFileNumberAndPath) { - if (std::string(std::get<1>(GetParam())) != "kFileMmap") return; - Options options = ConvertOptions(); - DestroyAndReopen(options); - ASSERT_OK(Put("same-file", "value")); - const auto registered = dbfull()->GetVersionSet()->GetMemTableFiles(); - ASSERT_EQ(registered.count(0), 1U); - ASSERT_EQ(registered.at(0).size(), 1U); - const uint64_t number = *registered.at(0).begin(); - const std::string path = MakeTableFileName(dbname_, number); - ASSERT_OK(env_->FileExists(path)); - ObserveConversion(); - ASSERT_OK(db_->Flush(FlushOptions())); - std::vector files; - db_->GetLiveFilesMetaData(&files); - ASSERT_EQ(files.size(), 1U); - ASSERT_EQ(files[0].file_number, number); - ASSERT_OK(env_->FileExists(path)); - const auto& after = dbfull()->GetVersionSet()->GetMemTableFiles(); - auto cf = after.find(0); - ASSERT_TRUE(cf == after.end() || cf->second.count(number) == 0); - Close(); - ASSERT_OK(TryReopen(options)); - ASSERT_EQ(Get("same-file"), "value"); -} - -#if !defined(OS_WIN) -TEST_P(DBMemtableConvertTest, FileMmapExclusiveCreationAndChrootConversion) { - if (std::string(std::get<1>(GetParam())) != "kFileMmap") return; - Options options = ConvertOptions(); - DestroyAndReopen(options); - Close(); - const std::string logical_path = MakeTableFileName("", 900000); - const std::string physical_path = dbname_ + logical_path; - auto factory = PluginFactorySP::AcquirePlugin( - std::get<0>(GetParam()) ? "OffsetSkipList" : "CSPPMemTab", - {{"mem_cap", 16777216}, {"convert_to_sst", "kFileMmap"}, - {"chroot_dir", dbname_}}, repo_); - InternalKeyComparator icmp(options.comparator); - MemTable::KeyComparator cmp(icmp); - MutableCFOptions moptions(options); - moptions.write_buffer_size = 1 << 20; - Arena arena; - if (std::get<0>(GetParam())) { - const std::string sentinel = "existing SST must remain intact"; - ASSERT_OK(WriteStringToFile(env_, sentinel, physical_path)); - ASSERT_THROW({ - std::unique_ptr collision(factory->CreateMemTableRep( - logical_path, moptions, cmp, &arena, nullptr, nullptr, 0)); - }, std::runtime_error); - std::string after; - ASSERT_OK(ReadFileToString(env_, physical_path, &after)); - ASSERT_EQ(after, sentinel); - ASSERT_OK(env_->DeleteFile(physical_path)); - } - - std::unique_ptr chroot_env(NewChrootEnv(env_, dbname_)); - options.env = chroot_env.get(); - options.cf_paths = {{"", 0}}; - ImmutableOptions ioptions(options); - IntTblPropCollectorFactories collectors; - const std::string cf_name = "default"; - TableBuilderOptions tbo(ioptions, moptions, icmp, &collectors, - options.compression, options.compression_opts, 0, - cf_name, 0); - std::unique_ptr rep(factory->CreateMemTableRep( - logical_path, moptions, cmp, &arena, nullptr, nullptr, 0)); - ASSERT_TRUE(rep->IsFileMmap()); - rep->InitSetMemTableAsLogIndex(false); - rep->MarkFileRegistered(); - ASSERT_TRUE(rep->InsertKeyValue(PackSequenceAndType(1, kTypeValue), "k", "v")); - rep->MarkReadOnly(); - FileMetaData meta; - meta.fd = FileDescriptor(900001, 0, 0); - meta.num_entries = 1; - ASSERT_TRUE(rep->ConvertToSST(&meta, tbo).IsInvalidArgument()); - rep.reset(); - ASSERT_OK(env_->FileExists(physical_path)); - meta.fd = FileDescriptor(900000, 0, 0); - meta.fd.largest_seqno = kMaxSequenceNumber; - ASSERT_OK(factory->RecoverCrashSafeMemTableToSST(logical_path, &meta, tbo)); - ASSERT_GT(meta.fd.GetFileSize(), 0U); - ASSERT_OK(env_->FileExists(physical_path)); - ASSERT_OK(env_->DeleteFile(physical_path)); - - // Retry conversion on the same live rep after a completed footer exists. - rep.reset(factory->CreateMemTableRep( - logical_path, moptions, cmp, &arena, nullptr, nullptr, 0)); - rep->InitSetMemTableAsLogIndex(false); - rep->MarkFileRegistered(); - ASSERT_TRUE(rep->InsertKeyValue(PackSequenceAndType(1, kTypeValue), "k", "v")); - rep->MarkReadOnly(); - meta.fd = FileDescriptor(900000, 0, 0); - ASSERT_OK(rep->ConvertToSST(&meta, tbo)); - const uint64_t first_size = meta.fd.GetFileSize(); - ASSERT_GT(first_size, 0U); - ASSERT_OK(rep->ConvertToSST(&meta, tbo)); - ASSERT_EQ(meta.fd.GetFileSize(), first_size); - uint64_t actual_size = 0; - ASSERT_OK(env_->GetFileSize(physical_path, &actual_size)); - ASSERT_EQ(actual_size, first_size); - std::string value; - rep->GetPIK(ReadOptions(), ParsedInternalKey("k", 1, kTypeValue), &value, - [](void* arg, const MemTableRep::KeyValuePair& kv) { - *static_cast(arg) = kv.value.ToString(); - return false; - }); - ASSERT_EQ(value, "v"); - rep.reset(); - meta.fd = FileDescriptor(900000, 0, 0); - meta.fd.largest_seqno = kMaxSequenceNumber; - ASSERT_OK(factory->RecoverCrashSafeMemTableToSST(logical_path, &meta, tbo)); - ASSERT_OK(env_->DeleteFile(physical_path)); -} -#endif - -TEST_P(DBMemtableConvertTest, FileMmapCloseKeepsEveryFileNumber) { - if (std::string(std::get<1>(GetParam())) != "kFileMmap") return; - Options options = ConvertOptions(); - DestroyAndReopen(options); - ASSERT_OK(Put("head", "1")); - ASSERT_OK(dbfull()->TEST_SwitchMemtable()); - ASSERT_OK(Put("tail", "2")); - const auto registered = dbfull()->GetVersionSet()->GetMemTableFiles(); - ASSERT_EQ(registered.count(0), 1U); - ASSERT_EQ(registered.at(0).size(), 2U); - ObserveConversion(); - ASSERT_OK(db_->Close()); - Close(); - ASSERT_EQ(converts_.load(), 2); - for (uint64_t number : registered.at(0)) { - ASSERT_OK(env_->FileExists(MakeTableFileName(dbname_, number))); - } - ASSERT_OK(TryReopen(options)); - std::vector files; - db_->GetLiveFilesMetaData(&files); - ASSERT_EQ(files.size(), 2U); - for (const auto& file : files) { - ASSERT_EQ(registered.at(0).count(file.file_number), 1U); - } - ASSERT_EQ(Get("head"), "1"); - ASSERT_EQ(Get("tail"), "2"); -} - TEST_P(DBMemtableConvertTest, CloseConvertsAllMemtables) { CheckClose(false); } diff --git a/db/flush_job.cc b/db/flush_job.cc index cc24a440a..9370b1a32 100644 --- a/db/flush_job.cc +++ b/db/flush_job.cc @@ -209,12 +209,7 @@ void FlushJob::PickMemTable() { edit_->SetColumnFamily(cfd_->GetID()); // path 0 for level 0 file. - const uint64_t backing_file_number = - mems_.size() == 1 && m->SupportConvertToSST() - ? m->GetBackingFileNumber() : 0; - meta_.fd = FileDescriptor(backing_file_number != 0 - ? backing_file_number : versions_->NewFileNumber(), - 0, 0); + meta_.fd = FileDescriptor(versions_->NewFileNumber(), 0, 0); meta_.epoch_number = cfd_->NewEpochNumber(); base_ = cfd_->current(); @@ -237,13 +232,6 @@ Status FlushJob::Run(LogsWithPrepTracker* prep_tracker, FileMetaData* file_meta, if (db_options_.memtable_as_log_index) { mempurge_threshold = 0; // not supported } - for (const auto* mem : mems_) { - if (mem->IsFileRegistered()) { - // File-backed inputs must retire their registration with an SST edit. - mempurge_threshold = 0; - break; - } - } AutoThreadOperationStageUpdater stage_run(ThreadStatus::STAGE_FLUSH_RUN); if (mems_.empty()) { @@ -1000,10 +988,8 @@ Status FlushJob::WriteLevel0Table() { cfd_->GetName().c_str(), job_context_->job_id, meta_.fd.GetNumber(), memtable->ApproximateMemoryUsage(), s.ToString().c_str()); - // Rebuilding a file-backed table here would overwrite its registered - // source at the same path. Preserve it for crash recovery. - if (flush_reason_ != FlushReason::kShutDown && - memtable->GetBackingFileNumber() == 0) { + // Do not turn a failed close conversion into a full table rebuild. + if (flush_reason_ != FlushReason::kShutDown) { goto UseBuildTable; } } else { @@ -1091,17 +1077,10 @@ Status FlushJob::WriteLevel0Table() { } base_->Unref(); - // Zero-sized output does not become an SST in the manifest. + // Note that if file_size is zero, the file has been deleted and + // should not be added to the manifest. const bool has_output = meta_.fd.GetFileSize() > 0; - if (s.ok()) { - for (const auto* mem : mems_) { - if (mem->IsFileRegistered()) { - edit_->DeleteMemTableFile(mem->GetBackingFileNumber()); - } - } - } - if (s.ok() && has_output) { TEST_SYNC_POINT("DBImpl::FlushJob:SSTFileCreated"); // if we have more than 1 background thread, then we cannot diff --git a/db/memtable.cc b/db/memtable.cc index 5f18a7ba4..f8e7545bd 100644 --- a/db/memtable.cc +++ b/db/memtable.cc @@ -22,7 +22,6 @@ #include "db/range_tombstone_fragmenter.h" #include "db/read_callback.h" #include "db/wide/wide_column_serialization.h" -#include "file/filename.h" #include "logging/logging.h" #include "memory/arena.h" #include "memory/memory_usage.h" @@ -73,8 +72,7 @@ MemTable::MemTable(const InternalKeyComparator& cmp, const ImmutableOptions& ioptions, const MutableCFOptions& mutable_cf_options, WriteBufferManager* write_buffer_manager, - SequenceNumber latest_seq, uint32_t column_family_id, - uint64_t backing_file_number) + SequenceNumber latest_seq, uint32_t column_family_id) : comparator_(cmp), moptions_(ioptions, mutable_cf_options), refs_(0), @@ -88,9 +86,7 @@ MemTable::MemTable(const InternalKeyComparator& cmp, : nullptr, mutable_cf_options.memtable_huge_page_size), table_(ioptions.memtable_factory->CreateMemTableRep( - backing_file_number == 0 - ? std::string() - : TableFileName(ioptions.cf_paths, backing_file_number, 0), + ioptions.cf_paths[0].path, // level0_dir mutable_cf_options, comparator_, &arena_, mutable_cf_options.prefix_extractor.get(), ioptions.logger, column_family_id)), @@ -110,7 +106,6 @@ MemTable::MemTable(const InternalKeyComparator& cmp, flush_in_progress_(false), flush_completed_(false), file_number_(0), - backing_file_number_(table_->IsFileMmap() ? backing_file_number : 0), first_seqno_(0), earliest_seqno_(latest_seq), creation_seq_(latest_seq), diff --git a/db/memtable.h b/db/memtable.h index cf25794e6..e1a69ac26 100644 --- a/db/memtable.h +++ b/db/memtable.h @@ -154,8 +154,7 @@ class MemTable : public CacheAlignedNewDelete { const ImmutableOptions& ioptions, const MutableCFOptions& mutable_cf_options, WriteBufferManager* write_buffer_manager, - SequenceNumber earliest_seq, uint32_t column_family_id, - uint64_t backing_file_number = 0); + SequenceNumber earliest_seq, uint32_t column_family_id); // No copying allowed MemTable(const MemTable&) = delete; MemTable& operator=(const MemTable&) = delete; @@ -586,12 +585,6 @@ class MemTable : public CacheAlignedNewDelete { void SetFlushCompleted(bool completed) { flush_completed_ = completed; } uint64_t GetFileNumber() const { return file_number_; } - uint64_t GetBackingFileNumber() const { return backing_file_number_; } - bool IsFileRegistered() const { return file_registered_; } - void MarkFileRegistered() { - table_->MarkFileRegistered(); - file_registered_ = true; - } void SetFileNumber(uint64_t file_num) { file_number_ = file_num; } @@ -671,9 +664,7 @@ class MemTable : public CacheAlignedNewDelete { bool needs_user_key_cmp_in_get_; bool support_convert_to_sst_; bool reject_memtable_as_log_index_; - bool file_registered_ = false; uint64_t file_number_; // filled up after flush is complete - uint64_t backing_file_number_; // MANIFEST identity, separate from flush output // The updates to be applied to the transaction log when this // memtable is flushed to storage. diff --git a/db/memtable_list.cc b/db/memtable_list.cc index 27aec027d..c02bc5f9e 100644 --- a/db/memtable_list.cc +++ b/db/memtable_list.cc @@ -621,16 +621,10 @@ Status MemTableList::TryInstallMemtableFlushResults( const auto manifest_write_cb = [this, cfd, batch_count, log_buffer, to_delete, mu](const Status& status) { - if (status.ok() && !cfd->IsDropped()) { - cfd->PublishRegisteredMemTableCache(); - TEST_SYNC_POINT("FlushJob::AfterManifest"); - } RemoveMemTablesOrRestoreFlags(status, cfd, batch_count, log_buffer, to_delete, mu); }; if (write_edits) { - cfd->AddPendingMemTableFileEdits(edit_list.front()); - TEST_SYNC_POINT("FlushJob::BeforeManifest"); // this can release and reacquire the mutex. s = vset->LogAndApply(cfd, mutable_cf_options, read_options, edit_list, mu, db_directory, /*new_descriptor_log=*/false, @@ -899,13 +893,11 @@ Status InstallMemtableAtomicFlushResults( autovector> edit_lists; uint32_t num_entries = 0; - for (size_t k = 0; k < mems_list.size(); ++k) { - const auto mems = mems_list[k]; + for (const auto mems : mems_list) { assert(mems != nullptr); autovector edits; assert(!mems->empty()); edits.emplace_back((*mems)[0]->GetEdits()); - cfds[k]->AddPendingMemTableFileEdits(edits.front()); ++num_entries; edit_lists.emplace_back(edits); } @@ -941,16 +933,11 @@ Status InstallMemtableAtomicFlushResults( assert(0 == num_entries); } - TEST_SYNC_POINT("FlushJob::BeforeManifest"); // this can release and reacquire the mutex. s = vset->LogAndApply(cfds, mutable_cf_options_list, read_options, edit_lists, mu, db_directory); for (size_t k = 0; k != cfds.size(); ++k) { - if (s.ok() && !cfds[k]->IsDropped()) { - cfds[k]->PublishRegisteredMemTableCache(); - TEST_SYNC_POINT("FlushJob::AfterManifest"); - } auto* imm = (imm_lists == nullptr) ? cfds[k]->imm() : imm_lists->at(k); imm->InstallNewVersion(); } diff --git a/db/version_edit_handler.cc b/db/version_edit_handler.cc index 8e2defc11..fb32f4db4 100644 --- a/db/version_edit_handler.cc +++ b/db/version_edit_handler.cc @@ -496,8 +496,7 @@ ColumnFamilyData* VersionEditHandler::CreateCfAndInit( const ColumnFamilyOptions& cf_options, const VersionEdit& edit) { uint32_t cf_id = edit.GetColumnFamily(); ColumnFamilyData* cfd = - version_set_->CreateColumnFamily(cf_options, read_options_, &edit, - /*create_memtable=*/!cf_options.memtable_factory->SupportCrashSafe()); + version_set_->CreateColumnFamily(cf_options, read_options_, &edit); assert(cfd != nullptr); cfd->set_initialized(); assert(builders_.find(cf_id) == builders_.end()); diff --git a/db/version_set.cc b/db/version_set.cc index 089877d6c..ad13602ab 100644 --- a/db/version_set.cc +++ b/db/version_set.cc @@ -7646,7 +7646,7 @@ void VersionSet::ApplyMemTableFileEdit(const VersionEdit& edit) { ColumnFamilyData* VersionSet::CreateColumnFamily( const ColumnFamilyOptions& cf_options, const ReadOptions& read_options, - const VersionEdit* edit, bool create_memtable) { + const VersionEdit* edit) { assert(edit->IsColumnFamilyAdd()); MutableCFOptions dummy_cf_options; @@ -7671,10 +7671,8 @@ ColumnFamilyData* VersionSet::CreateColumnFamily( AppendVersion(new_cfd, v); // GetLatestMutableCFOptions() is safe here without mutex since the // cfd is not available to client - if (create_memtable) { - new_cfd->CreateNewMemtable(*new_cfd->GetLatestMutableCFOptions(), + new_cfd->CreateNewMemtable(*new_cfd->GetLatestMutableCFOptions(), LastSequence()); - } new_cfd->SetLogNumber(edit->GetLogNumber()); return new_cfd; } diff --git a/db/version_set.h b/db/version_set.h index 8a49959a2..987258d1c 100644 --- a/db/version_set.h +++ b/db/version_set.h @@ -1678,8 +1678,7 @@ class VersionSet { ColumnFamilyData* CreateColumnFamily(const ColumnFamilyOptions& cf_options, const ReadOptions& read_options, - const VersionEdit* edit, - bool create_memtable = true); + const VersionEdit* edit); Status VerifyFileMetadata(const ReadOptions& read_options, ColumnFamilyData* cfd, const std::string& fpath, diff --git a/include/rocksdb/memtablerep.h b/include/rocksdb/memtablerep.h index 965e864a5..9f2ae0de3 100644 --- a/include/rocksdb/memtablerep.h +++ b/include/rocksdb/memtablerep.h @@ -212,10 +212,6 @@ class MemTableRep : public CacheAlignedNewDelete { // of time. Otherwise, RocksDB may be blocked. virtual void MarkFlushed() {} - // Registered backing files are retained for the DB's MANIFEST lifecycle. - virtual void MarkFileRegistered() {} - virtual bool IsFileMmap() const { return false; } - struct KeyValuePair { Slice ukey; uint64_t tag; @@ -364,10 +360,8 @@ class MemTableRepFactory : public Customizable { uint32_t /* column_family_id */) { return CreateMemTableRep(key_cmp, allocator, slice_transform, logger); } - // The DB supplies the final SST path for a file-backed memtable. An empty - // path requests an anonymous in-memory representation. virtual MemTableRep* CreateMemTableRep( - const std::string& /*memtable_file_path*/, + const std::string& /*level0_dir*/, const MutableCFOptions&, const MemTableRep::KeyComparator& key_cmp, Allocator* allocator, const SliceTransform* slice_transform, Logger* logger, @@ -393,6 +387,19 @@ class MemTableRepFactory : public Customizable { // Default: false virtual bool SupportCrashSafe() const { return false; } + // Append leftover crash-safe mmap paths under cf_dir (plus factory chroot). + // Default: no leftovers. + virtual void ListCrashSafeLeftovers(const std::string& /*cf_dir*/, + std::vector* /*leftovers*/) { + } + + // Read-only leftover probe for Recover check. Must not truncate/rename. + // wal_dir is used to verify each WAL fileno can still be opened. + virtual Status ProbeCrashSafeLeftover(const std::string& /*path*/, + const std::string& /*wal_dir*/) const { + return Status::OK(); + } + // Load leftover, cap visible entries, truncate, ConvertToSST. The caller // supplies the visibility bound in meta->fd.largest_seqno; preserve it for // subsequent table readers. From 0fd5be37bd6e84a94ec9d21e72d479eb1760c39b Mon Sep 17 00:00:00 2001 From: leipeng Date: Sun, 4 Oct 2026 18:04:45 +0800 Subject: [PATCH 23/58] Revert "Track recoverable MemTable files in the MANIFEST" This reverts commit e1e7809c096a74a177c6bf3c6872a07dcdf242d6. --- db/version_edit.cc | 54 ---------- db/version_edit.h | 22 +--- db/version_edit_handler.cc | 14 +-- db/version_edit_test.cc | 50 --------- db/version_set.cc | 78 +------------- db/version_set.h | 10 -- db/version_set_test.cc | 204 ------------------------------------- 7 files changed, 3 insertions(+), 429 deletions(-) diff --git a/db/version_edit.cc b/db/version_edit.cc index 29d9c8090..b0af0d87a 100644 --- a/db/version_edit.cc +++ b/db/version_edit.cc @@ -86,9 +86,6 @@ void VersionEdit::Clear() { wal_additions_.clear(); wal_deletion_.Reset(); column_family_ = 0; - memtable_file_additions_.clear(); - memtable_file_deletions_.clear(); - has_memtable_file_tracking_ = false; is_column_family_add_ = false; is_column_family_drop_ = false; column_family_name_.clear(); @@ -291,17 +288,6 @@ bool VersionEdit::EncodeTo(std::string* dst, } // 0 is default and does not need to be explicitly written - if (has_memtable_file_tracking_) { - PutVarint32(dst, kMemTableFileTracking); - } - for (uint64_t number : memtable_file_additions_) { - if (number == 0 || number > kFileNumberMask) return false; - PutVarint32Varint64(dst, kMemTableFileAddition, number); - } - for (uint64_t number : memtable_file_deletions_) { - if (number == 0 || number > kFileNumberMask) return false; - PutVarint32Varint64(dst, kMemTableFileDeletion, number); - } if (column_family_ != 0) { PutVarint32Varint32(dst, kColumnFamily, column_family_); } @@ -782,24 +768,6 @@ Status VersionEdit::DecodeFrom(const Slice& src) { is_column_family_drop_ = true; break; - case kMemTableFileTracking: - has_memtable_file_tracking_ = true; - break; - - case kMemTableFileAddition: - case kMemTableFileDeletion: { - uint64_t number; - if (!GetVarint64(&input, &number) || number == 0 || - number > kFileNumberMask) { - msg = "invalid MemTable file number"; - } else if (tag == kMemTableFileAddition) { - memtable_file_additions_.insert(number); - } else { - memtable_file_deletions_.insert(number); - } - break; - } - case kInAtomicGroup: is_in_atomic_group_ = true; if (!GetVarint32(&input, &remaining_entries_)) { @@ -980,15 +948,6 @@ std::string VersionEdit::DebugString(bool hex_key) const { r.append("\n ColumnFamily: "); AppendNumberTo(&r, column_family_); - if (has_memtable_file_tracking_) r.append("\n MemTableFileTracking: true"); - for (uint64_t number : memtable_file_additions_) { - r.append("\n AddMemTableFile: "); - AppendNumberTo(&r, number); - } - for (uint64_t number : memtable_file_deletions_) { - r.append("\n DeleteMemTableFile: "); - AppendNumberTo(&r, number); - } if (is_column_family_add_) { r.append("\n ColumnFamilyAdd: "); r.append(column_family_name_); @@ -1139,19 +1098,6 @@ std::string VersionEdit::DebugJSON(int edit_num, bool hex_key) const { } jw << "ColumnFamily" << column_family_; - if (has_memtable_file_tracking_) jw << "MemTableFileTracking" << true; - if (!memtable_file_additions_.empty()) { - jw << "MemTableFileAdditions"; - jw.StartArray(); - for (uint64_t number : memtable_file_additions_) jw << number; - jw.EndArray(); - } - if (!memtable_file_deletions_.empty()) { - jw << "MemTableFileDeletions"; - jw.StartArray(); - for (uint64_t number : memtable_file_deletions_) jw << number; - jw.EndArray(); - } if (is_column_family_add_) { jw << "ColumnFamilyAdd" << column_family_name_; diff --git a/db/version_edit.h b/db/version_edit.h index 1df20c5f1..ce35758d4 100644 --- a/db/version_edit.h +++ b/db/version_edit.h @@ -63,11 +63,6 @@ enum Tag : uint32_t { kBlobFileAddition = 400, kBlobFileGarbage, - // Required recovery metadata: older readers must reject these tags. - kMemTableFileAddition = 500, - kMemTableFileDeletion = 501, - kMemTableFileTracking = 502, - // Mask for an unidentified tag from the future which can be safely ignored. kTagSafeIgnoreMask = 1 << 13, @@ -657,8 +652,7 @@ class VersionEdit { size_t NumEntries() const { return new_files_.size() + deleted_files_.size() + blob_file_additions_.size() + blob_file_garbages_.size() + - wal_additions_.size() + !wal_deletion_.IsEmpty() + - memtable_file_additions_.size() + memtable_file_deletions_.size(); + wal_additions_.size() + !wal_deletion_.IsEmpty(); } void SetColumnFamily(uint32_t column_family_id) { @@ -666,17 +660,6 @@ class VersionEdit { } uint32_t GetColumnFamily() const { return column_family_; } - void AddMemTableFile(uint64_t number) { memtable_file_additions_.insert(number); } - void DeleteMemTableFile(uint64_t number) { memtable_file_deletions_.insert(number); } - const std::set& GetMemTableFileAdditions() const { - return memtable_file_additions_; - } - const std::set& GetMemTableFileDeletions() const { - return memtable_file_deletions_; - } - void SetMemTableFileTracking() { has_memtable_file_tracking_ = true; } - bool HasMemTableFileTracking() const { return has_memtable_file_tracking_; } - const std::string& GetColumnFamilyName() const { return column_family_name_; } // set column family ID by calling SetColumnFamily() @@ -791,9 +774,6 @@ class VersionEdit { // Each version edit record should have column_family_ set // If it's not set, it is default (0) uint32_t column_family_ = 0; - std::set memtable_file_additions_; - std::set memtable_file_deletions_; - bool has_memtable_file_tracking_ = false; // a version edit can be either column_family add or // column_family drop. If it's column family add, // it also includes column family name. diff --git a/db/version_edit_handler.cc b/db/version_edit_handler.cc index fb32f4db4..395d066d7 100644 --- a/db/version_edit_handler.cc +++ b/db/version_edit_handler.cc @@ -213,7 +213,6 @@ Status VersionEditHandler::ApplyVersionEdit(VersionEdit& edit, if (s.ok()) { assert(cfd != nullptr); s = ExtractInfoFromVersionEdit(*cfd, edit); - if (s.ok()) version_set_->ApplyMemTableFileEdit(edit); } return s; } @@ -641,18 +640,7 @@ Status VersionEditHandler::ExtractInfoFromVersionEdit(ColumnFamilyData* cfd, version_edit_params_.SetPrevLogNumber(edit.GetPrevLogNumber()); } if (edit.HasNextFile()) { - version_edit_params_.SetNextFile( - std::max(version_edit_params_.GetNextFile(), edit.GetNextFile())); - } - uint64_t next_file = version_edit_params_.GetNextFile(); - for (uint64_t number : edit.GetMemTableFileAdditions()) { - next_file = std::max(next_file, number + 1); - } - for (uint64_t number : edit.GetMemTableFileDeletions()) { - next_file = std::max(next_file, number + 1); - } - if (next_file != version_edit_params_.GetNextFile()) { - version_edit_params_.SetNextFile(next_file); + version_edit_params_.SetNextFile(edit.GetNextFile()); } if (edit.HasMaxColumnFamily()) { version_edit_params_.SetMaxColumnFamily(edit.GetMaxColumnFamily()); diff --git a/db/version_edit_test.cc b/db/version_edit_test.cc index 8767ef49e..252352069 100644 --- a/db/version_edit_test.cc +++ b/db/version_edit_test.cc @@ -40,56 +40,6 @@ static void TestEncodeDecode(const VersionEdit& edit) { class VersionEditTest : public testing::Test {}; -TEST_F(VersionEditTest, MemTableFilesEncodeDecodeAndClear) { - VersionEdit edit; - edit.SetColumnFamily(7); - edit.SetMemTableFileTracking(); - edit.AddMemTableFile(123); - edit.AddMemTableFile(kFileNumberMask); - edit.DeleteMemTableFile(42); - edit.MarkAtomicGroup(0); - std::string encoded; - ASSERT_TRUE(edit.EncodeTo(&encoded)); - VersionEdit parsed; - ASSERT_OK(parsed.DecodeFrom(encoded)); - ASSERT_EQ(parsed.GetColumnFamily(), 7U); - ASSERT_TRUE(parsed.HasMemTableFileTracking()); - ASSERT_EQ(parsed.GetMemTableFileAdditions(), edit.GetMemTableFileAdditions()); - ASSERT_EQ(parsed.GetMemTableFileDeletions(), edit.GetMemTableFileDeletions()); - ASSERT_TRUE(parsed.IsInAtomicGroup()); - ASSERT_EQ(parsed.GetRemainingEntries(), 0U); - parsed.Clear(); - ASSERT_FALSE(parsed.HasMemTableFileTracking()); - ASSERT_TRUE(parsed.GetMemTableFileAdditions().empty()); - ASSERT_TRUE(parsed.GetMemTableFileDeletions().empty()); - ASSERT_FALSE(parsed.IsInAtomicGroup()); -} - -TEST_F(VersionEditTest, MemTableFilesRejectInvalidNumbers) { - for (uint32_t tag : {uint32_t(kMemTableFileAddition), - uint32_t(kMemTableFileDeletion)}) { - for (uint64_t number : {uint64_t(0), kFileNumberMask + 1}) { - std::string encoded; - PutVarint32Varint64(&encoded, tag, number); - VersionEdit parsed; - ASSERT_TRUE(parsed.DecodeFrom(encoded).IsCorruption()); - VersionEdit invalid; - if (tag == kMemTableFileAddition) invalid.AddMemTableFile(number); - else invalid.DeleteMemTableFile(number); - encoded.clear(); - ASSERT_FALSE(invalid.EncodeTo(&encoded)); - } - std::string truncated; - PutVarint32(&truncated, tag); - truncated.push_back(char(0x80)); - VersionEdit parsed; - ASSERT_TRUE(parsed.DecodeFrom(truncated).IsCorruption()); - } - ASSERT_EQ(kMemTableFileAddition & kTagSafeIgnoreMask, 0U); - ASSERT_EQ(kMemTableFileDeletion & kTagSafeIgnoreMask, 0U); - ASSERT_EQ(kMemTableFileTracking & kTagSafeIgnoreMask, 0U); -} - TEST_F(VersionEditTest, EncodeDecode) { static const uint64_t kBig = 1ull << 50; static const uint32_t kBig32Bit = 1ull << 30; diff --git a/db/version_set.cc b/db/version_set.cc index ad13602ab..573c57e97 100644 --- a/db/version_set.cc +++ b/db/version_set.cc @@ -5505,8 +5505,6 @@ void VersionSet::Reset() { obsolete_files_.clear(); obsolete_manifests_.clear(); wals_.Reset(); - memtable_files_.clear(); - has_memtable_file_tracking_ = false; } void VersionSet::AppendVersion(ColumnFamilyData* column_family_data, @@ -5745,7 +5743,6 @@ Status VersionSet::ProcessManifestWrites( std::unordered_map curr_state; VersionEdit wal_additions; if (new_descriptor_log) { - if (has_memtable_file_tracking_) wal_additions.SetMemTableFileTracking(); pending_manifest_file_number_ = NewFileNumber(); batch_edits.back()->SetNextFile(next_file_number_.load()); @@ -5760,10 +5757,6 @@ Status VersionSet::ProcessManifestWrites( curr_state.emplace( cfd->GetID(), MutableCFState(cfd->GetLogNumber(), cfd->GetFullHistoryTsLow())); - auto memtable_iter = memtable_files_.find(cfd->GetID()); - if (memtable_iter != memtable_files_.end()) { - curr_state.at(cfd->GetID()).memtable_files = memtable_iter->second; - } } for (const auto& wal : wals_.GetWals()) { @@ -5963,45 +5956,6 @@ Status VersionSet::ProcessManifestWrites( // Install the new versions if (s.ok()) { - // Only a committed live batch can retire backing files. MANIFEST replay - // applies registry edits without scheduling historical deletions. - std::map retired_memtable_files; - for (const auto* edit : batch_edits) { - auto files = memtable_files_.find(edit->GetColumnFamily()); - if (files != memtable_files_.end()) { - auto* cfd = column_family_set_->GetColumnFamily(edit->GetColumnFamily()); - assert(cfd != nullptr); - const auto& path = cfd->ioptions()->cf_paths[0].path; - if (edit->IsColumnFamilyDrop()) { - for (uint64_t number : files->second) { - retired_memtable_files.emplace(number, path); - } - } else { - for (uint64_t number : edit->GetMemTableFileDeletions()) { - if (files->second.count(number)) { - retired_memtable_files.emplace(number, path); - } - } - } - } - ApplyMemTableFileEdit(*edit); - } - if (!retired_memtable_files.empty()) { - // In-place conversion transfers ownership to the installed SST version. - for (const auto* edit : batch_edits) { - for (const auto& file : edit->GetNewFiles()) { - retired_memtable_files.erase(file.second.fd.GetNumber()); - } - } - for (const auto& cf : memtable_files_) { - for (uint64_t number : cf.second) retired_memtable_files.erase(number); - } - for (const auto& file : retired_memtable_files) { - auto* metadata = new FileMetaData(); - metadata->fd = FileDescriptor(file.first, 0, 0); - obsolete_files_.emplace_back(metadata, file.second); - } - } if (first_writer.edit_list.front()->IsColumnFamilyAdd()) { assert(batch_edits.size() == 1); assert(new_cf_options != nullptr); @@ -6896,8 +6850,7 @@ Status VersionSet::WriteCurrentStateToManifest( } // Save WALs. - if (!wal_additions.GetWalAdditions().empty() || - wal_additions.HasMemTableFileTracking()) { + if (!wal_additions.GetWalAdditions().empty()) { TEST_SYNC_POINT_CALLBACK("VersionSet::WriteCurrentStateToManifest:SaveWal", const_cast(&wal_additions)); std::string record; @@ -6963,10 +6916,6 @@ Status VersionSet::WriteCurrentStateToManifest( VersionEdit edit; edit.SetColumnFamily(cfd->GetID()); - for (uint64_t number : curr_state.at(cfd->GetID()).memtable_files) { - edit.AddMemTableFile(number); - } - const auto* current = cfd->current(); assert(current); @@ -7619,31 +7568,6 @@ uint64_t VersionSet::GetObsoleteSstFilesSize() const { return ret; } -void VersionSet::ApplyMemTableFileEdit(const VersionEdit& edit) { - has_memtable_file_tracking_ |= edit.HasMemTableFileTracking(); - // Live allocation can race with MANIFEST installation; never move the - // atomic allocator backwards when reserving recovered/persisted numbers. - auto reserve_number = [this](uint64_t number) { - uint64_t next = next_file_number_.load(std::memory_order_relaxed); - while (next <= number && - !next_file_number_.compare_exchange_weak( - next, number + 1, std::memory_order_relaxed)) {} - }; - for (uint64_t number : edit.GetMemTableFileAdditions()) reserve_number(number); - for (uint64_t number : edit.GetMemTableFileDeletions()) reserve_number(number); - if (edit.IsColumnFamilyDrop()) { - memtable_files_.erase(edit.GetColumnFamily()); - return; - } - for (uint64_t number : edit.GetMemTableFileDeletions()) { - auto iter = memtable_files_.find(edit.GetColumnFamily()); - if (iter != memtable_files_.end()) iter->second.erase(number); - } - for (uint64_t number : edit.GetMemTableFileAdditions()) { - memtable_files_[edit.GetColumnFamily()].insert(number); - } -} - ColumnFamilyData* VersionSet::CreateColumnFamily( const ColumnFamilyOptions& cf_options, const ReadOptions& read_options, const VersionEdit* edit) { diff --git a/db/version_set.h b/db/version_set.h index 987258d1c..2c4b3bf7e 100644 --- a/db/version_set.h +++ b/db/version_set.h @@ -1370,13 +1370,6 @@ class VersionSet { // Allocate and return a new file number uint64_t NewFileNumber() { return next_file_number_.fetch_add(1); } - // Access requires the DB mutex, like other MANIFEST state. - const std::map>& GetMemTableFiles() const { - return memtable_files_; - } - bool HasMemTableFileTracking() const { return has_memtable_file_tracking_; } - void ApplyMemTableFileEdit(const VersionEdit& edit); - // Fetch And Add n new file number uint64_t FetchAddFileNumber(uint64_t n) { return next_file_number_.fetch_add(n); @@ -1662,7 +1655,6 @@ class VersionSet { struct MutableCFState { uint64_t log_number; std::string full_history_ts_low; - std::set memtable_files; explicit MutableCFState() = default; explicit MutableCFState(uint64_t _log_number, std::string ts_low) @@ -1686,8 +1678,6 @@ class VersionSet { // Protected by DB mutex. WalSet wals_; - std::map> memtable_files_; - bool has_memtable_file_tracking_ = false; std::unique_ptr column_family_set_; Cache* table_cache_; diff --git a/db/version_set_test.cc b/db/version_set_test.cc index f329ccb31..4703b1f44 100644 --- a/db/version_set_test.cc +++ b/db/version_set_test.cc @@ -1505,139 +1505,6 @@ TEST_F(VersionSetTest, SameColumnFamilyGroupCommit) { EXPECT_EQ(kGroupSize - 1, count); } -TEST_F(VersionSetTest, MemTableRegistryManifestRollover) { - NewDB(); - ASSERT_FALSE(versions_->HasMemTableFileTracking()); - VersionEdit addition; - addition.SetMemTableFileTracking(); - addition.AddMemTableFile(10000); - addition.AddMemTableFile(10001); - ASSERT_OK(LogAndApplyToDefaultCF(addition)); - ASSERT_EQ(versions_->GetMemTableFiles().at(0), - std::set({10000, 10001})); - CreateNewManifest(); - ReopenDB(); - ASSERT_TRUE(versions_->HasMemTableFileTracking()); - ASSERT_EQ(versions_->GetMemTableFiles().at(0), - std::set({10000, 10001})); - ASSERT_GT(versions_->current_next_file_number(), 10001U); - VersionEdit deletion; - deletion.DeleteMemTableFile(10000); - deletion.DeleteMemTableFile(10001); - ASSERT_OK(LogAndApplyToDefaultCF(deletion)); - CreateNewManifest(); - ReopenDB(); - ASSERT_TRUE(versions_->HasMemTableFileTracking()); - for (const auto& entry : versions_->GetMemTableFiles()) { - ASSERT_TRUE(entry.second.empty()); - } -} - -TEST_F(VersionSetTest, MemTableRegistryFailedCommitDoesNotEnableTracking) { - NewDB(); - VersionEdit invalid; - invalid.SetMemTableFileTracking(); - invalid.AddMemTableFile(0); - ASSERT_TRUE(LogAndApplyToDefaultCF(invalid).IsCorruption()); - ASSERT_FALSE(versions_->HasMemTableFileTracking()); - ASSERT_TRUE(versions_->GetMemTableFiles().empty()); -} - -TEST_F(VersionSetTest, MemTableRegistryRetirementQueuesObsoleteFile) { - NewDB(); - VersionEdit addition; - addition.SetMemTableFileTracking(); - addition.AddMemTableFile(10000); - ASSERT_OK(LogAndApplyToDefaultCF(addition)); - VersionEdit deletion; - deletion.DeleteMemTableFile(10000); - deletion.DeleteMemTableFile(10001); // Never registered; do not schedule it. - ASSERT_OK(LogAndApplyToDefaultCF(deletion)); - std::vector tables; - std::vector blobs; - std::vector manifests; - versions_->GetObsoleteFiles(&tables, &blobs, &manifests, 10000); - ASSERT_TRUE(tables.empty()); - manifests.clear(); - versions_->GetObsoleteFiles(&tables, &blobs, &manifests, 10001); - ASSERT_EQ(tables.size(), 1U); - ASSERT_EQ(tables[0].metadata->fd.GetNumber(), 10000U); - ASSERT_EQ(tables[0].path, - versions_->GetColumnFamilySet()->GetDefault()->ioptions()->cf_paths[0].path); - ASSERT_EQ(tables[0].metadata->table_reader_handle, nullptr); - tables[0].DeleteMetadata(); -} - -TEST_F(VersionSetTest, MemTableRegistryBatchReregistrationDoesNotRetire) { - NewDB(); - VersionEdit addition; - addition.SetMemTableFileTracking(); - addition.AddMemTableFile(10000); - ASSERT_OK(LogAndApplyToDefaultCF(addition)); - autovector> edits; - edits.emplace_back(new VersionEdit); - edits.back()->DeleteMemTableFile(10000); - edits.emplace_back(new VersionEdit); - edits.back()->AddMemTableFile(10000); - ASSERT_OK(LogAndApplyToDefaultCF(edits)); - ASSERT_EQ(versions_->GetMemTableFiles().at(0), std::set({10000})); - std::vector tables; - std::vector blobs; - std::vector manifests; - versions_->GetObsoleteFiles(&tables, &blobs, &manifests, 10001); - ASSERT_TRUE(tables.empty()); -} - -TEST_F(VersionSetTest, MemTableRegistryDropAndDiscardedEdit) { - NewDB(); - auto* cfd = versions_->GetColumnFamilySet()->GetColumnFamily(1); - ASSERT_NE(cfd, nullptr); - cfd->Ref(); - VersionEdit addition; - addition.SetColumnFamily(1); - addition.SetMemTableFileTracking(); - addition.AddMemTableFile(10000); - mutex_.Lock(); - Status status = versions_->LogAndApply( - cfd, mutable_cf_options_, read_options_, &addition, &mutex_, nullptr); - mutex_.Unlock(); - ASSERT_OK(status); - ASSERT_EQ(versions_->GetMemTableFiles().at(1), std::set({10000})); - VersionEdit drop; - drop.SetColumnFamily(1); - drop.DropColumnFamily(); - mutex_.Lock(); - status = versions_->LogAndApply( - cfd, mutable_cf_options_, read_options_, &drop, &mutex_, nullptr); - mutex_.Unlock(); - ASSERT_OK(status); - ASSERT_EQ(versions_->GetMemTableFiles().count(1), 0U); - std::vector retired_tables; - std::vector retired_blobs; - std::vector retired_manifests; - versions_->GetObsoleteFiles(&retired_tables, &retired_blobs, - &retired_manifests, 10001); - ASSERT_EQ(retired_tables.size(), 1U); - ASSERT_EQ(retired_tables[0].metadata->fd.GetNumber(), 10000U); - ASSERT_EQ(retired_tables[0].path, cfd->ioptions()->cf_paths[0].path); - retired_tables[0].DeleteMetadata(); - VersionEdit discarded; - discarded.SetColumnFamily(1); - discarded.AddMemTableFile(10001); - mutex_.Lock(); - status = versions_->LogAndApply( - cfd, mutable_cf_options_, read_options_, &discarded, &mutex_, nullptr); - cfd->UnrefAndTryDelete(); - mutex_.Unlock(); - ASSERT_TRUE(status.IsColumnFamilyDropped()); - ASSERT_EQ(versions_->GetMemTableFiles().count(1), 0U); - ASSERT_TRUE(versions_->HasMemTableFileTracking()); - CreateNewManifest(); - ReopenDB(); - ASSERT_TRUE(versions_->HasMemTableFileTracking()); - ASSERT_EQ(versions_->GetMemTableFiles().count(1), 0U); -} - TEST_F(VersionSetTest, PersistBlobFileStateInNewManifest) { // Initialize the database and add a couple of blob files, one with some // garbage in it, and one without any garbage. @@ -2757,54 +2624,6 @@ TEST_F(VersionSetAtomicGroupTest, EXPECT_EQ(num_initial_edits_ + kAtomicGroupSize, num_recovered_edits_); } -TEST_F(VersionSetAtomicGroupTest, MemTableRegistryAtomicTransition) { - SetupValidAtomicGroup(3); - edits_[0].SetMemTableFileTracking(); - edits_[0].AddMemTableFile(123); - edits_[1].DeleteMemTableFile(123); - edits_[1].AddMemTableFile(124); - // A later next-file record must not erase the allocator reservation. - edits_[2].SetNextFile(2); - AddNewEditsToLog(3); - ASSERT_OK(versions_->Recover(column_families_, false)); - ASSERT_TRUE(versions_->HasMemTableFileTracking()); - ASSERT_EQ(versions_->GetMemTableFiles().at(0), std::set({124})); - ASSERT_GT(versions_->current_next_file_number(), 124U); - auto* cfd = versions_->GetColumnFamilySet()->GetColumnFamily(0); - for (int level = 0; level < cfd->NumberLevels(); ++level) { - ASSERT_TRUE(cfd->current()->storage_info()->LevelFiles(level).empty()); - } - std::vector tables; - std::vector blobs; - std::vector manifests; - versions_->GetObsoleteFiles(&tables, &blobs, &manifests, 125); - ASSERT_TRUE(tables.empty()); // Historical replay must never schedule GC. -} - -TEST_F(VersionSetAtomicGroupTest, MemTableRegistryIncompleteAtomicGroup) { - SetupIncompleteTrailingAtomicGroup(3); - edits_[0].SetMemTableFileTracking(); - edits_[0].AddMemTableFile(123); - edits_[1].DeleteMemTableFile(123); - edits_[1].AddMemTableFile(124); - AddNewEditsToLog(2); - ASSERT_OK(versions_->Recover(column_families_, false)); - ASSERT_FALSE(versions_->HasMemTableFileTracking()); - ASSERT_TRUE(versions_->GetMemTableFiles().empty()); -} - -TEST_F(VersionSetAtomicGroupTest, MemTableRegistryTrackingSurvivesEmptyList) { - SetupValidAtomicGroup(3); - edits_[0].SetMemTableFileTracking(); - edits_[0].AddMemTableFile(123); - edits_[1].DeleteMemTableFile(123); - AddNewEditsToLog(3); - ASSERT_OK(versions_->Recover(column_families_, false)); - ASSERT_TRUE(versions_->HasMemTableFileTracking()); - ASSERT_TRUE(versions_->GetMemTableFiles().at(0).empty()); - ASSERT_GT(versions_->current_next_file_number(), 123U); -} - TEST_F(VersionSetAtomicGroupTest, HandleValidAtomicGroupWithReactiveVersionSetReadAndApply) { const int kAtomicGroupSize = 3; @@ -3862,29 +3681,6 @@ TEST_F(VersionSetTestMissingFiles, NoFileMissing) { } } -TEST_F(VersionSetTestMissingFiles, MemTableRegistryConversionKeepsSameNumberSst) { - NewDB(); - VersionEdit addition; - addition.SetMemTableFileTracking(); - addition.AddMemTableFile(100); - ASSERT_OK(LogAndApplyToDefaultCF(addition)); - SstInfo sst(100, kDefaultColumnFamilyName, "a", 0, 100); - std::vector file_metas; - CreateDummyTableFiles({sst}, &file_metas); - VersionEdit conversion; - conversion.DeleteMemTableFile(100); - conversion.AddFile(0, file_metas[0]); - ASSERT_OK(LogAndApplyToDefaultCF(conversion)); - ASSERT_TRUE(versions_->GetMemTableFiles().at(0).empty()); - std::vector tables; - std::vector blobs; - std::vector manifests; - versions_->GetObsoleteFiles(&tables, &blobs, &manifests, 101); - ASSERT_TRUE(tables.empty()); - ASSERT_EQ(versions_->GetColumnFamilySet()->GetDefault()->current() - ->storage_info()->LevelFiles(0).size(), 1U); -} - TEST_F(VersionSetTestMissingFiles, MinLogNumberToKeep2PC) { db_options_.allow_2pc = true; NewDB(); From afd4095745b36d73e55061572c786904a5e01476 Mon Sep 17 00:00:00 2001 From: leipeng Date: Sun, 4 Oct 2026 12:42:06 +0800 Subject: [PATCH 24/58] Test immutable MemTable conversion modes Conversion-mode changes must not invalidate the configuration assumptions made when a factory is created. Both MemTable factories and their table factory wrappers expose runtime updates, so the same restriction must hold through both entry points without partially changing other options. Other runtime tuning and repeated submission of the current mode remain valid; a dangerous-update opt-in no longer enables mode transitions. --- db/db_cspp_crash_safe_test.cc | 75 +++++++---------------------------- 1 file changed, 14 insertions(+), 61 deletions(-) diff --git a/db/db_cspp_crash_safe_test.cc b/db/db_cspp_crash_safe_test.cc index aa0268ea4..4539e6a63 100644 --- a/db/db_cspp_crash_safe_test.cc +++ b/db/db_cspp_crash_safe_test.cc @@ -243,11 +243,11 @@ TEST_F(DBCsppCrashSafeTest, CrashSafeRequiresFileMmap) { } } -TEST_F(DBCsppCrashSafeTest, DangerousFactoryUpdate) { +TEST_F(DBCsppCrashSafeTest, ImmutableFactoryConvertMode) { Close(); const SidePluginRepo repo; const json query = {{"html", false}}; - auto check = [&](const auto& factory, const auto* manip, bool allowed) { + auto check = [&](const auto& factory, const auto* manip, const char* mode) { auto state = [&] { return json::parse(manip->ToString(*factory, query, repo)); }; @@ -259,86 +259,39 @@ TEST_F(DBCsppCrashSafeTest, DangerousFactoryUpdate) { return s; } }; - ASSERT_EQ(state()["convert_to_sst"], "kFileMmap"); - ASSERT_EQ(state()["allow_dangerous_update"], allowed); + ASSERT_EQ(state()["convert_to_sst"], mode); ASSERT_OK(update({{"token_use_idle", false}})); ASSERT_EQ(state()["token_use_idle"], false); - ASSERT_OK(update({{"convert_to_sst", "kFileMmap"}})); - + ASSERT_OK(update({{"convert_to_sst", mode}})); const json before = state(); - Status s = update({{"allow_dangerous_update", !allowed}, - {"token_use_idle", true}, {"populate_read", false}}); - ASSERT_TRUE(s.IsInvalidArgument()); - ASSERT_NE(s.ToString().find("cannot be changed online"), std::string::npos); - ASSERT_EQ(state(), before); - s = update({{"allow_dangerous_update", !allowed}, - {"convert_to_sst", "kDontConvert"}}); - ASSERT_TRUE(s.IsInvalidArgument()); - ASSERT_EQ(state(), before); - s = update({{"convert_to_sst", "invalid"}, {"token_use_idle", true}}); - ASSERT_TRUE(s.IsInvalidArgument()); - ASSERT_EQ(state(), before); - ASSERT_OK(update({{"allow_dangerous_update", allowed}, - {"convert_to_sst", "kFileMmap"}})); - if (!allowed) { - s = update({{"convert_to_sst", "kDontConvert"}, - {"token_use_idle", true}, {"populate_read", false}}); + for (const char* other : {"kDontConvert", "kDumpMem", "kFileMmap", "invalid"}) { + if (std::string(other) == mode) continue; + SCOPED_TRACE(other); + Status s = update({{"convert_to_sst", other}, + {"token_use_idle", true}, {"populate_read", false}}); ASSERT_TRUE(s.IsInvalidArgument()); - ASSERT_NE(s.ToString().find("allow_dangerous_update=true"), std::string::npos); ASSERT_EQ(state(), before); - return; } - ASSERT_OK(update({{"convert_to_sst", "kDumpMem"}})); - ASSERT_EQ(state()["convert_to_sst"], "kDumpMem"); - ASSERT_OK(update({{"allow_dangerous_update", true}, - {"convert_to_sst", "kDontConvert"}})); - ASSERT_EQ(state()["convert_to_sst"], "kDontConvert"); - ASSERT_OK(update({{"convert_to_sst", "kFileMmap"}})); - ASSERT_EQ(state()["convert_to_sst"], "kFileMmap"); - ASSERT_EQ(state()["allow_dangerous_update"], true); }; - for (bool allowed : {false, true}) { - SCOPED_TRACE(allowed); - json params = {{"mem_cap", 16777216}, {"convert_to_sst", "kFileMmap"}}; - if (allowed) params["allow_dangerous_update"] = true; + for (const char* mode : {"kDontConvert", "kDumpMem", "kFileMmap"}) { + SCOPED_TRACE(mode); + const json params = {{"mem_cap", 16777216}, {"convert_to_sst", mode}}; for (const char* cls : {"CSPPMemTab", "OffsetSkipList"}) { SCOPED_TRACE(cls); auto factory = PluginFactorySP::AcquirePlugin( cls, params, repo); auto* manip = PluginManip::AcquirePlugin(cls, {}, repo); - check(factory, manip, allowed); + check(factory, manip, mode); } for (const char* cls : {"CSPPMemTabTable", "OffsetSkipListTable"}) { SCOPED_TRACE(cls); auto factory = PluginFactorySP::AcquirePlugin(cls, params, repo); auto* manip = PluginManip::AcquirePlugin(cls, {}, repo); - check(factory, manip, allowed); + check(factory, manip, mode); } } } -TEST_F(DBCsppCrashSafeTest, DangerousUpdateAffectsOnlyNewMemtables) { - Close(); - const SidePluginRepo repo; - InternalKeyComparator icmp(BytewiseComparator()); - MemTable::KeyComparator cmp(icmp); - for (const char* cls : {"CSPPMemTab", "OffsetSkipList"}) { - SCOPED_TRACE(cls); - Arena arena; - auto factory = PluginFactorySP::AcquirePlugin( - cls, {{"mem_cap", 16777216}, {"allow_dangerous_update", true}}, repo); - auto* manip = PluginManip::AcquirePlugin(cls, {}, repo); - std::unique_ptr before( - factory->CreateMemTableRep(cmp, &arena, nullptr, nullptr)); - ASSERT_FALSE(before->SupportConvertToSST()); - manip->Update(factory.get(), {}, {{"convert_to_sst", "kDumpMem"}}, repo); - std::unique_ptr after( - factory->CreateMemTableRep(cmp, &arena, nullptr, nullptr)); - ASSERT_FALSE(before->SupportConvertToSST()); - ASSERT_TRUE(after->SupportConvertToSST()); - } -} - TEST_F(DBCsppCrashSafeTest, RecoveredTableUsesFileSequenceBound) { Close(); for (bool osl : {false, true}) { From 0b1a3d481957df998f91ea90bca48e09a6bd2f23 Mon Sep 17 00:00:00 2001 From: leipeng Date: Sun, 4 Oct 2026 14:19:01 +0800 Subject: [PATCH 25/58] Fail flush when ConvertToSST fails ConvertToSST avoids the cost of rebuilding a table. Falling back to BuildTable after conversion fails hides the original error and unexpectedly performs that expensive work. Preserve the conversion failure for the caller to handle. --- db/flush_job.cc | 5 ----- 1 file changed, 5 deletions(-) diff --git a/db/flush_job.cc b/db/flush_job.cc index 9370b1a32..f1e67ca3d 100644 --- a/db/flush_job.cc +++ b/db/flush_job.cc @@ -988,10 +988,6 @@ Status FlushJob::WriteLevel0Table() { cfd_->GetName().c_str(), job_context_->job_id, meta_.fd.GetNumber(), memtable->ApproximateMemoryUsage(), s.ToString().c_str()); - // Do not turn a failed close conversion into a full table rebuild. - if (flush_reason_ != FlushReason::kShutDown) { - goto UseBuildTable; - } } else { meta_.fd.smallest_seqno = std::min(memtable->GetEarliestSequenceNumber(), memtable->GetFirstSequenceNumber()); @@ -1004,7 +1000,6 @@ Status FlushJob::WriteLevel0Table() { memtables.clear(); } else { // call BuildTable -UseBuildTable: uint64_t num_input_entries = 0; uint64_t memtable_payload_bytes = 0; uint64_t memtable_garbage_bytes = 0; From 63ebb2e77ab365d23970aac8c05c8153472ac9e2 Mon Sep 17 00:00:00 2001 From: leipeng Date: Sun, 4 Oct 2026 17:26:15 +0800 Subject: [PATCH 26/58] Use cached MemTable conversion capability MemTable already caches its conversion capability at construction. Reusing that flag avoids an unnecessary pointer chase on each capability check. --- db/memtable.h | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/db/memtable.h b/db/memtable.h index e1a69ac26..aaa19e326 100644 --- a/db/memtable.h +++ b/db/memtable.h @@ -550,7 +550,7 @@ class MemTable : public CacheAlignedNewDelete { void FinishHint(void* hint) const { table_->FinishHint(hint); } bool SupportConvertToSST() const { - return table_->SupportConvertToSST() && is_range_del_table_empty_; + return support_convert_to_sst_ && is_range_del_table_empty_; } Status ConvertToSST(struct FileMetaData*, const struct TableBuilderOptions&); From eaeba37a0009e18fd2d71556deca416e0407e263 Mon Sep 17 00:00:00 2001 From: leipeng Date: Mon, 5 Oct 2026 20:31:32 +0800 Subject: [PATCH 27/58] Restore FastRange64 for lock stripe selection Lock stripe counts are runtime values, so modulo adds integer division to the transaction lock hot path. It also changes the hash bits used for stripe selection: with 16 stripes, k1, k2, and k3 all map to the same stripe, introducing mutex contention between unrelated keys and causing zero-timeout locking to fail. --- utilities/transactions/lock/point/point_lock_manager.cc | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/utilities/transactions/lock/point/point_lock_manager.cc b/utilities/transactions/lock/point/point_lock_manager.cc index 03845958c..43ac47389 100644 --- a/utilities/transactions/lock/point/point_lock_manager.cc +++ b/utilities/transactions/lock/point/point_lock_manager.cc @@ -309,7 +309,7 @@ PointLockManager::PointLockManager(PessimisticTransactionDB* txn_db, terark_forceinline size_t LockMap::GetStripe(const LockString& key, size_t hash) const { assert(num_stripes_ > 0); - auto col = hash % num_stripes_; + auto col = FastRange64(hash, num_stripes_); if (1 == super_stripes_) { return col; } else { @@ -317,7 +317,7 @@ size_t LockMap::GetStripe(const LockString& key, size_t hash) const { size_t plen = std::min(size_t(key_prefix_len_), key.size()); ROCKSDB_ASSUME(plen <= sizeof(pref)); memcpy(&pref, key.data(), plen); - size_t row = pref % super_stripes_; + size_t row = FastRange64(pref, super_stripes_); return row * num_stripes_ + col; } } From 5e5b05ac15c074549f0b3335917492536e873488 Mon Sep 17 00:00:00 2001 From: leipeng Date: Mon, 5 Oct 2026 10:26:44 +0800 Subject: [PATCH 28/58] Separate AVX pubseq publication from the odd/even protocol AVX publishes the record with one store and does not need the odd/even protocol used by scalar publication. Sharing that protocol obscures the different guarantees and adds unnecessary work to the AVX write path. An interrupted scalar publication can leave an odd generation. Successful recovery must discard that incomplete cursor and establish an even initial generation before writes resume; otherwise AVX increments would preserve the invalid state indefinitely. --- db/db_cspp_crash_safe_test.cc | 3 +++ db/db_impl/db_impl_open.cc | 18 ++++++++++++------ 2 files changed, 15 insertions(+), 6 deletions(-) diff --git a/db/db_cspp_crash_safe_test.cc b/db/db_cspp_crash_safe_test.cc index 4539e6a63..31bb05154 100644 --- a/db/db_cspp_crash_safe_test.cc +++ b/db/db_cspp_crash_safe_test.cc @@ -1273,6 +1273,9 @@ TEST_F(DBCsppCrashSafeTest, OddGenerationPublishesEvenAfterWalRecovery) { ASSERT_TRUE(SetPublishedSeqGeneration(dbname_, rec.generation | 1)); ASSERT_OK(TryReopen(options)); ASSERT_EQ(Get("before"), "recovery"); + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.generation % 2, 0U); + ASSERT_EQ(rec.pubseq, 0U); #if !defined(__AVX__) int odd = 0; SyncPoint::GetInstance()->SetCallBack( diff --git a/db/db_impl/db_impl_open.cc b/db/db_impl/db_impl_open.cc index 5665a55ae..92c91fae1 100644 --- a/db/db_impl/db_impl_open.cc +++ b/db/db_impl/db_impl_open.cc @@ -745,13 +745,9 @@ void DBImpl::PersistPublishedSequence(SequenceNumber seq, uint64_t wal_number, } TEST_SYNC_POINT("DBImpl::PersistPublishedSequence:SameSeq"); } - // Recovery may leave an odd generation until the first new publication. - const uint64_t g = pubseq_mmap_->generation | 1; #if defined(__AVX__) // One aligned 32-byte store. A process crash falls between instructions, - // so the record is all old or all new. Publish an even generation only - // after this store, including when recovery left it odd. A host without it - // uses the odd/even bracket below after the file is moved. + // so the record is all old or all new. // PublishedSeqRecord next; // next.pubseq = seq; // next.wal_number = wal_number; @@ -764,7 +760,11 @@ void DBImpl::PersistPublishedSequence(SequenceNumber seq, uint64_t wal_number, // volatile loads/stores. This keeps the AVX store a single 32-byte instruction. // https://llvm.org/docs/LangRef.html#volatile-memory-accesses *(volatile __m256i*)rec = packed; + pubseq_mmap_->generation += 2; #else + // Use the odd/even bracket also for files moved from an AVX host. + // A failed publication may leave an odd generation. + const uint64_t g = pubseq_mmap_->generation | 1; // Keep generation odd throughout the field stores, even if already odd. // Acquire keeps the field stores after this exchange; the release below // keeps them before publication of the next even generation. @@ -774,9 +774,9 @@ void DBImpl::PersistPublishedSequence(SequenceNumber seq, uint64_t wal_number, rec->wal_number = wal_number; rec->wal_offset = wal_offset; rec->pubseq = seq; -#endif terark::as_atomic(pubseq_mmap_->generation) .store(g + 1, std::memory_order_release); +#endif TEST_SYNC_POINT("DBImpl::PersistPublishedSequence:AfterCommit"); } @@ -1673,6 +1673,12 @@ Status DBImpl::LogAndApplyForRecovery(const RecoveryContext& recovery_ctx) { if (s.ok() && recovery_ctx.restore_published_seq_) { pubseq_mmap_->generation += 1; } + if (s.ok() && pubseq_mmap_ != nullptr && (pubseq_mmap_->generation & 1)) { + // Discard a torn cursor before enabling publication. + pubseq_mmap_->rec = PublishedSeqRecord{}; + terark::as_atomic(pubseq_mmap_->generation) + .store(0, std::memory_order_release); + } return s; } From 6437b71b706478b1fefd20e600f847ef0b143cce Mon Sep 17 00:00:00 2001 From: leipeng Date: Mon, 5 Oct 2026 09:03:27 +0800 Subject: [PATCH 29/58] Exercise MemTable precreation in unit tests Unit tests must exercise the same MemTable cache as production. Compiling that cache out hides allocation, configuration, and handoff behavior that the production write path depends on. --- db/column_family.cc | 32 +++++++++++++++++--------------- db/column_family.h | 2 -- db/db_range_del_test.cc | 4 ++++ db/db_test.cc | 4 ++++ db/db_test2.cc | 7 +++++++ 5 files changed, 32 insertions(+), 17 deletions(-) diff --git a/db/column_family.cc b/db/column_family.cc index 2ea7cb857..a0f3b18b7 100644 --- a/db/column_family.cc +++ b/db/column_family.cc @@ -40,6 +40,7 @@ #include "util/autovector.h" #include "util/cast_util.h" #include "util/compression.h" +#include namespace ROCKSDB_NAMESPACE { @@ -1164,7 +1165,12 @@ uint64_t ColumnFamilyData::GetLiveSstFilesSize() const { void ColumnFamilyData::PrepareNewMemtableInBackground( const MutableCFOptions& mutable_cf_options) { - #if !defined(ROCKSDB_UNIT_TEST) + bool use_cache = true; + TEST_SYNC_POINT_CALLBACK( + "ColumnFamilyData::PrepareNewMemtableInBackground:UseCache", &use_cache); + if (!use_cache) { + return; + } { std::lock_guard lk(precreated_memtable_mutex_); if (precreated_memtable_list_.full()) { @@ -1172,11 +1178,11 @@ void ColumnFamilyData::PrepareNewMemtableInBackground( return; } } - auto beg = ioptions_.clock->NowNanos(); + auto beg = terark::qtime::now(); auto tab = new MemTable(internal_comparator_, ioptions_, mutable_cf_options, write_buffer_manager_, 0/*earliest_seq*/, id_); - auto end = ioptions_.clock->NowNanos(); - RecordInHistogram(ioptions_.stats, MEMTAB_CONSTRUCT_NANOS, end - beg); + auto end = terark::qtime::now(); + RecordInHistogram(ioptions_.stats, MEMTAB_CONSTRUCT_NANOS, (end - beg).ns()); { std::lock_guard lk(precreated_memtable_mutex_); if (LIKELY(!precreated_memtable_list_.full())) { @@ -1191,34 +1197,30 @@ void ColumnFamilyData::PrepareNewMemtableInBackground( "precreated_memtable_list_ is full, discard the newly created memtab"); delete tab; } - #endif } MemTable* ColumnFamilyData::ConstructNewMemtable( const MutableCFOptions& mutable_cf_options, SequenceNumber earliest_seq) { MemTable* tab = nullptr; - #if !defined(ROCKSDB_UNIT_TEST) - { + bool use_cache = true; + TEST_SYNC_POINT_CALLBACK("ColumnFamilyData::ConstructNewMemtable:UseCache", + &use_cache); + if (use_cache) { std::lock_guard lk(precreated_memtable_mutex_); if (!precreated_memtable_list_.empty()) { tab = precreated_memtable_list_.front().release(); precreated_memtable_list_.pop_front(); } } - #endif if (tab) { tab->SetCreationSeq(earliest_seq); tab->SetEarliestSequenceNumber(earliest_seq); } else { - #if !defined(ROCKSDB_UNIT_TEST) - auto beg = ioptions_.clock->NowNanos(); - #endif + auto beg = terark::qtime::now(); tab = new MemTable(internal_comparator_, ioptions_, mutable_cf_options, write_buffer_manager_, earliest_seq, id_); - #if !defined(ROCKSDB_UNIT_TEST) - auto end = ioptions_.clock->NowNanos(); - RecordInHistogram(ioptions_.stats, MEMTAB_CONSTRUCT_NANOS, end - beg); - #endif + auto end = terark::qtime::now(); + RecordInHistogram(ioptions_.stats, MEMTAB_CONSTRUCT_NANOS, (end - beg).ns()); } return tab; } diff --git a/db/column_family.h b/db/column_family.h index 367b94160..0aadf8506 100644 --- a/db/column_family.h +++ b/db/column_family.h @@ -612,11 +612,9 @@ class ColumnFamilyData { WriteBufferManager* write_buffer_manager_; - #if !defined(ROCKSDB_UNIT_TEST) // precreated_memtable_list_.size() is normally 1 terark::fixed_circular_queue, 4> precreated_memtable_list_; std::mutex precreated_memtable_mutex_; - #endif MemTable* mem_; MemTableList imm_; diff --git a/db/db_range_del_test.cc b/db/db_range_del_test.cc index a84cad4c5..8000ecd66 100644 --- a/db/db_range_del_test.cc +++ b/db/db_range_del_test.cc @@ -3484,6 +3484,10 @@ TEST_F(DBRangeDelTest, NonBottommostCompactionDropRangetombstone) { } TEST_F(DBRangeDelTest, MemtableMaxRangeDeletions) { + SyncPoint::GetInstance()->SetCallBack( + "ColumnFamilyData::ConstructNewMemtable:UseCache", + [](void* arg) { *static_cast(arg) = false; }); + SyncPoint::GetInstance()->EnableProcessing(); // Tests option `memtable_max_range_deletions`. Options options = CurrentOptions(); options.level_compaction_dynamic_file_size = false; diff --git a/db/db_test.cc b/db/db_test.cc index ca7e2716a..f02bcab13 100644 --- a/db/db_test.cc +++ b/db/db_test.cc @@ -4497,6 +4497,10 @@ TEST_F(DBTest, ManualFlushWalAndWriteRace) { } TEST_F(DBTest, DynamicMemtableOptions) { + SyncPoint::GetInstance()->SetCallBack( + "ColumnFamilyData::ConstructNewMemtable:UseCache", + [](void* arg) { *static_cast(arg) = false; }); + SyncPoint::GetInstance()->EnableProcessing(); const uint64_t k64KB = 1 << 16; const uint64_t k128KB = 1 << 17; const uint64_t k5KB = 5 * 1024; diff --git a/db/db_test2.cc b/db/db_test2.cc index 4ca59a5ec..020fa4178 100644 --- a/db/db_test2.cc +++ b/db/db_test2.cc @@ -324,6 +324,9 @@ TEST_P(DBTestSharedWriteBufferAcrossCFs, SharedWriteBufferAcrossCFs) { static_cast*>(arg); *std::get<0>(*pair) = *std::get<1>(*pair); }); + SyncPoint::GetInstance()->SetCallBack( + "ColumnFamilyData::PrepareNewMemtableInBackground:UseCache", + [](void* arg) { *static_cast(arg) = false; }); ROCKSDB_NAMESPACE::SyncPoint::GetInstance()->EnableProcessing(); // The total soft write buffer size is about 105000 @@ -5770,6 +5773,10 @@ TEST_F(DBTest2, SeekFileRangeDeleteTail) { } TEST_F(DBTest2, BackgroundPurgeTest) { + SyncPoint::GetInstance()->SetCallBack( + "ColumnFamilyData::PrepareNewMemtableInBackground:UseCache", + [](void* arg) { *static_cast(arg) = false; }); + SyncPoint::GetInstance()->EnableProcessing(); Options options = CurrentOptions(); options.write_buffer_manager = std::make_shared(1 << 20); From a18eb1056a3e22518b4cbb5e8bce8e039310f25c Mon Sep 17 00:00:00 2001 From: leipeng Date: Mon, 5 Oct 2026 09:03:28 +0800 Subject: [PATCH 30/58] Inject WAL seek errors after the seek operation Failure injection should replace the result of an operation that can actually fail. A freshly constructed OK status bypasses the seek and does not exercise the recovery path after a real seek failure. --- db/db_cspp_crash_safe_test.cc | 2 +- db/db_impl/db_impl_open.cc | 4 +++- db/log_reader.cc | 6 +----- 3 files changed, 5 insertions(+), 7 deletions(-) diff --git a/db/db_cspp_crash_safe_test.cc b/db/db_cspp_crash_safe_test.cc index 31bb05154..d009f7cb1 100644 --- a/db/db_cspp_crash_safe_test.cc +++ b/db/db_cspp_crash_safe_test.cc @@ -1575,7 +1575,7 @@ TEST_F(DBCsppCrashSafeTest, SeekIOFailureReopensWal) { options.env = fault_env.get(); options.log_readahead_size = 0; SyncPoint::GetInstance()->SetCallBack( - "CrashSafeRecover::SeekToFileOffset:InjectStatus", + "CrashSafeRecover::SeekToFileOffset:Before", [&](void*) { fs->armed = true; }); SyncPoint::GetInstance()->EnableProcessing(); const Status s = TryReopen(options); diff --git a/db/db_impl/db_impl_open.cc b/db/db_impl/db_impl_open.cc index 92c91fae1..c828e4aed 100644 --- a/db/db_impl/db_impl_open.cc +++ b/db/db_impl/db_impl_open.cc @@ -1981,7 +1981,9 @@ Status DBImpl::RecoverLogFiles(const std::vector& wal_numbers, } else { const uint64_t record_start = PublishedWalRecordStart( crash_safe_wal_offset, crash_safe_wal_offset_kind); - const IOStatus seek_s = reader.SeekToFileOffset(record_start); + IOStatus seek_s = reader.SeekToFileOffset(record_start); + TEST_SYNC_POINT_CALLBACK( + "CrashSafeRecover::SeekToFileOffset:InjectStatus", &seek_s); if (!seek_s.ok()) { ROCKS_LOG_WARN(immutable_db_options_.info_log, "Crash-safe SeekToFileOffset(%" PRIu64 diff --git a/db/log_reader.cc b/db/log_reader.cc index cbbc39d64..6beaa929d 100644 --- a/db/log_reader.cc +++ b/db/log_reader.cc @@ -70,12 +70,8 @@ void Reader::InitSetMemTableAsLogIndex(FileSystem& fs) { } IOStatus Reader::SeekToFileOffset(uint64_t file_offset) { + TEST_SYNC_POINT("CrashSafeRecover::SeekToFileOffset:Before"); IOStatus io_s; - TEST_SYNC_POINT_CALLBACK("CrashSafeRecover::SeekToFileOffset:InjectStatus", - &io_s); - if (!io_s.ok()) { - return io_s; - } buffer_ = Slice(); eof_ = false; read_error_ = false; From f63e72c987b33903e17497df6bc5c3e48ee28d2a Mon Sep 17 00:00:00 2001 From: leipeng Date: Mon, 5 Oct 2026 09:03:29 +0800 Subject: [PATCH 31/58] Share immutable MemTable conversion capabilities Conversion mode is fixed when the factory is created. Keeping its meaning in the common Rep and factory classes gives callers one consistent capability contract instead of duplicated virtual checks and independently interpreted state in each implementation. --- include/rocksdb/memtablerep.h | 25 +++++++++++++++++++++++-- 1 file changed, 23 insertions(+), 2 deletions(-) diff --git a/include/rocksdb/memtablerep.h b/include/rocksdb/memtablerep.h index 9f2ae0de3..41297e6ea 100644 --- a/include/rocksdb/memtablerep.h +++ b/include/rocksdb/memtablerep.h @@ -42,6 +42,7 @@ #include #include "rocksdb/customizable.h" +#include "rocksdb/enum_reflection.h" #include "rocksdb/slice.h" namespace ROCKSDB_NAMESPACE { @@ -63,6 +64,8 @@ extern const char* EncodeKey(std::string* scratch, const Slice& target); class MemTableRep : public CacheAlignedNewDelete { public: + ROCKSDB_ENUM_CLASS_INCLASS(ConvertKind, uint8_t, + kDontConvert, kDumpMem, kFileMmap); // KeyComparator provides a means to compare keys, which are internal keys // concatenated with values. class KeyComparator { @@ -308,7 +311,13 @@ class MemTableRep : public CacheAlignedNewDelete { virtual void FinishHint(void*); virtual void InitSetMemTableAsLogIndex(bool) {} virtual bool SupportMemTableAsLogIndex() const { return false; } - virtual bool SupportConvertToSST() const { return false; } + ConvertKind GetConvertKind() const { return m_convert_to_sst; } + bool SupportConvertToSST() const { + return m_convert_to_sst != ConvertKind::kDontConvert; + } + bool SupportCrashSafe() const { + return m_convert_to_sst == ConvertKind::kFileMmap; + } virtual Status ConvertToSST(struct FileMetaData*, const struct TableBuilderOptions&); protected: @@ -335,12 +344,15 @@ class MemTableRep : public CacheAlignedNewDelete { virtual Slice UserKey(const char* key) const; Allocator* allocator_; + ConvertKind m_convert_to_sst = ConvertKind::kDontConvert; }; // This is the base class for all factories that are used by RocksDB to create // new MemTableRep objects class MemTableRepFactory : public Customizable { public: + using ConvertKind = MemTableRep::ConvertKind; + ~MemTableRepFactory() override {} static const char* Type() { return "MemTableRepFactory"; } @@ -382,10 +394,16 @@ class MemTableRepFactory : public Customizable { // Default: false virtual bool CanHandleDuplicatedKey() const { return false; } + bool SupportConvertToSST() const { + return convert_to_sst != ConvertKind::kDontConvert; + } + // Return true if leftover mmap memtables created by this factory can be // RO-loaded after a crash and ConvertToSST during Recover. // Default: false - virtual bool SupportCrashSafe() const { return false; } + bool SupportCrashSafe() const { + return convert_to_sst == ConvertKind::kFileMmap; + } // Append leftover crash-safe mmap paths under cf_dir (plus factory chroot). // Default: no leftovers. @@ -409,6 +427,9 @@ class MemTableRepFactory : public Customizable { const struct TableBuilderOptions&) { return Status::NotSupported("RecoverCrashSafeMemTableToSST"); } + + protected: + ConvertKind convert_to_sst = ConvertKind::kDontConvert; }; // This uses a skip list to store keys. It is the default. From e39dcf4708e5a86e4ff93ff28889d4ba1ea5d9eb Mon Sep 17 00:00:00 2001 From: leipeng Date: Mon, 5 Oct 2026 09:03:31 +0800 Subject: [PATCH 32/58] Sanitize MemTable merge thresholds for ConvertToSST ConvertToSST converts one existing MemTable rather than rebuilding a merged table. Waiting for several MemTables is incompatible with that contract. Normalize the merge threshold for converting factories while retaining the requested threshold for ordinary MemTables. --- db/column_family.cc | 9 ++++++++ db/db_memtable_convert_test.cc | 42 ++++++++++++++++++++++++++++++++++ db/flush_job.cc | 3 ++- 3 files changed, 53 insertions(+), 1 deletion(-) diff --git a/db/column_family.cc b/db/column_family.cc index a0f3b18b7..ec759c960 100644 --- a/db/column_family.cc +++ b/db/column_family.cc @@ -469,6 +469,15 @@ ColumnFamilyOptions SanitizeOptions(const ImmutableDBOptions& db_options, } #endif + if (result.min_write_buffer_number_to_merge > 1 && + result.memtable_factory->SupportConvertToSST()) { + ROCKS_LOG_WARN(db_options.logger, + "ConvertToSST converts each memtable separately; " + "min_write_buffer_number_to_merge > 1 is incompatible " + "and is sanitized to 1"); + result.min_write_buffer_number_to_merge = 1; + } + return result; } diff --git a/db/db_memtable_convert_test.cc b/db/db_memtable_convert_test.cc index d3d20c436..85b60976d 100644 --- a/db/db_memtable_convert_test.cc +++ b/db/db_memtable_convert_test.cc @@ -112,6 +112,48 @@ TEST_P(DBMemtableConvertTest, ManualFlushConverts) { Close(); } +TEST_P(DBMemtableConvertTest, ConversionSanitizesMergeThreshold) { + // Avoid the existing atomic-flush sanitizer masking the conversion rule. + if (std::get<2>(GetParam())) return; + Options options = ConvertOptions(); + ASSERT_GT(options.min_write_buffer_number_to_merge, 1); + DestroyAndReopen(options); + ASSERT_EQ(db_->GetOptions().min_write_buffer_number_to_merge, 1); + ColumnFamilyHandle* cf = nullptr; + ASSERT_OK(db_->CreateColumnFamily(options, "converted", &cf)); + ASSERT_EQ(db_->GetOptions(cf).min_write_buffer_number_to_merge, 1); + ASSERT_OK(db_->DestroyColumnFamilyHandle(cf)); + Close(); +} + +TEST_P(DBMemtableConvertTest, NonConversionKeepsMergeThreshold) { + // One run per plugin suffices for the disabled conversion and default + // factory controls; the converting modes are covered independently above. + if (std::get<2>(GetParam()) || + std::string(std::get<1>(GetParam())) != "kDumpMem") return; + for (bool skip_list : {false, true}) { + SCOPED_TRACE(skip_list); + Options options = CurrentOptions(); + options.max_write_buffer_number = 8; + options.min_write_buffer_number_to_merge = 2; + options.atomic_flush = false; + if (skip_list) { + options.memtable_factory = std::make_shared(); + } else { + options.memtable_factory = PluginFactorySP::AcquirePlugin( + std::get<0>(GetParam()) ? "OffsetSkipList" : "CSPPMemTab", + {{"mem_cap", 16777216}, {"convert_to_sst", "kDontConvert"}}, repo_); + } + DestroyAndReopen(options); + ASSERT_EQ(db_->GetOptions().min_write_buffer_number_to_merge, 2); + ColumnFamilyHandle* cf = nullptr; + ASSERT_OK(db_->CreateColumnFamily(options, "plain", &cf)); + ASSERT_EQ(db_->GetOptions(cf).min_write_buffer_number_to_merge, 2); + ASSERT_OK(db_->DestroyColumnFamilyHandle(cf)); + Close(); + } +} + TEST_P(DBMemtableConvertTest, CloseConvertsAllMemtables) { CheckClose(false); } diff --git a/db/flush_job.cc b/db/flush_job.cc index f1e67ca3d..4e84fe839 100644 --- a/db/flush_job.cc +++ b/db/flush_job.cc @@ -963,7 +963,8 @@ Status FlushJob::WriteLevel0Table() { TableFileCreationReason::kFlush, oldest_key_time, current_time, db_id_, db_session_id_, 0 /* target_file_size */, meta_.fd.GetNumber()); - if (mems_.size() == 1 && mems_.front()->SupportConvertToSST()) { + if (mems_.front()->SupportConvertToSST()) { + ROCKSDB_ASSERT_EQ(mems_.size(), 1); // convert MemTable to sst MemTable* memtable = mems_.front(); // pass these fields to ConvertToSST, to fill TableProperties From b74db85a6543a38feb09bbbd9c8f176c785f519c Mon Sep 17 00:00:00 2001 From: leipeng Date: Mon, 5 Oct 2026 09:03:32 +0800 Subject: [PATCH 33/58] Skip MemPurge for FileMmap MemTables FileMmap MemTables already have a backing file for ConvertToSST. Rebuilding their contents through MemPurge is outside that lifecycle and must not replace the existing file-backed table. --- db/flush_job.cc | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/db/flush_job.cc b/db/flush_job.cc index 4e84fe839..0dce27e1f 100644 --- a/db/flush_job.cc +++ b/db/flush_job.cc @@ -229,7 +229,8 @@ Status FlushJob::Run(LogsWithPrepTracker* prep_tracker, FileMetaData* file_meta, double mempurge_threshold = mutable_cf_options_.experimental_mempurge_threshold; - if (db_options_.memtable_as_log_index) { + if (db_options_.memtable_as_log_index || + cfd_->ioptions()->memtable_factory->SupportCrashSafe()) { mempurge_threshold = 0; // not supported } From 7e8b8b032b68c0b8f8a0244f03bebf7979ff1c46 Mon Sep 17 00:00:00 2001 From: leipeng Date: Mon, 5 Oct 2026 09:03:34 +0800 Subject: [PATCH 34/58] Validate FileMmap recovery configuration Crash-safe recovery requires every column family to provide recoverable FileMmap data. Accepting an incompatible factory would silently defeat that guarantee. Read-only and secondary opens must not create writable FileMmap backing files; disabling crash-safe recovery still leaves ordinary read-write factory choices unrestricted. --- db/column_family.cc | 6 +++++ db/db_cspp_crash_safe_test.cc | 44 +++++++++++++++++++++++++++++---- db/db_impl/db_impl_open.cc | 8 ++++++ db/db_impl/db_impl_secondary.cc | 6 +++++ db/db_options_test.cc | 16 +++++++++++- 5 files changed, 74 insertions(+), 6 deletions(-) diff --git a/db/column_family.cc b/db/column_family.cc index ec759c960..cae542dd0 100644 --- a/db/column_family.cc +++ b/db/column_family.cc @@ -1484,6 +1484,12 @@ void ColumnFamilyData::ResetThreadLocalSuperVersions() { Status ColumnFamilyData::ValidateOptions( const DBOptions& db_options, const ColumnFamilyOptions& cf_options) { + if (db_options.memtable_crash_safe_recover && + !cf_options.memtable_factory->SupportCrashSafe()) { + return Status::InvalidArgument( + "memtable_crash_safe_recover requires a FileMmap memtable factory", + cf_options.memtable_factory->Name()); + } Status s; s = CheckCompressionSupported(cf_options); if (s.ok() && db_options.allow_concurrent_memtable_write) { diff --git a/db/db_cspp_crash_safe_test.cc b/db/db_cspp_crash_safe_test.cc index d009f7cb1..c2bb3ca0a 100644 --- a/db/db_cspp_crash_safe_test.cc +++ b/db/db_cspp_crash_safe_test.cc @@ -230,6 +230,35 @@ class DBCsppCrashSafeTest : public DBTestBase { : DBTestBase("db_cspp_crash_safe_test", /*env_do_fsync=*/false) {} }; +TEST_F(DBCsppCrashSafeTest, FileMmapRejectsReadOnlyAndSecondary) { + Close(); + for (bool osl : {false, true}) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "WalFilterFallsBackToWal", + osl ? "10" : "00"), 0); + std::vector before; + ASSERT_OK(env_->GetChildren(dbname_, &before)); + std::sort(before.begin(), before.end()); + for (bool secondary : {false, true}) { + SCOPED_TRACE(osl); + SCOPED_TRACE(secondary); + DB* rejected = nullptr; + const Status s = secondary + ? DB::OpenAsSecondary(options, dbname_, dbname_ + "_secondary", + &rejected) + : DB::OpenForReadOnly(options, dbname_, &rejected); + ASSERT_TRUE(s.IsInvalidArgument()) << s.ToString(); + ASSERT_EQ(rejected, nullptr); + std::vector after; + ASSERT_OK(env_->GetChildren(dbname_, &after)); + std::sort(after.begin(), after.end()); + ASSERT_EQ(after, before); + } + } +} + TEST_F(DBCsppCrashSafeTest, CrashSafeRequiresFileMmap) { for (const char* cls : {"CSPPMemTab", "OffsetSkipList"}) { for (const char* mode : {"kDontConvert", "kDumpMem", "kFileMmap"}) { @@ -790,12 +819,17 @@ TEST_F(DBCsppCrashSafeTest, RecoverOnCreatesPublishedSeqAndFlushWalBuffer) { ASSERT_OK(env_->FileExists(CrashSafePubSeqFileName(dbname_))); } -TEST_F(DBCsppCrashSafeTest, SkipListPlusRecoverOpensAndUsesFullWal) { +TEST_F(DBCsppCrashSafeTest, SkipListRejectsRecoverAndUsesWalWhenDisabled) { Close(); Options options = CurrentOptions(); + options.memtable_factory = std::make_shared(); options.memtable_crash_safe_recover = true; options.create_if_missing = true; Destroy(options); + ASSERT_TRUE(TryReopen(options).IsInvalidArgument()); + ASSERT_EQ(db_, nullptr); + options.memtable_crash_safe_recover = false; + options.avoid_flush_during_shutdown = true; ASSERT_OK(TryReopen(options)); ASSERT_OK(Put("a", "1")); Close(); @@ -962,7 +996,7 @@ TEST_F(DBCsppCrashSafeTest, OslLeftoverWithWriterLock) { } TEST_F(CrashChild, DISABLED_MixedDumpMemUsesFullWal) { - Options options = BaseCrashSafeOptions(dbname_, true, false); + Options options = BaseCrashSafeOptions(dbname_, false, false); const bool osl = arg_ == "OSL"; if (osl) SetupOsl(&options, true); DB* child_db = nullptr; @@ -982,7 +1016,7 @@ TEST_F(DBCsppCrashSafeTest, MixedDumpMemUsesFullWal) { for (bool osl : {false, true}) { SCOPED_TRACE(osl); Close(); - Options options = BaseCrashSafeOptions(dbname_, true, false); + Options options = BaseCrashSafeOptions(dbname_, false, false); if (osl) SetupOsl(&options, true); Destroy(options); ASSERT_EQ(RunCrashChild(dbname_, "MixedDumpMemUsesFullWal", @@ -2519,9 +2553,9 @@ TEST_F(DBCsppCrashSafeTest, KindPrepConvertFailFallsBackToWal) { } #endif -TEST_F(DBCsppCrashSafeTest, DontConvertFallsBackToWal) { +TEST_F(DBCsppCrashSafeTest, DontConvertUsesWal) { Close(); - Options options = BaseCrashSafeOptions(dbname_, true, false); + Options options = BaseCrashSafeOptions(dbname_, false, false); SetupCspp(&options, false); Destroy(options); ASSERT_OK(TryReopen(options)); diff --git a/db/db_impl/db_impl_open.cc b/db/db_impl/db_impl_open.cc index c828e4aed..8beab8bbd 100644 --- a/db/db_impl/db_impl_open.cc +++ b/db/db_impl/db_impl_open.cc @@ -1022,6 +1022,14 @@ Status DBImpl::Recover( bool error_if_wal_file_exists, bool error_if_data_exists_in_wals, uint64_t* recovered_seq, RecoveryContext* recovery_ctx) { mutex_.AssertHeld(); + if (read_only) { + for (const auto& cf : column_families) { + if (cf.options.memtable_factory->SupportCrashSafe()) { + return Status::InvalidArgument( + "FileMmap memtable is not supported in read-only mode", cf.name); + } + } + } std::vector leftover_snapshot; if (immutable_db_options_.memtable_crash_safe_recover && !read_only) { diff --git a/db/db_impl/db_impl_secondary.cc b/db/db_impl/db_impl_secondary.cc index 8f4d2e641..0d2af8b4a 100644 --- a/db/db_impl/db_impl_secondary.cc +++ b/db/db_impl/db_impl_secondary.cc @@ -36,6 +36,12 @@ Status DBImplSecondary::Recover( bool /*error_if_data_exists_in_wals*/, uint64_t*, RecoveryContext* /*recovery_ctx*/) { mutex_.AssertHeld(); + for (const auto& cf : column_families) { + if (cf.options.memtable_factory->SupportCrashSafe()) { + return Status::InvalidArgument( + "FileMmap memtable is not supported in secondary mode", cf.name); + } + } JobContext job_context(0); Status s; diff --git a/db/db_options_test.cc b/db/db_options_test.cc index 99fcb5941..b217e26ce 100644 --- a/db/db_options_test.cc +++ b/db/db_options_test.cc @@ -28,6 +28,10 @@ namespace ROCKSDB_NAMESPACE { +#ifdef HAS_TOPLING_CSPP_MEMTABLE +std::shared_ptr EasyNewMemTableRep(Slice cls, Slice js); +#endif + class DBOptionsTest : public DBTestBase { public: DBOptionsTest() : DBTestBase("db_options_test", /*env_do_fsync=*/true) {} @@ -159,11 +163,12 @@ TEST_F(DBOptionsTest, ReadOnlyDisablesCrashSafeAndWarns) { TEST_F(DBOptionsTest, ReadOnlyReplaysWalWithCrashSafeDisabled) { Options options = CurrentOptions(); - options.memtable_crash_safe_recover = true; + options.memtable_crash_safe_recover = false; options.avoid_flush_during_shutdown = true; DestroyAndReopen(options); ASSERT_OK(Put("k", "v")); Close(); + options.memtable_crash_safe_recover = true; DB* ro = nullptr; ASSERT_OK(DB::OpenForReadOnly(options, dbname_, &ro)); std::unique_ptr read_only_db(ro); @@ -177,6 +182,15 @@ TEST_F(DBOptionsTest, CrashSafeForcesWal) { for (bool crash_safe : {false, true}) { SCOPED_TRACE(crash_safe); Options options = CurrentOptions(); + if (crash_safe) { +#ifdef HAS_TOPLING_CSPP_MEMTABLE + options.memtable_factory = EasyNewMemTableRep( + "CSPPMemTab", R"({"mem_cap":16777216,"convert_to_sst":"kFileMmap"})"); + options.avoid_flush_during_shutdown = true; +#else + continue; +#endif + } options.memtable_as_log_index = false; options.memtable_crash_safe_recover = crash_safe; options.statistics = CreateDBStatistics(); From 91c49a8592a3e8c76d0847eeb8be38c3d4e682e0 Mon Sep 17 00:00:00 2001 From: leipeng Date: Mon, 5 Oct 2026 09:03:35 +0800 Subject: [PATCH 35/58] Encode recoverable MemTable files in VersionEdit Recovery needs a durable inventory to distinguish a missing MemTable file from one that never existed. These records describe that inventory without treating mutable files as installed SSTs. Older readers must reject the records rather than silently ignore required recovery state. --- db/version_edit.cc | 54 +++++++++++++++++++++++++++++++++++++++++ db/version_edit.h | 20 +++++++++++++++ db/version_edit_test.cc | 50 ++++++++++++++++++++++++++++++++++++++ 3 files changed, 124 insertions(+) diff --git a/db/version_edit.cc b/db/version_edit.cc index b0af0d87a..29d9c8090 100644 --- a/db/version_edit.cc +++ b/db/version_edit.cc @@ -86,6 +86,9 @@ void VersionEdit::Clear() { wal_additions_.clear(); wal_deletion_.Reset(); column_family_ = 0; + memtable_file_additions_.clear(); + memtable_file_deletions_.clear(); + has_memtable_file_tracking_ = false; is_column_family_add_ = false; is_column_family_drop_ = false; column_family_name_.clear(); @@ -288,6 +291,17 @@ bool VersionEdit::EncodeTo(std::string* dst, } // 0 is default and does not need to be explicitly written + if (has_memtable_file_tracking_) { + PutVarint32(dst, kMemTableFileTracking); + } + for (uint64_t number : memtable_file_additions_) { + if (number == 0 || number > kFileNumberMask) return false; + PutVarint32Varint64(dst, kMemTableFileAddition, number); + } + for (uint64_t number : memtable_file_deletions_) { + if (number == 0 || number > kFileNumberMask) return false; + PutVarint32Varint64(dst, kMemTableFileDeletion, number); + } if (column_family_ != 0) { PutVarint32Varint32(dst, kColumnFamily, column_family_); } @@ -768,6 +782,24 @@ Status VersionEdit::DecodeFrom(const Slice& src) { is_column_family_drop_ = true; break; + case kMemTableFileTracking: + has_memtable_file_tracking_ = true; + break; + + case kMemTableFileAddition: + case kMemTableFileDeletion: { + uint64_t number; + if (!GetVarint64(&input, &number) || number == 0 || + number > kFileNumberMask) { + msg = "invalid MemTable file number"; + } else if (tag == kMemTableFileAddition) { + memtable_file_additions_.insert(number); + } else { + memtable_file_deletions_.insert(number); + } + break; + } + case kInAtomicGroup: is_in_atomic_group_ = true; if (!GetVarint32(&input, &remaining_entries_)) { @@ -948,6 +980,15 @@ std::string VersionEdit::DebugString(bool hex_key) const { r.append("\n ColumnFamily: "); AppendNumberTo(&r, column_family_); + if (has_memtable_file_tracking_) r.append("\n MemTableFileTracking: true"); + for (uint64_t number : memtable_file_additions_) { + r.append("\n AddMemTableFile: "); + AppendNumberTo(&r, number); + } + for (uint64_t number : memtable_file_deletions_) { + r.append("\n DeleteMemTableFile: "); + AppendNumberTo(&r, number); + } if (is_column_family_add_) { r.append("\n ColumnFamilyAdd: "); r.append(column_family_name_); @@ -1098,6 +1139,19 @@ std::string VersionEdit::DebugJSON(int edit_num, bool hex_key) const { } jw << "ColumnFamily" << column_family_; + if (has_memtable_file_tracking_) jw << "MemTableFileTracking" << true; + if (!memtable_file_additions_.empty()) { + jw << "MemTableFileAdditions"; + jw.StartArray(); + for (uint64_t number : memtable_file_additions_) jw << number; + jw.EndArray(); + } + if (!memtable_file_deletions_.empty()) { + jw << "MemTableFileDeletions"; + jw.StartArray(); + for (uint64_t number : memtable_file_deletions_) jw << number; + jw.EndArray(); + } if (is_column_family_add_) { jw << "ColumnFamilyAdd" << column_family_name_; diff --git a/db/version_edit.h b/db/version_edit.h index ce35758d4..aa6954cfe 100644 --- a/db/version_edit.h +++ b/db/version_edit.h @@ -63,6 +63,11 @@ enum Tag : uint32_t { kBlobFileAddition = 400, kBlobFileGarbage, + // Required recovery metadata: older readers must reject these tags. + kMemTableFileAddition = 500, + kMemTableFileDeletion = 501, + kMemTableFileTracking = 502, + // Mask for an unidentified tag from the future which can be safely ignored. kTagSafeIgnoreMask = 1 << 13, @@ -652,6 +657,7 @@ class VersionEdit { size_t NumEntries() const { return new_files_.size() + deleted_files_.size() + blob_file_additions_.size() + blob_file_garbages_.size() + + memtable_file_additions_.size() + memtable_file_deletions_.size() + wal_additions_.size() + !wal_deletion_.IsEmpty(); } @@ -660,6 +666,17 @@ class VersionEdit { } uint32_t GetColumnFamily() const { return column_family_; } + void AddMemTableFile(uint64_t number) { memtable_file_additions_.insert(number); } + void DeleteMemTableFile(uint64_t number) { memtable_file_deletions_.insert(number); } + const std::set& GetMemTableFileAdditions() const { + return memtable_file_additions_; + } + const std::set& GetMemTableFileDeletions() const { + return memtable_file_deletions_; + } + void SetMemTableFileTracking() { has_memtable_file_tracking_ = true; } + bool HasMemTableFileTracking() const { return has_memtable_file_tracking_; } + const std::string& GetColumnFamilyName() const { return column_family_name_; } // set column family ID by calling SetColumnFamily() @@ -774,6 +791,9 @@ class VersionEdit { // Each version edit record should have column_family_ set // If it's not set, it is default (0) uint32_t column_family_ = 0; + std::set memtable_file_additions_; + std::set memtable_file_deletions_; + bool has_memtable_file_tracking_ = false; // a version edit can be either column_family add or // column_family drop. If it's column family add, // it also includes column family name. diff --git a/db/version_edit_test.cc b/db/version_edit_test.cc index 252352069..8767ef49e 100644 --- a/db/version_edit_test.cc +++ b/db/version_edit_test.cc @@ -40,6 +40,56 @@ static void TestEncodeDecode(const VersionEdit& edit) { class VersionEditTest : public testing::Test {}; +TEST_F(VersionEditTest, MemTableFilesEncodeDecodeAndClear) { + VersionEdit edit; + edit.SetColumnFamily(7); + edit.SetMemTableFileTracking(); + edit.AddMemTableFile(123); + edit.AddMemTableFile(kFileNumberMask); + edit.DeleteMemTableFile(42); + edit.MarkAtomicGroup(0); + std::string encoded; + ASSERT_TRUE(edit.EncodeTo(&encoded)); + VersionEdit parsed; + ASSERT_OK(parsed.DecodeFrom(encoded)); + ASSERT_EQ(parsed.GetColumnFamily(), 7U); + ASSERT_TRUE(parsed.HasMemTableFileTracking()); + ASSERT_EQ(parsed.GetMemTableFileAdditions(), edit.GetMemTableFileAdditions()); + ASSERT_EQ(parsed.GetMemTableFileDeletions(), edit.GetMemTableFileDeletions()); + ASSERT_TRUE(parsed.IsInAtomicGroup()); + ASSERT_EQ(parsed.GetRemainingEntries(), 0U); + parsed.Clear(); + ASSERT_FALSE(parsed.HasMemTableFileTracking()); + ASSERT_TRUE(parsed.GetMemTableFileAdditions().empty()); + ASSERT_TRUE(parsed.GetMemTableFileDeletions().empty()); + ASSERT_FALSE(parsed.IsInAtomicGroup()); +} + +TEST_F(VersionEditTest, MemTableFilesRejectInvalidNumbers) { + for (uint32_t tag : {uint32_t(kMemTableFileAddition), + uint32_t(kMemTableFileDeletion)}) { + for (uint64_t number : {uint64_t(0), kFileNumberMask + 1}) { + std::string encoded; + PutVarint32Varint64(&encoded, tag, number); + VersionEdit parsed; + ASSERT_TRUE(parsed.DecodeFrom(encoded).IsCorruption()); + VersionEdit invalid; + if (tag == kMemTableFileAddition) invalid.AddMemTableFile(number); + else invalid.DeleteMemTableFile(number); + encoded.clear(); + ASSERT_FALSE(invalid.EncodeTo(&encoded)); + } + std::string truncated; + PutVarint32(&truncated, tag); + truncated.push_back(char(0x80)); + VersionEdit parsed; + ASSERT_TRUE(parsed.DecodeFrom(truncated).IsCorruption()); + } + ASSERT_EQ(kMemTableFileAddition & kTagSafeIgnoreMask, 0U); + ASSERT_EQ(kMemTableFileDeletion & kTagSafeIgnoreMask, 0U); + ASSERT_EQ(kMemTableFileTracking & kTagSafeIgnoreMask, 0U); +} + TEST_F(VersionEditTest, EncodeDecode) { static const uint64_t kBig = 1ull << 50; static const uint32_t kBig32Bit = 1ull << 30; From af59c716cf11689d6e7ebb169c4e595d3b192269 Mon Sep 17 00:00:00 2001 From: leipeng Date: Mon, 5 Oct 2026 09:03:36 +0800 Subject: [PATCH 36/58] Maintain the committed MemTable file inventory per column family The MANIFEST inventory must survive replay, rollover, and an empty live set. Failed edits must not change it, and replaying historical deletions must not schedule live-file cleanup. Keeping the inventory with its column family also makes file retirement and same-number SST ownership transfer follow the committed state. --- db/column_family.cc | 14 +++ db/column_family.h | 4 + db/version_edit_handler.cc | 2 + db/version_set.cc | 52 ++++++++- db/version_set.h | 5 + db/version_set_test.cc | 233 +++++++++++++++++++++++++++++++++++++ 6 files changed, 309 insertions(+), 1 deletion(-) diff --git a/db/column_family.cc b/db/column_family.cc index cae542dd0..48fe989cd 100644 --- a/db/column_family.cc +++ b/db/column_family.cc @@ -1234,6 +1234,20 @@ MemTable* ColumnFamilyData::ConstructNewMemtable( return tab; } +void ColumnFamilyData::ApplyMemTableFileEdit(const VersionEdit& edit) { + ROCKSDB_ASSERT_EQ(edit.GetColumnFamily(), id_); + if (edit.IsColumnFamilyDrop()) { + memtable_files_.clear(); + return; + } + for (uint64_t number : edit.GetMemTableFileDeletions()) { + memtable_files_.erase(number); + } + for (uint64_t number : edit.GetMemTableFileAdditions()) { + memtable_files_.insert(number); + } +} + void ColumnFamilyData::CreateNewMemtable( const MutableCFOptions& mutable_cf_options, SequenceNumber earliest_seq) { if (mem_ != nullptr) { diff --git a/db/column_family.h b/db/column_family.h index 0aadf8506..3e9aad600 100644 --- a/db/column_family.h +++ b/db/column_family.h @@ -373,6 +373,9 @@ class ColumnFamilyData { uint64_t OldestLogToKeep(); void PrepareNewMemtableInBackground(const MutableCFOptions&); + // DB mutex must be held for registration and live-file collection. + const std::set& GetMemTableFiles() const { return memtable_files_; } + void ApplyMemTableFileEdit(const VersionEdit& edit); // See Memtable constructor for explanation of earliest_seq param. MemTable* ConstructNewMemtable(const MutableCFOptions& mutable_cf_options, @@ -618,6 +621,7 @@ class ColumnFamilyData { MemTable* mem_; MemTableList imm_; + std::set memtable_files_; // Protected by DB mutex. SuperVersion* super_version_; // An ordinal representing the current SuperVersion. Updated by diff --git a/db/version_edit_handler.cc b/db/version_edit_handler.cc index 395d066d7..06c92dcb2 100644 --- a/db/version_edit_handler.cc +++ b/db/version_edit_handler.cc @@ -213,6 +213,7 @@ Status VersionEditHandler::ApplyVersionEdit(VersionEdit& edit, if (s.ok()) { assert(cfd != nullptr); s = ExtractInfoFromVersionEdit(*cfd, edit); + if (s.ok()) version_set_->ApplyMemTableFileEdit(edit); } return s; } @@ -528,6 +529,7 @@ ColumnFamilyData* VersionEditHandler::DestroyCfAndCleanup( ColumnFamilyData* ret = version_set_->GetColumnFamilySet()->GetColumnFamily(cf_id); assert(ret != nullptr); + ret->ApplyMemTableFileEdit(edit); ret->SetDropped(); ret->UnrefAndTryDelete(); ret = nullptr; diff --git a/db/version_set.cc b/db/version_set.cc index 573c57e97..b462fa626 100644 --- a/db/version_set.cc +++ b/db/version_set.cc @@ -5505,6 +5505,7 @@ void VersionSet::Reset() { obsolete_files_.clear(); obsolete_manifests_.clear(); wals_.Reset(); + has_memtable_file_tracking_ = false; } void VersionSet::AppendVersion(ColumnFamilyData* column_family_data, @@ -5743,6 +5744,7 @@ Status VersionSet::ProcessManifestWrites( std::unordered_map curr_state; VersionEdit wal_additions; if (new_descriptor_log) { + if (has_memtable_file_tracking_) wal_additions.SetMemTableFileTracking(); pending_manifest_file_number_ = NewFileNumber(); batch_edits.back()->SetNextFile(next_file_number_.load()); @@ -5956,6 +5958,40 @@ Status VersionSet::ProcessManifestWrites( // Install the new versions if (s.ok()) { + // Only a committed live batch can retire backing files. MANIFEST replay + // applies registry edits without scheduling historical deletions. + std::map retired_memtable_files; + for (const auto* edit : batch_edits) { + auto* cfd = column_family_set_->GetColumnFamily(edit->GetColumnFamily()); + if (cfd != nullptr) { + const auto& files = cfd->GetMemTableFiles(); + const auto& path = cfd->ioptions()->cf_paths[0].path; + if (edit->IsColumnFamilyDrop()) { + for (uint64_t number : files) { + retired_memtable_files.emplace(number, path); + } + } else { + for (uint64_t number : edit->GetMemTableFileDeletions()) { + assert(files.count(number)); + retired_memtable_files.emplace(number, path); + } + } + } + ApplyMemTableFileEdit(*edit); + } + if (!retired_memtable_files.empty()) { + // In-place conversion transfers ownership to the installed SST version. + for (const auto* edit : batch_edits) { + for (const auto& file : edit->GetNewFiles()) { + retired_memtable_files.erase(file.second.fd.GetNumber()); + } + } + for (const auto& file : retired_memtable_files) { + auto* metadata = new FileMetaData(); + metadata->fd = FileDescriptor(file.first, 0, 0); + obsolete_files_.emplace_back(metadata, file.second); + } + } if (first_writer.edit_list.front()->IsColumnFamilyAdd()) { assert(batch_edits.size() == 1); assert(new_cf_options != nullptr); @@ -6850,7 +6886,8 @@ Status VersionSet::WriteCurrentStateToManifest( } // Save WALs. - if (!wal_additions.GetWalAdditions().empty()) { + if (!wal_additions.GetWalAdditions().empty() || + wal_additions.HasMemTableFileTracking()) { TEST_SYNC_POINT_CALLBACK("VersionSet::WriteCurrentStateToManifest:SaveWal", const_cast(&wal_additions)); std::string record; @@ -6916,6 +6953,11 @@ Status VersionSet::WriteCurrentStateToManifest( VersionEdit edit; edit.SetColumnFamily(cfd->GetID()); + // Only the serialized MANIFEST writer changes this registry. + for (uint64_t number : cfd->GetMemTableFiles()) { + edit.AddMemTableFile(number); + } + const auto* current = cfd->current(); assert(current); @@ -7568,6 +7610,14 @@ uint64_t VersionSet::GetObsoleteSstFilesSize() const { return ret; } +void VersionSet::ApplyMemTableFileEdit(const VersionEdit& edit) { + has_memtable_file_tracking_ |= edit.HasMemTableFileTracking(); + // Recovery may skip CFs that were not requested. + if (auto* cfd = column_family_set_->GetColumnFamily(edit.GetColumnFamily())) { + cfd->ApplyMemTableFileEdit(edit); + } +} + ColumnFamilyData* VersionSet::CreateColumnFamily( const ColumnFamilyOptions& cf_options, const ReadOptions& read_options, const VersionEdit* edit) { diff --git a/db/version_set.h b/db/version_set.h index 2c4b3bf7e..5aa84e0bb 100644 --- a/db/version_set.h +++ b/db/version_set.h @@ -1370,6 +1370,10 @@ class VersionSet { // Allocate and return a new file number uint64_t NewFileNumber() { return next_file_number_.fetch_add(1); } + // Access requires the DB mutex, like other MANIFEST state. + bool HasMemTableFileTracking() const { return has_memtable_file_tracking_; } + void ApplyMemTableFileEdit(const VersionEdit& edit); + // Fetch And Add n new file number uint64_t FetchAddFileNumber(uint64_t n) { return next_file_number_.fetch_add(n); @@ -1678,6 +1682,7 @@ class VersionSet { // Protected by DB mutex. WalSet wals_; + bool has_memtable_file_tracking_ = false; std::unique_ptr column_family_set_; Cache* table_cache_; diff --git a/db/version_set_test.cc b/db/version_set_test.cc index 4703b1f44..a2a4b269c 100644 --- a/db/version_set_test.cc +++ b/db/version_set_test.cc @@ -1505,6 +1505,156 @@ TEST_F(VersionSetTest, SameColumnFamilyGroupCommit) { EXPECT_EQ(kGroupSize - 1, count); } +TEST_F(VersionSetTest, MemTableRegistryManifestRollover) { + NewDB(); + const uint64_t first = versions_->NewFileNumber(); + const uint64_t second = versions_->NewFileNumber(); + ASSERT_FALSE(versions_->HasMemTableFileTracking()); + VersionEdit addition; + addition.SetMemTableFileTracking(); + addition.AddMemTableFile(first); + addition.AddMemTableFile(second); + ASSERT_OK(LogAndApplyToDefaultCF(addition)); + ASSERT_EQ(versions_->GetColumnFamilySet()->GetDefault()->GetMemTableFiles(), + std::set({first, second})); + CreateNewManifest(); + ReopenDB(); + ASSERT_TRUE(versions_->HasMemTableFileTracking()); + ASSERT_EQ(versions_->GetColumnFamilySet()->GetDefault()->GetMemTableFiles(), + std::set({first, second})); + ASSERT_GT(versions_->current_next_file_number(), second); + VersionEdit deletion; + deletion.DeleteMemTableFile(first); + deletion.DeleteMemTableFile(second); + ASSERT_OK(LogAndApplyToDefaultCF(deletion)); + CreateNewManifest(); + ReopenDB(); + ASSERT_TRUE(versions_->HasMemTableFileTracking()); + for (auto* cfd : *versions_->GetColumnFamilySet()) { + ASSERT_TRUE(cfd->GetMemTableFiles().empty()); + } +} + +TEST_F(VersionSetTest, MemTableRegistryFailedCommitDoesNotEnableTracking) { + NewDB(); + VersionEdit invalid; + invalid.SetMemTableFileTracking(); + invalid.AddMemTableFile(0); + ASSERT_TRUE(LogAndApplyToDefaultCF(invalid).IsCorruption()); + ASSERT_FALSE(versions_->HasMemTableFileTracking()); + for (auto* cfd : *versions_->GetColumnFamilySet()) { + ASSERT_TRUE(cfd->GetMemTableFiles().empty()); + } +} + +TEST_F(VersionSetTest, MemTableRegistryReadOnlySkipsUnopenedCf) { + NewDB(); + const uint64_t first = versions_->NewFileNumber(); + const uint64_t second = versions_->NewFileNumber(); + VersionEdit addition; + addition.SetMemTableFileTracking(); + addition.AddMemTableFile(first); + ASSERT_OK(LogAndApplyToDefaultCF(addition)); + auto* cfd = versions_->GetColumnFamilySet()->GetColumnFamily(1); + addition.Clear(); + addition.SetColumnFamily(1); + addition.AddMemTableFile(second); + mutex_.Lock(); + Status s = versions_->LogAndApply(cfd, mutable_cf_options_, read_options_, + &addition, &mutex_, nullptr); + mutex_.Unlock(); + ASSERT_OK(s); + ASSERT_EQ(cfd->GetMemTableFiles(), std::set({second})); + ASSERT_EQ(versions_->GetColumnFamilySet()->GetDefault()->GetMemTableFiles(), + std::set({first})); + VersionSet read_only(dbname_, &db_options_, env_options_, table_cache_.get(), + &write_buffer_manager_, &write_controller_, nullptr, + nullptr, "", "", "", nullptr); + ASSERT_OK(read_only.Recover({column_families_.front()}, true)); + ASSERT_EQ(read_only.GetColumnFamilySet()->GetColumnFamily(1), nullptr); + ASSERT_EQ(read_only.GetColumnFamilySet()->GetDefault()->GetMemTableFiles(), + std::set({first})); + ASSERT_TRUE(read_only.HasMemTableFileTracking()); + ASSERT_GT(read_only.current_next_file_number(), second); +} + +TEST_F(VersionSetTest, MemTableRegistryRetirementQueuesObsoleteFile) { + NewDB(); + const uint64_t first = versions_->NewFileNumber(); + VersionEdit addition; + addition.SetMemTableFileTracking(); + addition.AddMemTableFile(first); + ASSERT_OK(LogAndApplyToDefaultCF(addition)); + VersionEdit deletion; + deletion.DeleteMemTableFile(first); + ASSERT_OK(LogAndApplyToDefaultCF(deletion)); + std::vector tables; + std::vector blobs; + std::vector manifests; + versions_->GetObsoleteFiles(&tables, &blobs, &manifests, first); + ASSERT_TRUE(tables.empty()); + manifests.clear(); + versions_->GetObsoleteFiles(&tables, &blobs, &manifests, first + 1); + ASSERT_EQ(tables.size(), 1U); + ASSERT_EQ(tables[0].metadata->fd.GetNumber(), first); + ASSERT_EQ(tables[0].path, + versions_->GetColumnFamilySet()->GetDefault()->ioptions()->cf_paths[0].path); + ASSERT_EQ(tables[0].metadata->table_reader_handle, nullptr); + tables[0].DeleteMetadata(); +} + +TEST_F(VersionSetTest, MemTableRegistryDropAndDiscardedEdit) { + NewDB(); + const uint64_t first = versions_->NewFileNumber(); + auto* cfd = versions_->GetColumnFamilySet()->GetColumnFamily(1); + ASSERT_NE(cfd, nullptr); + cfd->Ref(); + VersionEdit addition; + addition.SetColumnFamily(1); + addition.SetMemTableFileTracking(); + addition.AddMemTableFile(first); + mutex_.Lock(); + Status status = versions_->LogAndApply( + cfd, mutable_cf_options_, read_options_, &addition, &mutex_, nullptr); + mutex_.Unlock(); + ASSERT_OK(status); + ASSERT_EQ(cfd->GetMemTableFiles(), std::set({first})); + VersionEdit drop; + drop.SetColumnFamily(1); + drop.DropColumnFamily(); + mutex_.Lock(); + status = versions_->LogAndApply( + cfd, mutable_cf_options_, read_options_, &drop, &mutex_, nullptr); + mutex_.Unlock(); + ASSERT_OK(status); + ASSERT_TRUE(cfd->GetMemTableFiles().empty()); + std::vector retired_tables; + std::vector retired_blobs; + std::vector retired_manifests; + versions_->GetObsoleteFiles(&retired_tables, &retired_blobs, + &retired_manifests, first + 1); + ASSERT_EQ(retired_tables.size(), 1U); + ASSERT_EQ(retired_tables[0].metadata->fd.GetNumber(), first); + ASSERT_EQ(retired_tables[0].path, cfd->ioptions()->cf_paths[0].path); + retired_tables[0].DeleteMetadata(); + VersionEdit discarded; + discarded.SetColumnFamily(1); + discarded.AddMemTableFile(versions_->NewFileNumber()); + mutex_.Lock(); + status = versions_->LogAndApply( + cfd, mutable_cf_options_, read_options_, &discarded, &mutex_, nullptr); + EXPECT_TRUE(cfd->GetMemTableFiles().empty()); + cfd->UnrefAndTryDelete(); + mutex_.Unlock(); + ASSERT_TRUE(status.IsColumnFamilyDropped()); + ASSERT_EQ(versions_->GetColumnFamilySet()->GetColumnFamily(1), nullptr); + ASSERT_TRUE(versions_->HasMemTableFileTracking()); + CreateNewManifest(); + ReopenDB(); + ASSERT_TRUE(versions_->HasMemTableFileTracking()); + ASSERT_EQ(versions_->GetColumnFamilySet()->GetColumnFamily(1), nullptr); +} + TEST_F(VersionSetTest, PersistBlobFileStateInNewManifest) { // Initialize the database and add a couple of blob files, one with some // garbage in it, and one without any garbage. @@ -2624,6 +2774,64 @@ TEST_F(VersionSetAtomicGroupTest, EXPECT_EQ(num_initial_edits_ + kAtomicGroupSize, num_recovered_edits_); } +TEST_F(VersionSetAtomicGroupTest, MemTableRegistryAtomicTransition) { + SetupValidAtomicGroup(3); + const uint64_t first = versions_->NewFileNumber(); + const uint64_t second = versions_->NewFileNumber(); + edits_[0].SetMemTableFileTracking(); + edits_[0].AddMemTableFile(first); + edits_[1].DeleteMemTableFile(first); + edits_[1].AddMemTableFile(second); + edits_[2].SetNextFile(versions_->current_next_file_number()); + AddNewEditsToLog(3); + ASSERT_OK(versions_->Recover(column_families_, false)); + ASSERT_TRUE(versions_->HasMemTableFileTracking()); + ASSERT_EQ(versions_->GetColumnFamilySet()->GetDefault()->GetMemTableFiles(), + std::set({second})); + ASSERT_GT(versions_->current_next_file_number(), second); + auto* cfd = versions_->GetColumnFamilySet()->GetColumnFamily(0); + for (int level = 0; level < cfd->NumberLevels(); ++level) { + ASSERT_TRUE(cfd->current()->storage_info()->LevelFiles(level).empty()); + } + std::vector tables; + std::vector blobs; + std::vector manifests; + versions_->GetObsoleteFiles(&tables, &blobs, &manifests, second + 1); + ASSERT_TRUE(tables.empty()); // Historical replay must never schedule GC. +} + +TEST_F(VersionSetAtomicGroupTest, MemTableRegistryIncompleteAtomicGroup) { + SetupIncompleteTrailingAtomicGroup(3); + const uint64_t first = versions_->NewFileNumber(); + const uint64_t second = versions_->NewFileNumber(); + edits_[0].SetMemTableFileTracking(); + edits_[0].AddMemTableFile(first); + edits_[1].DeleteMemTableFile(first); + edits_[1].AddMemTableFile(second); + edits_[1].SetNextFile(versions_->current_next_file_number()); + AddNewEditsToLog(2); + ASSERT_OK(versions_->Recover(column_families_, false)); + ASSERT_FALSE(versions_->HasMemTableFileTracking()); + for (auto* cfd : *versions_->GetColumnFamilySet()) { + ASSERT_TRUE(cfd->GetMemTableFiles().empty()); + } +} + +TEST_F(VersionSetAtomicGroupTest, MemTableRegistryTrackingSurvivesEmptyList) { + SetupValidAtomicGroup(3); + const uint64_t first = versions_->NewFileNumber(); + edits_[0].SetMemTableFileTracking(); + edits_[0].AddMemTableFile(first); + edits_[1].DeleteMemTableFile(first); + edits_[2].SetNextFile(versions_->current_next_file_number()); + AddNewEditsToLog(3); + ASSERT_OK(versions_->Recover(column_families_, false)); + ASSERT_TRUE(versions_->HasMemTableFileTracking()); + ASSERT_TRUE(versions_->GetColumnFamilySet()->GetDefault() + ->GetMemTableFiles().empty()); + ASSERT_GT(versions_->current_next_file_number(), first); +} + TEST_F(VersionSetAtomicGroupTest, HandleValidAtomicGroupWithReactiveVersionSetReadAndApply) { const int kAtomicGroupSize = 3; @@ -3681,6 +3889,31 @@ TEST_F(VersionSetTestMissingFiles, NoFileMissing) { } } +TEST_F(VersionSetTestMissingFiles, MemTableRegistryConversionKeepsSameNumberSst) { + NewDB(); + const uint64_t first = versions_->NewFileNumber(); + VersionEdit addition; + addition.SetMemTableFileTracking(); + addition.AddMemTableFile(first); + ASSERT_OK(LogAndApplyToDefaultCF(addition)); + SstInfo sst(first, kDefaultColumnFamilyName, "a", 0, 100); + std::vector file_metas; + CreateDummyTableFiles({sst}, &file_metas); + VersionEdit conversion; + conversion.DeleteMemTableFile(first); + conversion.AddFile(0, file_metas[0]); + ASSERT_OK(LogAndApplyToDefaultCF(conversion)); + ASSERT_TRUE(versions_->GetColumnFamilySet()->GetDefault() + ->GetMemTableFiles().empty()); + std::vector tables; + std::vector blobs; + std::vector manifests; + versions_->GetObsoleteFiles(&tables, &blobs, &manifests, first + 1); + ASSERT_TRUE(tables.empty()); + ASSERT_EQ(versions_->GetColumnFamilySet()->GetDefault()->current() + ->storage_info()->LevelFiles(0).size(), 1U); +} + TEST_F(VersionSetTestMissingFiles, MinLogNumberToKeep2PC) { db_options_.allow_2pc = true; NewDB(); From 9eca109604dbfe865b09976346395ec463f8d05d Mon Sep 17 00:00:00 2001 From: leipeng Date: Mon, 5 Oct 2026 09:04:03 +0800 Subject: [PATCH 37/58] Track FileMmap MemTables through creation, flush, and recovery A directory scan cannot reveal a missing member of a MemTable set. Recoverable files need stable DB identities and a committed inventory before accepting writes. The DB must retain those files until flush or recovery commits their retirement, including failed conversions and another process crash during recovery. Creation, cache handoff, garbage collection, SST installation, and recovery share this ownership contract and must become active together. Normal switches use precreated, registered tables; the rare cache-miss path may register synchronously instead of waiting indefinitely. --- db/column_family.cc | 58 +- db/column_family.h | 4 + db/db_cspp_crash_safe_test.cc | 1408 +++++++++++++++++++++++++++----- db/db_impl/db_impl.cc | 20 + db/db_impl/db_impl.h | 8 +- db/db_impl/db_impl_files.cc | 8 + db/db_impl/db_impl_open.cc | 295 +++---- db/db_impl/db_impl_write.cc | 50 ++ db/db_memtable_convert_test.cc | 156 ++++ db/flush_job.cc | 15 +- db/memtable.cc | 11 +- db/memtable.h | 9 +- db/memtable_list.cc | 35 +- db/memtable_list.h | 1 + db/version_set.cc | 13 +- db/version_set.h | 1 + include/rocksdb/memtablerep.h | 17 +- 17 files changed, 1685 insertions(+), 424 deletions(-) diff --git a/db/column_family.cc b/db/column_family.cc index 48fe989cd..f69a82d77 100644 --- a/db/column_family.cc +++ b/db/column_family.cc @@ -1188,8 +1188,10 @@ void ColumnFamilyData::PrepareNewMemtableInBackground( } } auto beg = terark::qtime::now(); + uint64_t number = ioptions_.memtable_factory->SupportCrashSafe() + ? dummy_versions_->version_set()->NewFileNumber() : 0; auto tab = new MemTable(internal_comparator_, ioptions_, mutable_cf_options, - write_buffer_manager_, 0/*earliest_seq*/, id_); + write_buffer_manager_, 0/*earliest_seq*/, id_, number); auto end = terark::qtime::now(); RecordInHistogram(ioptions_.stats, MEMTAB_CONSTRUCT_NANOS, (end - beg).ns()); { @@ -1226,14 +1228,23 @@ MemTable* ColumnFamilyData::ConstructNewMemtable( tab->SetEarliestSequenceNumber(earliest_seq); } else { auto beg = terark::qtime::now(); + // dummy_versions_ remains alive for the lifetime of this CF, unlike current_. + uint64_t number = ioptions_.memtable_factory->SupportCrashSafe() + ? dummy_versions_->version_set()->NewFileNumber() : 0; tab = new MemTable(internal_comparator_, ioptions_, mutable_cf_options, - write_buffer_manager_, earliest_seq, id_); + write_buffer_manager_, earliest_seq, id_, number); auto end = terark::qtime::now(); RecordInHistogram(ioptions_.stats, MEMTAB_CONSTRUCT_NANOS, (end - beg).ns()); } return tab; } +MemTable* ColumnFamilyData::PeekPrecreatedMemtable() { + std::lock_guard lk(precreated_memtable_mutex_); + return precreated_memtable_list_.empty() + ? nullptr : precreated_memtable_list_.front().get(); +} + void ColumnFamilyData::ApplyMemTableFileEdit(const VersionEdit& edit) { ROCKSDB_ASSERT_EQ(edit.GetColumnFamily(), id_); if (edit.IsColumnFamilyDrop()) { @@ -1248,6 +1259,49 @@ void ColumnFamilyData::ApplyMemTableFileEdit(const VersionEdit& edit) { } } +void ColumnFamilyData::AddMemTableFileEdits(VersionEdit* edit) { + if (!ioptions_.memtable_crash_safe_recover) return; + std::lock_guard lk(precreated_memtable_mutex_); + auto add = [&](MemTable* mem) { + if (!mem->IsFileRegistered()) { + edit->AddMemTableFile(mem->GetFileNumber()); + edit->SetMemTableFileTracking(); + } + }; + add(mem_); + for (size_t i = 0; i < precreated_memtable_list_.size(); ++i) { + add(precreated_memtable_list_[i].get()); + } +} + +void ColumnFamilyData::PublishRegisteredMemTables() { + if (!ioptions_.memtable_crash_safe_recover) return; + TEST_SYNC_POINT("FlushJob::MemTableCache:BeforePublish"); + { + std::lock_guard lk(precreated_memtable_mutex_); + auto publish = [&](MemTable* mem) { + if (memtable_files_.count(mem->GetFileNumber())) { + mem->MarkFileRegistered(); + } + }; + publish(mem_); + for (size_t i = 0; i < precreated_memtable_list_.size(); ++i) { + publish(precreated_memtable_list_[i].get()); + } + } + TEST_SYNC_POINT("FlushJob::MemTableCache:AfterPublish"); +} + +void ColumnFamilyData::AddMemTableFileNumbers(std::vector* live) { + if (!ioptions_.memtable_factory->SupportCrashSafe()) return; + if (mem_ != nullptr) live->push_back(mem_->GetFileNumber()); + imm_.AddMemTableFileNumbers(live); + std::lock_guard lk(precreated_memtable_mutex_); + for (size_t i = 0; i < precreated_memtable_list_.size(); ++i) { + live->push_back(precreated_memtable_list_[i]->GetFileNumber()); + } +} + void ColumnFamilyData::CreateNewMemtable( const MutableCFOptions& mutable_cf_options, SequenceNumber earliest_seq) { if (mem_ != nullptr) { diff --git a/db/column_family.h b/db/column_family.h index 3e9aad600..47b5229aa 100644 --- a/db/column_family.h +++ b/db/column_family.h @@ -373,9 +373,13 @@ class ColumnFamilyData { uint64_t OldestLogToKeep(); void PrepareNewMemtableInBackground(const MutableCFOptions&); + MemTable* PeekPrecreatedMemtable(); // DB mutex must be held for registration and live-file collection. const std::set& GetMemTableFiles() const { return memtable_files_; } void ApplyMemTableFileEdit(const VersionEdit& edit); + void AddMemTableFileEdits(VersionEdit* edit); + void AddMemTableFileNumbers(std::vector* live); + void PublishRegisteredMemTables(); // See Memtable constructor for explanation of earliest_seq param. MemTable* ConstructNewMemtable(const MutableCFOptions& mutable_cf_options, diff --git a/db/db_cspp_crash_safe_test.cc b/db/db_cspp_crash_safe_test.cc index c2bb3ca0a..55e220bc7 100644 --- a/db/db_cspp_crash_safe_test.cc +++ b/db/db_cspp_crash_safe_test.cc @@ -8,8 +8,15 @@ #include #include #include +#include +#include +#include #include #include +#include +#include +#include +#include #include #include #include @@ -26,7 +33,9 @@ #include "db/db_test_util.h" #include "db/log_reader.h" #include "db/log_writer.h" +#include "db/memtable.h" #include "db/pre_release_callback.h" +#include "db/version_set.h" #include "file/filename.h" #include "file/file_util.h" #include "file/sequence_file_reader.h" @@ -34,7 +43,9 @@ #include "port/port.h" #include "port/stack_trace.h" #include "rocksdb/io_status.h" +#include "rocksdb/convenience.h" #include "rocksdb/statistics.h" +#include "rocksdb/utilities/checkpoint.h" #include "rocksdb/utilities/transaction_db.h" #include "rocksdb/wal_filter.h" #include "table/get_context.h" @@ -107,8 +118,74 @@ Options BaseCrashSafeOptions(const std::string& dbname, bool recover, std::vector ListLeftovers(const Options& options, const std::string& dir) { std::vector leftovers; - if (options.memtable_factory) { - options.memtable_factory->ListCrashSafeLeftovers(dir, &leftovers); + Env* env = options.env ? options.env : Env::Default(); + std::string current; + Status s = env->FileExists(CurrentFileName(dir)); + if (s.IsNotFound()) return leftovers; + EXPECT_OK(s); + if (!s.ok()) return leftovers; + s = ReadFileToString(env, CurrentFileName(dir), ¤t); + if (!s.ok()) { + ADD_FAILURE() << s.ToString(); + return leftovers; + } + EXPECT_FALSE(current.empty()); + if (current.empty()) return leftovers; + if (current.back() == '\n') current.pop_back(); + const std::string manifest = dir + "/" + current; + std::unique_ptr file; + s = env->GetFileSystem()->NewSequentialFile(manifest, FileOptions(), &file, + nullptr); + EXPECT_OK(s); + if (!s.ok()) return leftovers; + struct Reporter : log::Reader::Reporter { + void Corruption(size_t, const Status& status) override { + ADD_FAILURE() << status.ToString(); + } + } reporter; + auto input = std::make_unique(std::move(file), manifest); + log::Reader reader(nullptr, std::move(input), &reporter, true, 0); + std::map> registered; + auto apply = [&](const VersionEdit& edit) { + const uint32_t cf = edit.GetColumnFamily(); + if (edit.IsColumnFamilyDrop()) { + registered.erase(cf); + return; + } + for (uint64_t number : edit.GetMemTableFileDeletions()) { + registered[cf].erase(number); + } + for (uint64_t number : edit.GetMemTableFileAdditions()) { + registered[cf].insert(number); + } + }; + AtomicGroupReadBuffer group; + Slice record; + std::string scratch; + while (reader.ReadRecord(&record, &scratch)) { + VersionEdit edit; + s = edit.DecodeFrom(record); + EXPECT_OK(s); + if (!s.ok()) break; + s = group.AddEdit(&edit); + EXPECT_OK(s); + if (!s.ok()) break; + if (!edit.IsInAtomicGroup()) { + apply(edit); + } else if (group.IsFull()) { + for (const auto& member : group.replay_buffer()) apply(member); + group.Clear(); + } + } + const std::string path = !options.cf_paths.empty() + ? options.cf_paths[0].path + : !options.db_paths.empty() + ? options.db_paths[0].path + : dir; + for (const auto& cf : registered) { + for (uint64_t number : cf.second) { + leftovers.push_back(MakeTableFileName(path, number)); + } } return leftovers; } @@ -230,6 +307,171 @@ class DBCsppCrashSafeTest : public DBTestBase { : DBTestBase("db_cspp_crash_safe_test", /*env_do_fsync=*/false) {} }; +#if !defined(OS_WIN) +TEST_F(CrashChild, DISABLED_RegisteredMemTables) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (arg_ == "osl") SetupOsl(&options, true); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "first", "1")); + ASSERT_OK(static_cast(child_db)->TEST_SwitchMemtable()); + ASSERT_OK(child_db->Put(WriteOptions(), "second", "2")); + ASSERT_OK(static_cast(child_db)->TEST_SwitchMemtable()); + // The registered empty active file must also exist for prefix recovery. + ::_exit(42); +} + +TEST_F(DBCsppCrashSafeTest, MissingRegisteredMemTableUsesFullWal) { + Close(); + for (bool osl : {false, true}) { + for (int missing : {0, 1, 2, 3}) { + SCOPED_TRACE(osl); + SCOPED_TRACE(missing); + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "RegisteredMemTables", + osl ? "osl" : "cspp"), 42); + const auto registered = ListLeftovers(options, dbname_); + ASSERT_EQ(registered.size(), 3U); + if (missing == 3) { + for (const auto& path : registered) ASSERT_OK(env_->DeleteFile(path)); + } else { + ASSERT_OK(env_->DeleteFile(registered[missing])); + } + std::atomic converted{0}; + SyncPoint::GetInstance()->SetCallBack( + "MemTableRep::ConvertToSST:After", + [&](void*) { ++converted; }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(options)); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + // Recovery may convert an intact prefix before discovering the missing + // registered file, but must discard that prefix and replay the full WAL. + ASSERT_EQ(converted.load(), missing == 3 ? 0 : missing); + ASSERT_EQ(Get("first"), "1"); + ASSERT_EQ(Get("second"), "2"); + ASSERT_EQ(NumTableFilesAtLevel(0), 0); + ASSERT_OK(Flush()); + ASSERT_EQ(NumTableFilesAtLevel(0), 1); + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("first"), "1"); + ASSERT_EQ(Get("second"), "2"); + ASSERT_EQ(NumTableFilesAtLevel(0), 1); + Close(); + } + } +} + +TEST_F(DBCsppCrashSafeTest, FailedRegistrationCannotAcceptWrites) { + Close(); + for (bool osl : {false, true}) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("before", "safe")); + ASSERT_OK(dbfull()->TEST_SwitchMemtable()); + const auto caller = std::this_thread::get_id(); + std::atomic failed{0}; + std::atomic registering_cache{false}; + SyncPoint::GetInstance()->SetCallBack( + "FlushJob::BeforeManifest", [&](void*) { registering_cache.store(true); }); + SyncPoint::GetInstance()->SetCallBack( + "VersionSet::ProcessManifestWrites:AfterSyncManifest", [&](void* p) { + if (!registering_cache.load()) return; + EXPECT_NE(std::this_thread::get_id(), caller); + ++failed; + *static_cast(p) = IOStatus::IOError("register injection"); + }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_NOK(Flush()); + ASSERT_GT(failed.load(), 0); + auto* pending = dbfull()->GetVersionSet()->GetColumnFamilySet() + ->GetDefault()->PeekPrecreatedMemtable(); + ASSERT_NE(pending, nullptr); + ASSERT_FALSE(pending->IsFileRegistered()); + ASSERT_NOK(dbfull()->TEST_SwitchMemtable()); + ASSERT_NOK(Put("unregistered", "must-not-commit")); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_EQ(Get("before"), "safe"); + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("before"), "safe"); + ASSERT_EQ(Get("unregistered"), "NOT_FOUND"); + Close(); + } +} + +TEST_F(DBCsppCrashSafeTest, FailedInitialRegistrationClearsDbPointer) { + Close(); + for (bool osl : {false, true}) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + Destroy(options); + std::atomic failed{0}; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::RegisterMemTableFile:AfterLogAndApply", [&](void* p) { + ++failed; + *static_cast(p) = Status::IOError("initial register injection"); + }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* opened = nullptr; + const Status status = DB::Open(options, dbname_, &opened); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_NOK(status); + ASSERT_GT(failed.load(), 0); + ASSERT_EQ(opened, nullptr); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("after-failure", "safe")); + Close(); + } +} + + + +TEST_F(DBCsppCrashSafeTest, FailedNewColumnFamilyRegistrationRemainsReopenable) { + Close(); + for (bool osl : {false, true}) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("existing", "preserved")); + std::atomic failed{0}; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::RegisterMemTableFile:AfterLogAndApply", [&](void* p) { + ++failed; + *static_cast(p) = Status::IOError("new CF register injection"); + }); + SyncPoint::GetInstance()->EnableProcessing(); + ColumnFamilyHandle* handle = nullptr; + const Status created = db_->CreateColumnFamily(options, "failed", &handle); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_NOK(created); + ASSERT_GT(failed.load(), 0); + ASSERT_EQ(handle, nullptr); + ASSERT_EQ(Get("existing"), "preserved"); + // Closing and reopening also exercises manifest snapshots over this CF. + Close(); + ASSERT_OK(TryReopenWithColumnFamilies({kDefaultColumnFamilyName, "failed"}, + options)); + ASSERT_EQ(Get(0, "existing"), "preserved"); + ASSERT_EQ(Get(1, "existing"), "NOT_FOUND"); + std::unique_ptr iterator(db_->NewIterator(ReadOptions(), handles_[1])); + iterator->SeekToFirst(); + ASSERT_FALSE(iterator->Valid()); + ASSERT_OK(iterator->status()); + iterator.reset(); + Close(); + } +} + TEST_F(DBCsppCrashSafeTest, FileMmapRejectsReadOnlyAndSecondary) { Close(); for (bool osl : {false, true}) { @@ -259,19 +501,533 @@ TEST_F(DBCsppCrashSafeTest, FileMmapRejectsReadOnlyAndSecondary) { } } -TEST_F(DBCsppCrashSafeTest, CrashSafeRequiresFileMmap) { - for (const char* cls : {"CSPPMemTab", "OffsetSkipList"}) { - for (const char* mode : {"kDontConvert", "kDumpMem", "kFileMmap"}) { - SCOPED_TRACE(cls); +TEST_F(DBCsppCrashSafeTest, ReadWriteWalRecoveryFailureDoesNotAbort) { + class CorruptSecondRecord final : public WalFilter { + public: + int calls = 0; + const char* Name() const override { return "CorruptSecondRecord"; } + WalProcessingOption LogRecordFound(unsigned long long, const std::string&, + const WriteBatch&, WriteBatch*, + bool*) override { + return ++calls == 1 ? WalProcessingOption::kContinueProcessing + : WalProcessingOption::kCorruptedRecord; + } + }; + Close(); + for (bool osl : {false, true}) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "WalFilterFallsBackToWal", + osl ? "10" : "00"), 0); + auto list_sst = [&] { + std::vector children; + EXPECT_OK(env_->GetChildren(dbname_, &children)); + std::vector files; + for (const auto& child : children) { + if (child.size() >= 4 && child.compare(child.size() - 4, 4, ".sst") == 0) + files.push_back(child); + } + std::sort(files.begin(), files.end()); + return files; + }; + const auto before = list_sst(); + ASSERT_FALSE(before.empty()); + std::map original_files; + for (const auto& name : before) { + ASSERT_OK(ReadFileToString(env_, dbname_ + "/" + name, &original_files[name])); + } + CorruptSecondRecord filter; + options.memtable_crash_safe_recover = false; + options.wal_filter = &filter; + options.wal_recovery_mode = WALRecoveryMode::kAbsoluteConsistency; + DB* failed = nullptr; + const Status status = DB::Open(options, dbname_, &failed); + ASSERT_TRUE(status.IsCorruption()) << status.ToString(); + ASSERT_EQ(failed, nullptr); + ASSERT_EQ(filter.calls, 2); + for (const auto& file : original_files) { + std::string after; + ASSERT_OK(ReadFileToString(env_, dbname_ + "/" + file.first, &after)); + ASSERT_EQ(after, file.second); + } + options.wal_filter = nullptr; + options.memtable_crash_safe_recover = true; + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("a"), std::string(128, 'a')); + ASSERT_EQ(Get("b"), std::string(128, 'b')); + Close(); + } +} + +TEST_F(DBCsppCrashSafeTest, ManifestRolloverPreservesMemTableRegistry) { + Close(); + for (bool osl : {false, true}) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.max_manifest_file_size = 1; + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("first", "1")); + std::string before; + ASSERT_OK(ReadFileToString(env_, CurrentFileName(dbname_), &before)); + ASSERT_OK(Flush()); + ASSERT_OK(Put("second", "2")); + std::string after; + ASSERT_OK(ReadFileToString(env_, CurrentFileName(dbname_), &after)); + ASSERT_NE(before, after); + const auto registered = dbfull()->GetVersionSet()->GetColumnFamilySet() + ->GetDefault()->GetMemTableFiles(); + ASSERT_EQ(registered.size(), 2U); + const auto disk = ListLeftovers(options, dbname_); + ASSERT_EQ(disk.size(), 2U); + for (uint64_t number : registered) { + const auto path = MakeTableFileName(dbname_, number); + ASSERT_EQ(std::count(disk.begin(), disk.end(), path), 1); + ASSERT_OK(env_->FileExists(path)); + } + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("first"), "1"); + ASSERT_EQ(Get("second"), "2"); + Close(); + } +} + +TEST_F(CrashChild, DISABLED_LegacyManifestWal) { + Options options = BaseCrashSafeOptions(dbname_, false, false); + options.memtable_factory = std::make_shared(); + options.table_factory = Options().table_factory; + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + WriteOptions write; + write.sync = true; + ASSERT_OK(child_db->Put(write, "legacy-first", "one")); + ASSERT_OK(child_db->Put(write, "legacy-second", "two")); + ::_exit(42); +} + +TEST_F(DBCsppCrashSafeTest, LegacyManifestWithoutTrackingUsesFullWal) { + Close(); + for (bool osl : {false, true}) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "LegacyManifestWal"), 42); + ASSERT_TRUE(ListLeftovers(options, dbname_).empty()); + std::vector children; + ASSERT_OK(env_->GetChildren(dbname_, &children)); + uint64_t wal_number = 0; + for (const auto& child : children) { + uint64_t number; + FileType type; + if (ParseFileName(child, &number, &type) && type == kWalFile) + wal_number = std::max(wal_number, number); + } + ASSERT_NE(wal_number, 0U); + uint64_t wal_size = 0; + ASSERT_OK(env_->GetFileSize(LogFileName(dbname_, wal_number), &wal_size)); + ASSERT_GT(wal_size, 0U); + // Valid classic sidecar deliberately points past both keys. An inventory + // without tracking cannot justify skipping this WAL prefix. + PublishedSeqOnDisk record; + record.magic = 0x5145534255505343ULL; + record.version = 1; + record.header_size = sizeof(record); + record.wal_offset_kind = 1; + record.kind_since_wal = static_cast(wal_number); + record.generation = 2; + record.pubseq = 2; + record.wal_number = wal_number; + record.wal_offset = wal_size; + std::string sidecar(4096, '\0'); + std::memcpy(&sidecar[0], &record, sizeof(record)); + ASSERT_OK(WriteStringToFile(env_, sidecar, CrashSafePubSeqFileName(dbname_))); + std::atomic converted{0}; + std::atomic reads{0}; + SyncPoint::GetInstance()->SetCallBack( + "MemTableRep::ConvertToSST:After", [&](void*) { ++converted; }); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::RecoverLogFiles:BeforeReadWal", [&](void*) { ++reads; }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(options)); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_EQ(converted.load(), 0); + ASSERT_GT(reads.load(), 0); + ASSERT_EQ(Get("legacy-first"), "one"); + ASSERT_EQ(Get("legacy-second"), "two"); + ASSERT_TRUE(dbfull()->GetVersionSet()->HasMemTableFileTracking()); + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("legacy-first"), "one"); + ASSERT_EQ(Get("legacy-second"), "two"); + Close(); + } +} + + + +TEST_F(DBCsppCrashSafeTest, GarbageCollectionKeepsActiveAndCachedMemTables) { + Close(); + for (bool osl : {false, true}) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + Destroy(options); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(options)); + std::atomic published{0}; + SyncPoint::GetInstance()->SetCallBack( + "FlushJob::MemTableCache:BeforePublish", [&](void*) { + std::vector expected; + for (auto* cfd : *dbfull()->GetVersionSet()->GetColumnFamilySet()) { + for (uint64_t number : cfd->GetMemTableFiles()) { + expected.push_back(MakeTableFileName(dbname_, number)); + } + } + ASSERT_EQ(ListLeftovers(options, dbname_), expected); + for (const auto& path : expected) ASSERT_OK(env_->FileExists(path)); + }); + SyncPoint::GetInstance()->SetCallBack( + "FlushJob::MemTableCache:AfterPublish", [&](void*) { ++published; }); + ASSERT_OK(Put("first", "1")); + ASSERT_OK(Flush()); + ASSERT_GT(published.load(), 0); + ASSERT_OK(Put("active", "2")); + const auto registered = dbfull()->GetVersionSet()->GetColumnFamilySet() + ->GetDefault()->GetMemTableFiles(); + ASSERT_GE(registered.size(), 2U); + ASSERT_OK(db_->DisableFileDeletions()); + ASSERT_OK(db_->EnableFileDeletions(true)); + for (uint64_t number : registered) { + ASSERT_OK(env_->FileExists(MakeTableFileName(dbname_, number))); + } + ASSERT_EQ(Get("first"), "1"); + ASSERT_EQ(Get("active"), "2"); + Close(); + const auto kept = ListLeftovers(options, dbname_); + ASSERT_FALSE(kept.empty()); + for (const auto& path : kept) ASSERT_OK(env_->FileExists(path)); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("first"), "1"); + ASSERT_EQ(Get("active"), "2"); + Close(); + } +} + +TEST_F(DBCsppCrashSafeTest, EmptyFlushRetiresSourceFileWithoutFullScan) { + Close(); + for (bool osl : {false, true}) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.delete_obsolete_files_period_micros = UINT64_MAX; + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_OK(TryReopen(options)); + auto* cfd = dbfull()->GetVersionSet()->GetColumnFamilySet()->GetDefault(); + const std::string source = + MakeTableFileName(dbname_, cfd->mem()->GetFileNumber()); + ASSERT_OK(env_->FileExists(source)); + ASSERT_OK(dbfull()->TEST_SwitchMemtable()); + ASSERT_OK(Flush()); + ASSERT_OK(dbfull()->TEST_WaitForBackgroundWork()); + ASSERT_OK(dbfull()->TEST_WaitForPurge()); + std::vector files; + db_->GetLiveFilesMetaData(&files); + ASSERT_TRUE(files.empty()); + ASSERT_TRUE(env_->FileExists(source).IsNotFound()); + const auto active = ListLeftovers(options, dbname_); + ASSERT_EQ(active.size(), 2U); + for (const auto& path : active) ASSERT_OK(env_->FileExists(path)); + const auto converted = MakeTableFileName(dbname_, cfd->mem()->GetFileNumber()); + ASSERT_OK(Put("converted", "preserved")); + ASSERT_OK(Flush()); + ASSERT_OK(dbfull()->TEST_WaitForBackgroundWork()); + ASSERT_OK(dbfull()->TEST_WaitForPurge()); + db_->GetLiveFilesMetaData(&files); + ASSERT_EQ(files.size(), 1U); + ASSERT_EQ(MakeTableFileName(dbname_, files[0].file_number), converted); + ASSERT_OK(env_->FileExists(converted)); + for (const auto& path : ListLeftovers(options, dbname_)) + ASSERT_OK(env_->FileExists(path)); + ASSERT_EQ(Get("converted"), "preserved"); + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("converted"), "preserved"); + Close(); + } +} + +TEST_F(DBCsppCrashSafeTest, DroppedCfRetiresSourceFilesWithoutFullScan) { + Close(); + for (bool osl : {false, true}) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.delete_obsolete_files_period_micros = UINT64_MAX; + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_OK(TryReopen(options)); + CreateAndReopenWithCF({"retire"}, options); + ASSERT_OK(Put(0, "keep", "one")); + ASSERT_OK(Put(1, "drop", "two")); + const auto registered = dbfull()->GetVersionSet()->GetColumnFamilySet() + ->GetColumnFamily(1)->GetMemTableFiles(); + ASSERT_EQ(registered.size(), 2U); + for (uint64_t number : registered) + ASSERT_OK(env_->FileExists(MakeTableFileName(dbname_, number))); + ASSERT_OK(db_->DropColumnFamily(handles_[1])); + ASSERT_OK(db_->DestroyColumnFamilyHandle(handles_[1])); + handles_.pop_back(); + // An ordinary flush provides normal obsolete-file GC, without a scan. + ASSERT_OK(Flush(0)); + ASSERT_OK(dbfull()->TEST_WaitForBackgroundWork()); + ASSERT_OK(dbfull()->TEST_WaitForPurge()); + for (uint64_t number : registered) + ASSERT_TRUE(env_->FileExists(MakeTableFileName(dbname_, number)).IsNotFound()); + ASSERT_EQ(dbfull()->GetVersionSet()->GetColumnFamilySet() + ->GetColumnFamily(1), nullptr); + for (const auto& path : ListLeftovers(options, dbname_)) + ASSERT_OK(env_->FileExists(path)); + std::vector files; + db_->GetLiveFilesMetaData(&files); + ASSERT_EQ(files.size(), 1U); + ASSERT_OK(env_->FileExists(MakeTableFileName(dbname_, files[0].file_number))); + ASSERT_EQ(Get(0, "keep"), "one"); + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("keep"), "one"); + Close(); + } +} + +TEST_F(DBCsppCrashSafeTest, CheckpointDoesNotHardLinkWritableMemTable) { + Close(); + for (bool osl : {false, true}) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("checkpoint", "original")); + const auto registered = dbfull()->GetVersionSet()->GetColumnFamilySet() + ->GetDefault()->GetMemTableFiles(); + ASSERT_EQ(registered.size(), 2U); + const uint64_t number = dbfull()->GetVersionSet()->GetColumnFamilySet() + ->GetDefault()->mem()->GetFileNumber(); + ASSERT_EQ(registered.count(number), 1U); + const std::string checkpoint_dir = dbname_ + ".checkpoint"; + Options copy_options = options; + copy_options.wal_dir = checkpoint_dir; + ASSERT_OK(DestroyDB(checkpoint_dir, copy_options)); + Checkpoint* raw = nullptr; + ASSERT_OK(Checkpoint::Create(db_, &raw)); + std::unique_ptr checkpoint(raw); + ASSERT_OK(checkpoint->CreateCheckpoint(checkpoint_dir, UINT64_MAX)); + // This memtable is still mutable, and hence must not be a live SST. + const auto& after = dbfull()->GetVersionSet()->GetColumnFamilySet() + ->GetDefault()->GetMemTableFiles(); + if (after.count(number)) { + ASSERT_TRUE(env_->FileExists(MakeTableFileName(checkpoint_dir, number)) + .IsNotFound()); + } + ASSERT_OK(Put("checkpoint", "source-changed")); + DB* copy_raw = nullptr; + ASSERT_OK(DB::Open(copy_options, checkpoint_dir, ©_raw)); + std::unique_ptr copy(copy_raw); + std::string value; + ASSERT_OK(copy->Get(ReadOptions(), "checkpoint", &value)); + ASSERT_EQ(value, "original"); + copy.reset(); + checkpoint.reset(); + ASSERT_OK(DestroyDB(checkpoint_dir, copy_options)); + Close(); + } +} + + + +TEST_F(DBCsppCrashSafeTest, OpenAndCreateColumnFamilyRegisterNextMemTable) { + Close(); + for (bool osl : {false, true}) { + for (bool atomic : {false, true}) { + SCOPED_TRACE(osl); + SCOPED_TRACE(atomic); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.atomic_flush = atomic; + options.avoid_flush_during_shutdown = true; + if (osl) SetupOsl(&options, true); + Destroy(options); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(options)); + auto* default_cfd = dbfull()->GetVersionSet()->GetColumnFamilySet() + ->GetColumnFamily(0); + dbfull()->TEST_LockMutex(); + auto* default_head = default_cfd->PeekPrecreatedMemtable(); + const uint64_t default_cache = default_head ? default_head->GetFileNumber() : 0; + dbfull()->TEST_UnlockMutex(); + ASSERT_NE(default_cache, 0U); + ASSERT_EQ(default_cfd->GetMemTableFiles() + .count(default_cache), 1U); + + ColumnFamilyHandle* handle = nullptr; + ASSERT_OK(db_->CreateColumnFamily(options, "bootstrap", &handle)); + auto* created_cfd = dbfull()->GetVersionSet()->GetColumnFamilySet() + ->GetColumnFamily(handle->GetID()); + dbfull()->TEST_LockMutex(); + auto* created_head = created_cfd->PeekPrecreatedMemtable(); + const uint64_t created_cache = created_head ? created_head->GetFileNumber() : 0; + dbfull()->TEST_UnlockMutex(); + ASSERT_NE(created_cache, 0U); + ASSERT_EQ(created_cfd->GetMemTableFiles() + .count(created_cache), 1U); + std::atomic front_registrations{0}; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::RegisterMemTableFile:BeforeLogAndApply", + [&](void*) { ++front_registrations; }); + ASSERT_OK(Put("default", "1")); + ASSERT_OK(Flush()); + ASSERT_EQ(default_cfd->mem()->GetFileNumber(), default_cache); + ASSERT_TRUE(default_cfd->mem()->IsFileRegistered()); + ASSERT_OK(db_->Put(WriteOptions(), handle, "created", "2")); + ASSERT_OK(db_->Flush(FlushOptions(), handle)); + ASSERT_EQ(created_cfd->mem()->GetFileNumber(), created_cache); + ASSERT_TRUE(created_cfd->mem()->IsFileRegistered()); + ASSERT_EQ(front_registrations.load(), 0); + ASSERT_OK(db_->DestroyColumnFamilyHandle(handle)); + Close(); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + } + } +} + + + + +#endif + +TEST_F(DBCsppCrashSafeTest, CrashSafeRequiresFileMmapFactories) { + Close(); + for (bool osl : {false, true}) { + for (const char* mode : {"kDontConvert", "kDumpMem", "SkipList"}) { + SCOPED_TRACE(osl); SCOPED_TRACE(mode); - const SidePluginRepo repo; - auto factory = PluginFactorySP::AcquirePlugin( - cls, {{"convert_to_sst", mode}}, repo); - ASSERT_EQ(factory->SupportCrashSafe(), std::string(mode) == "kFileMmap"); + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + Options unsupported = options; + unsupported.memtable_factory = std::string(mode) == "SkipList" + ? std::shared_ptr(new SkipListFactory) + : EasyNewMemTableRep(osl ? "OffsetSkipList" : "CSPPMemTab", + json({{"mem_cap", 16777216}, {"convert_to_sst", mode}}).dump()); + Destroy(options); + DB* rejected = nullptr; + ASSERT_TRUE(DB::Open(unsupported, dbname_, &rejected).IsInvalidArgument()); + ASSERT_EQ(rejected, nullptr); + ASSERT_OK(TryReopen(options)); + ColumnFamilyHandle* handle = nullptr; + ASSERT_TRUE(db_->CreateColumnFamily(unsupported, "mixed", &handle) + .IsInvalidArgument()); + ASSERT_EQ(handle, nullptr); + Close(); + options.memtable_crash_safe_recover = false; + options.avoid_flush_during_shutdown = true; + unsupported.memtable_crash_safe_recover = false; + ASSERT_OK(TryReopen(options)); + ASSERT_OK(db_->CreateColumnFamily(unsupported, "mixed", &handle)); + handles_.push_back(handle); + ASSERT_OK(Put("file", "value")); + ASSERT_OK(db_->Put(WriteOptions(), handle, "mixed", "value")); + if (unsupported.memtable_factory->SupportConvertToSST()) { + ASSERT_OK(db_->Flush(FlushOptions(), handle)); + } + for (auto* cfd : *dbfull()->GetVersionSet()->GetColumnFamilySet()) { + ASSERT_TRUE(cfd->GetMemTableFiles().empty()); + } + Close(); + ASSERT_OK(TryReopenWithColumnFamilies( + {"default", "mixed"}, std::vector{options, unsupported})); + ASSERT_EQ(Get(0, "file"), "value"); + ASSERT_EQ(Get(1, "mixed"), "value"); + Close(); } } } +TEST_F(DBCsppCrashSafeTest, RecoverOffKeepsUnregisteredMemTableFiles) { + Close(); + for (bool osl : {false, true}) { + SCOPED_TRACE(osl); + Options options = BaseCrashSafeOptions(dbname_, false, false); + options.experimental_mempurge_threshold = 2.0; + if (osl) SetupOsl(&options, true); + Destroy(options); + int registrations = 0, waits = 0, inspected = 0; + std::atomic converts{0}, purges{0}; + SyncPoint::GetInstance()->SetCallBack( + "MemTableRep::ConvertToSST:After", [&](void*) { ++converts; }); + for (const char* point : {"DBImpl::FlushJob:MemPurgeSuccessful", + "DBImpl::FlushJob:MemPurgeUnsuccessful"}) { + SyncPoint::GetInstance()->SetCallBack(point, [&](void*) { ++purges; }); + } + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::RegisterMemTableFile:BeforeLogAndApply", + [&](void*) { ++registrations; }); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::SwitchMemtable:MemTableCacheMiss", + [&](void*) { ++waits; }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(options)); + // Open reuses this hook for ordinary recovery MANIFEST writes. + registrations = 0; + ASSERT_OK(Put("sst", "1")); + ASSERT_OK(Flush()); + ASSERT_GT(converts.load(), 0); + ASSERT_EQ(purges.load(), 0); + auto* cfd = dbfull()->GetVersionSet()->GetColumnFamilySet()->GetDefault(); + ASSERT_NE(cfd->PeekPrecreatedMemtable(), nullptr); + const uint64_t cached = cfd->PeekPrecreatedMemtable()->GetFileNumber(); + ASSERT_OK(Put("imm", "2")); + const uint64_t immutable = cfd->mem()->GetFileNumber(); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::SwitchMemtable:BeforeInstallMemTable", [&](void*) { + ASSERT_OK(db_->DisableFileDeletions()); + ASSERT_OK(db_->EnableFileDeletions(true)); + ASSERT_OK(env_->FileExists(MakeTableFileName(dbname_, cached))); + ++inspected; + }); + ASSERT_OK(dbfull()->TEST_SwitchMemtable()); + SyncPoint::GetInstance()->ClearCallBack( + "DBImpl::SwitchMemtable:BeforeInstallMemTable"); + ASSERT_EQ(inspected, 1); + ASSERT_EQ(cfd->mem()->GetFileNumber(), cached); + ASSERT_OK(Put("active", "3")); + // Refill through the ordinary cache producer to check all three live roles. + cfd->PrepareNewMemtableInBackground(*cfd->GetLatestMutableCFOptions()); + auto* precreated = cfd->PeekPrecreatedMemtable(); + ASSERT_NE(precreated, nullptr); + const uint64_t next = precreated->GetFileNumber(); + for (auto* cfd : *dbfull()->GetVersionSet()->GetColumnFamilySet()) { + ASSERT_TRUE(cfd->GetMemTableFiles().empty()); + } + ASSERT_OK(db_->DisableFileDeletions()); + ASSERT_OK(db_->EnableFileDeletions(true)); + for (uint64_t number : {immutable, cached, next}) { + ASSERT_OK(env_->FileExists(MakeTableFileName(dbname_, number))); + } + ASSERT_EQ(registrations, 0); + ASSERT_EQ(waits, 0); + ASSERT_EQ(Get("imm"), "2"); + ASSERT_EQ(Get("active"), "3"); + Close(); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("sst"), "1"); + ASSERT_EQ(Get("imm"), "2"); + ASSERT_EQ(Get("active"), "3"); + Close(); + } +} + TEST_F(DBCsppCrashSafeTest, ImmutableFactoryConvertMode) { Close(); const SidePluginRepo repo; @@ -309,8 +1065,27 @@ TEST_F(DBCsppCrashSafeTest, ImmutableFactoryConvertMode) { SCOPED_TRACE(cls); auto factory = PluginFactorySP::AcquirePlugin( cls, params, repo); + ASSERT_EQ(factory->SupportCrashSafe(), std::string(mode) == "kFileMmap"); auto* manip = PluginManip::AcquirePlugin(cls, {}, repo); check(factory, manip, mode); + InternalKeyComparator icmp(BytewiseComparator()); + MemTable::KeyComparator cmp(icmp); + Arena arena; + MutableCFOptions moptions(Options{}); + const std::string path = MakeTableFileName(dbname_, 900000); + if (std::string(mode) == "kFileMmap") { + ASSERT_DEATH(factory->CreateMemTableRep( + "", moptions, cmp, &arena, nullptr, nullptr, 0), "memtable_file_path"); + } + std::unique_ptr rep(factory->CreateMemTableRep( + path, moptions, cmp, &arena, nullptr, nullptr, 0)); + ASSERT_EQ(enum_stdstr(rep->GetConvertKind()), mode); + ASSERT_EQ(rep->SupportConvertToSST(), std::string(mode) != "kDontConvert"); + ASSERT_EQ(rep->SupportCrashSafe(), std::string(mode) == "kFileMmap"); + rep.reset(); + if (std::string(mode) == "kFileMmap") { + ASSERT_OK(env_->DeleteFile(path)); + } } for (const char* cls : {"CSPPMemTabTable", "OffsetSkipListTable"}) { SCOPED_TRACE(cls); @@ -337,23 +1112,21 @@ TEST_F(DBCsppCrashSafeTest, RecoveredTableUsesFileSequenceBound) { MutableCFOptions moptions(options); WriteBufferManager wb(options.db_write_buffer_size); std::unique_ptr mem(new MemTable( - icmp, ioptions, moptions, &wb, kMaxSequenceNumber, 0)); + icmp, ioptions, moptions, &wb, kMaxSequenceNumber, 0, 1)); ASSERT_OK(mem->Add(1, kTypeValue, "key", "old", nullptr)); ASSERT_OK(mem->Add(3, kTypeValue, "key", "new", nullptr)); ASSERT_OK(mem->Add(4, kTypeValue, "ghost", "unpublished", nullptr)); mem->MarkImmutable(); - const auto leftovers = ListLeftovers(options, dbname_); - ASSERT_EQ(leftovers.size(), 1U); - const std::string leftover = leftovers[0] + ".recovery"; - CopyFile(leftovers[0], leftover); + const std::string leftover = MakeTableFileName(dbname_, 2); + CopyFile(MakeTableFileName(dbname_, 1), leftover); mem.reset(); IntTblPropCollectorFactories collectors; TableBuilderOptions tbo(ioptions, moptions, icmp, &collectors, options.compression, options.compression_opts, 0, - "default", 0); + kDefaultColumnFamilyName, 0); FileMetaData meta; - meta.fd = FileDescriptor(1, 0, 0); + meta.fd = FileDescriptor(2, 0, 0); meta.fd.smallest_seqno = 0; // The published bound need not be the sequence of any physical entry. meta.fd.largest_seqno = 2; @@ -363,7 +1136,7 @@ TEST_F(DBCsppCrashSafeTest, RecoveredTableUsesFileSequenceBound) { ASSERT_EQ(meta.fd.smallest_seqno, 0U); ASSERT_EQ(meta.fd.largest_seqno, 2U); - const std::string fname = TableFileName(options.cf_paths, 1, 0); + const std::string fname = TableFileName(options.cf_paths, 2, 0); for (SequenceNumber limit : {meta.fd.largest_seqno, SequenceNumber(4), SequenceNumber(0)}) { SCOPED_TRACE(limit); @@ -420,7 +1193,7 @@ TEST_F(DBCsppCrashSafeTest, RecoveredTableIteratorUsesFileSequenceBound) { MutableCFOptions moptions(options); WriteBufferManager wb(options.db_write_buffer_size); auto mem = std::make_unique( - icmp, ioptions, moptions, &wb, kMaxSequenceNumber, 0); + icmp, ioptions, moptions, &wb, kMaxSequenceNumber, 0, 1); std::vector physical; for (const auto& entry : {std::make_pair("b", 1), {"d", 2}, {"b", 3}, {"d", 5}, {"c", 6}, {"b", 7}, {"a", 8}, @@ -435,22 +1208,20 @@ TEST_F(DBCsppCrashSafeTest, RecoveredTableIteratorUsesFileSequenceBound) { }; std::sort(physical.begin(), physical.end(), less); mem->MarkImmutable(); - const auto leftovers = ListLeftovers(options, dbname_); - ASSERT_EQ(leftovers.size(), 1U); - const std::string leftover = leftovers[0] + ".iterator"; - CopyFile(leftovers[0], leftover); + const std::string leftover = MakeTableFileName(dbname_, 2); + CopyFile(MakeTableFileName(dbname_, 1), leftover); mem.reset(); IntTblPropCollectorFactories collectors; TableBuilderOptions tbo(ioptions, moptions, icmp, &collectors, options.compression, options.compression_opts, 0, - "default", 0); + kDefaultColumnFamilyName, 0); FileMetaData meta; - meta.fd = FileDescriptor(1, 0, 0); + meta.fd = FileDescriptor(2, 0, 0); meta.fd.smallest_seqno = 0; meta.fd.largest_seqno = 4; ASSERT_OK(options.memtable_factory->RecoverCrashSafeMemTableToSST( leftover, &meta, tbo)); - const std::string fname = TableFileName(options.cf_paths, 1, 0); + const std::string fname = TableFileName(options.cf_paths, 2, 0); for (SequenceNumber limit : {SequenceNumber(0), SequenceNumber(4), SequenceNumber(9), kMaxSequenceNumber}) { SCOPED_TRACE(limit); @@ -595,7 +1366,7 @@ TEST_F(DBCsppCrashSafeTest, ConvertedTableVisibilityFilter) { MutableCFOptions moptions(options); WriteBufferManager wb(options.db_write_buffer_size); auto mem = std::make_unique( - icmp, ioptions, moptions, &wb, kMaxSequenceNumber, 0); + icmp, ioptions, moptions, &wb, kMaxSequenceNumber, 0, 1); ASSERT_OK(mem->Add(1, kTypeValue, "a", "1", nullptr)); ASSERT_OK(mem->Add(2, kTypeValue, "b", "2", nullptr)); ASSERT_OK(mem->Add(3, kTypeValue, "b", "3", nullptr)); @@ -603,7 +1374,7 @@ TEST_F(DBCsppCrashSafeTest, ConvertedTableVisibilityFilter) { IntTblPropCollectorFactories collectors; TableBuilderOptions tbo(ioptions, moptions, icmp, &collectors, options.compression, options.compression_opts, 0, - "default", 0); + kDefaultColumnFamilyName, 0); FileMetaData meta; meta.fd = FileDescriptor(1, 0, 0); meta.fd.smallest_seqno = 1; @@ -656,6 +1427,133 @@ TEST_F(DBCsppCrashSafeTest, ConvertedTableVisibilityFilter) { } } +TEST_F(DBCsppCrashSafeTest, CsppSelfMmapUnmapsWholeFile) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + ASSERT_OK(env_->CreateDirIfMissing(dbname_)); + options.cf_paths = {{dbname_, 0}}; + InternalKeyComparator icmp(options.comparator); + ImmutableOptions ioptions(options); + MutableCFOptions moptions(options); + WriteBufferManager wb(options.db_write_buffer_size); + std::unique_ptr mem(new MemTable( + icmp, ioptions, moptions, &wb, kMaxSequenceNumber, 0, 1)); + ASSERT_OK(mem->Add(1, kTypeValue, "key", "value", nullptr)); + const std::string path = MakeTableFileName(dbname_, 2); + CopyFile(MakeTableFileName(dbname_, 1), path); + mem.reset(); + const int fd = ::open(path.c_str(), O_RDWR); + ASSERT_GE(fd, 0); + terark::DFA_MmapHeader hdr{}; + ASSERT_EQ(::pread(fd, &hdr, sizeof(hdr), 0), + static_cast(sizeof(hdr))); + ASSERT_EQ(::ftruncate(fd, hdr.file_size + 2 * ::sysconf(_SC_PAGESIZE)), 0); + const size_t physical_size = hdr.file_size + 2 * ::sysconf(_SC_PAGESIZE); + const auto original = hdr; + std::vector buffer((physical_size + 7) / 8); + ASSERT_EQ(::pread(fd, buffer.data(), physical_size, 0), + static_cast(physical_size)); + auto check_unmapped = [&] { + std::ifstream maps("/proc/self/maps"); + ASSERT_TRUE(maps.good()); + for (std::string line; std::getline(maps, line);) { + EXPECT_EQ(line.find(path), std::string::npos) << "leaked mapping: " << line; + } + }; + for (int entry = 0; entry < 4; ++entry) { + // self(path), self(fd), load(path), load(fd). + for (int damage = 0; damage < 7; ++damage) { + SCOPED_TRACE(entry); + SCOPED_TRACE(damage); + if (damage == 6 && entry < 2) continue; // load-only format check. + hdr = original; + size_t length = physical_size; + if (damage == 1) hdr.num_blocks = 0; // finish_load_mmap failure. + if (damage == 2) length = 0; + if (damage == 3) length = sizeof(hdr) - 1; + if (damage == 4) hdr.file_size = sizeof(hdr) - 1; + if (damage == 5) hdr.file_size = physical_size + 1; + if (damage == 6) hdr.magic[0] = '!'; + ASSERT_EQ(::ftruncate(fd, physical_size), 0); + ASSERT_EQ(::pwrite(fd, buffer.data(), physical_size, 0), + static_cast(physical_size)); + ASSERT_EQ(::pwrite(fd, &hdr, sizeof(hdr), 0), + static_cast(sizeof(hdr))); + ASSERT_EQ(::ftruncate(fd, length), 0); + auto open = [&] { + if (entry < 2) { + terark::MainPatricia trie(0, 16 << 20, + terark::Patricia::NoWriteReadOnly); + if (entry == 0) trie.self_mmap(path); + else trie.self_mmap(fd, false); + EXPECT_EQ(trie.get_mmap().size(), original.file_size); + } else { + std::unique_ptr trie(entry == 2 + ? terark::BaseDFA::load_mmap(path, false) + : terark::BaseDFA::load_mmap(fd)); + EXPECT_EQ(trie->get_mmap().size(), original.file_size); + } + }; + if (damage == 0) { + ASSERT_NO_THROW(open()); + } else { + try { + open(); + FAIL() << "expected invalid_argument"; + } catch (const std::invalid_argument& ex) { + if ((entry == 0 || entry == 2) && damage >= 2 && damage <= 5) { + EXPECT_NE(std::string(ex.what()).find(path), std::string::npos); + } + } + } + ASSERT_NE(::fcntl(fd, F_GETFD), -1); // Caller retains its descriptor. + check_unmapped(); + } + } + for (bool load : {false, true}) { + for (int damage = 0; damage < 5; ++damage) { + SCOPED_TRACE(load); + SCOPED_TRACE(damage); + auto* header = reinterpret_cast(buffer.data()); + *header = original; + const void* data = buffer.data(); + size_t length = physical_size; + if (damage == 1) { data = nullptr; length = 0; } + if (damage == 2) length = sizeof(original) - 1; + if (damage == 3) header->file_size = sizeof(original) - 1; + if (damage == 4) header->file_size = physical_size + 1; + buffer.back() = 0x87654321; + auto borrow = [&] { + if (load) { + std::unique_ptr trie( + terark::BaseDFA::load_mmap_user_mem(data, length)); + ASSERT_NE(trie, nullptr); + EXPECT_EQ(trie->get_mmap().size(), original.file_size); + } else { + terark::MainPatricia trie(0, 16 << 20, + terark::Patricia::NoWriteReadOnly); + trie.self_mmap_user_mem(data, length); + EXPECT_EQ(trie.get_mmap().size(), original.file_size); + } + }; + if (damage == 0) { + ASSERT_NO_THROW(borrow()); + } else { + ASSERT_THROW(borrow(), std::invalid_argument); + } + // Borrowers neither free the buffer nor alter its logical or extra tail. + ASSERT_EQ(header->file_size, damage == 3 ? sizeof(original) - 1 + : damage == 4 ? physical_size + 1 + : original.file_size); + ASSERT_EQ(buffer.back(), 0x87654321U); + buffer.back() = 0x12345678; + ASSERT_EQ(buffer.back(), 0x12345678U); + } + } + ::close(fd); +} + TEST_F(DBCsppCrashSafeTest, OslRecoveryUnmapsWholeFile) { Close(); Options options = BaseCrashSafeOptions(dbname_, true, false); @@ -668,26 +1566,24 @@ TEST_F(DBCsppCrashSafeTest, OslRecoveryUnmapsWholeFile) { MutableCFOptions moptions(options); WriteBufferManager wb(options.db_write_buffer_size); std::unique_ptr mem(new MemTable( - icmp, ioptions, moptions, &wb, kMaxSequenceNumber, 0)); + icmp, ioptions, moptions, &wb, kMaxSequenceNumber, 0, 1)); ASSERT_OK(mem->Add(1, kTypeValue, "key", "value", nullptr)); - const auto leftovers = ListLeftovers(options, dbname_); - ASSERT_EQ(leftovers.size(), 1U); - const std::string leftover = leftovers[0] + ".review-leak"; - CopyFile(leftovers[0], leftover); + const std::string leftover = MakeTableFileName(dbname_, 2); + CopyFile(MakeTableFileName(dbname_, 1), leftover); mem.reset(); IntTblPropCollectorFactories collectors; TableBuilderOptions tbo(ioptions, moptions, icmp, &collectors, options.compression, options.compression_opts, 0, - "default", 0); + kDefaultColumnFamilyName, 0); FileMetaData meta; - meta.fd = FileDescriptor(1, 0, 0); + meta.fd = FileDescriptor(2, 0, 0); meta.fd.smallest_seqno = 0; meta.fd.largest_seqno = 1; ASSERT_OK(options.memtable_factory->RecoverCrashSafeMemTableToSST( leftover, &meta, tbo)); std::ifstream maps("/proc/self/maps"); ASSERT_TRUE(maps.good()); - const auto fname = TableFileName(options.cf_paths, 1, 0); + const auto fname = TableFileName(options.cf_paths, 2, 0); for (std::string line; std::getline(maps, line);) { EXPECT_EQ(line.find(fname), std::string::npos) << "leaked mapping: " << line; } @@ -746,7 +1642,9 @@ TEST_F(DBCsppCrashSafeTest, SecondCrashAfterConvertFailure) { ASSERT_EQ(RunCrashChild(dbname_, "SecondCrashRecover", child_options), 43); PublishedSeqOnDisk rec; ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); - ASSERT_EQ(rec.generation & 1, after_open ? 0U : 1U); + // Recovery does not invalidate a valid published cursor: registered + // source files survive conversion failure and can be retried. + ASSERT_EQ(rec.generation & 1, 0U); ASSERT_OK(TryReopen(options)); EXPECT_EQ(Get("a"), "1"); EXPECT_EQ(Get("b"), "2"); @@ -863,6 +1761,11 @@ TEST_F(CrashChild, DISABLED_LogRefRecoveryIgnoresCounters) { std::string(128, 'v'))); } ASSERT_OK(child_db->Put(WriteOptions(), "inline", "v")); + auto* cfd = static_cast(child_db)->GetVersionSet() + ->GetColumnFamilySet()->GetDefault(); + ASSERT_OK(WriteStringToFile( + options.env, MakeTableFileName(dbname_, cfd->mem()->GetFileNumber()), + dbname_ + "/active-file")); ::_exit(42); } @@ -875,8 +1778,11 @@ TEST_F(DBCsppCrashSafeTest, LogRefRecoveryIgnoresCounters) { Destroy(options); ASSERT_EQ(RunCrashChild(dbname_, "LogRefRecoveryIgnoresCounters", config), 42); const auto leftovers = ListLeftovers(options, dbname_); - ASSERT_EQ(leftovers.size(), 1U); - const int fd = ::open(leftovers[0].c_str(), O_RDONLY); + ASSERT_EQ(leftovers.size(), 2U); + std::string active; + ASSERT_OK(ReadFileToString(env_, dbname_ + "/active-file", &active)); + ASSERT_EQ(std::count(leftovers.begin(), leftovers.end(), active), 1); + const int fd = ::open(active.c_str(), O_RDWR); ASSERT_GE(fd, 0); // Header statistics are not a source of truth for WAL references. const size_t offset = config[0] == 'O' @@ -884,21 +1790,27 @@ TEST_F(DBCsppCrashSafeTest, LogRefRecoveryIgnoresCounters) { : offsetof(terark::DFA_MmapHeader, reserved) + 4 * sizeof(uint32_t); uint64_t wal[3]; // fileno, cnt, bytes const ssize_t n = ::pread(fd, wal, sizeof(wal), offset); - ::close(fd); ASSERT_EQ(n, static_cast(sizeof(wal))); + if (config[2] == 'S') { + // Simulate a crash before TLS statistics were flushed to the mapped header. + // A zero approximate count must not discard the real WAL references. + wal[1] = 0; + wal[2] = 0; + ASSERT_EQ(::pwrite(fd, wal, sizeof(wal), offset), + static_cast(sizeof(wal))); + } + ::close(fd); ASSERT_NE(wal[0], 0U); - ASSERT_EQ(wal[1], 0U); - ASSERT_EQ(wal[2], 0U); - uint64_t wal_size = 0; - ASSERT_OK(env_->GetFileSize(LogFileName(options.wal_dir, wal[0]), &wal_size)); for (int reopen = 0; reopen < 2; ++reopen) { ASSERT_OK(TryReopen(options)); ASSERT_GT(CountL0(db_), 0); ColumnFamilyMetaData cf_meta; db_->GetColumnFamilyMetaData(&cf_meta); ASSERT_EQ(cf_meta.blob_files.size(), 1U); - ASSERT_EQ(cf_meta.blob_files[0].total_blob_count, 1U); - ASSERT_EQ(cf_meta.blob_files[0].total_blob_bytes, wal_size); + ASSERT_EQ(cf_meta.blob_files[0].total_blob_count, + std::max(wal[1], 1)); + ASSERT_EQ(cf_meta.blob_files[0].total_blob_bytes, + std::max(wal[2], 1)); ASSERT_EQ(Get("0"), std::string(128, 'v')); ASSERT_EQ(Get("1"), std::string(128, 'v')); ASSERT_EQ(Get("inline"), "v"); @@ -907,6 +1819,74 @@ TEST_F(DBCsppCrashSafeTest, LogRefRecoveryIgnoresCounters) { } } +TEST_F(CrashChild, DISABLED_LogRefRecoveryMultipleWals) { + Options options = LogRefCrashOptions(dbname_, arg_); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + Options auxiliary = options; + ColumnFamilyHandle* handle = nullptr; + ASSERT_OK(child_db->CreateColumnFamily(auxiliary, "rotate", &handle)); + auto* impl = static_cast(child_db); + auto* cfd = impl->GetVersionSet()->GetColumnFamilySet()->GetColumnFamily( + handle->GetID()); + for (int i = 0; i < 3; ++i) { + ASSERT_OK(child_db->Put(WriteOptions(), std::to_string(i), + std::string(128, 'a' + i))); + if (i != 2) { + ASSERT_OK(impl->TEST_SwitchMemtable(cfd)); + } + } + // Only the auxiliary CF switches: all three WAL slots belong to one primary + // memtable, exercising the parallel mapping array rather than three memtables. + ::_exit(42); +} + +TEST_F(DBCsppCrashSafeTest, LogRefRecoveryMultipleWals) { + for (const char* config : {"CPS", "CSS", "OPS", "OSS"}) { + SCOPED_TRACE(config); + Close(); + Options options = LogRefCrashOptions(dbname_, config); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "LogRefRecoveryMultipleWals", config), 42); + const auto leftovers = ListLeftovers(options, dbname_); + // The empty rotating CF is now FileMmap too; its registered sources follow + // the primary CF's earlier file number in the complete inventory. + ASSERT_GE(leftovers.size(), 2U); + const int fd = ::open(leftovers.front().c_str(), O_RDONLY); + ASSERT_GE(fd, 0); + uint32_t num_wals = 0; + const size_t num_wals_offset = config[0] == 'O' + ? offsetof(terark::OSL_MmapHeader, reserved) + sizeof(uint32_t) + : offsetof(terark::DFA_MmapHeader, reserved) + 2 * sizeof(uint32_t); + ASSERT_EQ(::pread(fd, &num_wals, sizeof(num_wals), + num_wals_offset), + static_cast(sizeof(num_wals))); + ::close(fd); + ASSERT_EQ(num_wals, 3U); + for (int reopen = 0; reopen < 2; ++reopen) { + ASSERT_OK(TryReopenWithColumnFamilies({"default", "rotate"}, options)); + ASSERT_EQ(CountL0(db_), 1); + ColumnFamilyMetaData meta; + db_->GetColumnFamilyMetaData(&meta); + ASSERT_EQ(meta.blob_files.size(), 3U); + std::unique_ptr it(db_->NewIterator(ReadOptions())); + it->SeekToFirst(); + for (int i = 0; i < 3; ++i) { + const std::string value(128, 'a' + i); + ASSERT_EQ(Get(0, std::to_string(i)), value); + ASSERT_TRUE(it->Valid()); + ASSERT_EQ(it->key().ToString(), std::to_string(i)); + ASSERT_EQ(it->value().ToString(), value); + it->Next(); + } + ASSERT_FALSE(it->Valid()); + ASSERT_OK(it->status()); + it.reset(); + Close(); + } + } +} + TEST_F(CrashChild, DISABLED_ChangedFactoryWithOtherCfLeftover) { Options options = BaseCrashSafeOptions(dbname_, true, false); Options other = options; @@ -947,9 +1927,10 @@ TEST_F(CrashChild, DISABLED_OslLeftoverWithWriterLock) { ASSERT_OK(DB::Open(options, dbname_, &child_db)); const std::string key = "locked-osl-key"; ASSERT_OK(child_db->Put(WriteOptions(), key, "value")); - const auto leftovers = ListLeftovers(options, dbname_); - ASSERT_EQ(leftovers.size(), 1U); - int fd = ::open(leftovers[0].c_str(), O_RDWR); + auto* cfd = static_cast(child_db)->GetVersionSet() + ->GetColumnFamilySet()->GetDefault(); + const auto active = MakeTableFileName(dbname_, cfd->mem()->GetFileNumber()); + int fd = ::open(active.c_str(), O_RDWR); ASSERT_GE(fd, 0); char data[4096]; const ssize_t n = ::pread(fd, data, sizeof(data), 0); @@ -1070,32 +2051,60 @@ TEST_F(DBCsppCrashSafeTest, CloseConvertsLeftovers) { Destroy(options); ASSERT_EQ(RunCrashChild(dbname_, "CloseConvertsLeftovers", std::to_string(osl) + std::to_string(atomic)), 0); - ASSERT_TRUE(ListLeftovers(options, dbname_).empty()); + ASSERT_EQ(ListLeftovers(options, dbname_).size(), 3U); + for (const auto& path : ListLeftovers(options, dbname_)) + ASSERT_OK(env_->FileExists(path)); ASSERT_OK(env_->FileExists(CrashSafePubSeqFileName(dbname_))); ASSERT_OK(TryReopen(options)); ASSERT_EQ(Get("k1"), "v1"); ASSERT_EQ(Get("k2"), "v2"); ASSERT_GE(CountL0(db_), 1); Close(); - ASSERT_TRUE(ListLeftovers(options, dbname_).empty()); + ASSERT_EQ(ListLeftovers(options, dbname_).size(), 2U); + for (const auto& path : ListLeftovers(options, dbname_)) + ASSERT_OK(env_->FileExists(path)); } } } #endif -TEST_F(DBCsppCrashSafeTest, AvoidFlushDuringShutdownLeavesNoLeftoverThenWal) { - Close(); - Options options = BaseCrashSafeOptions(dbname_, true, false); - options.avoid_flush_during_shutdown = true; - Destroy(options); - ASSERT_OK(TryReopen(options)); - ASSERT_OK(Put("k", "v")); +TEST_F(DBCsppCrashSafeTest, AvoidFlushDuringShutdownKeepsRegisteredMemTable) { Close(); - ASSERT_TRUE(ListLeftovers(options, dbname_).empty()); - ASSERT_OK(env_->FileExists(CrashSafePubSeqFileName(dbname_))); - options.avoid_flush_during_shutdown = false; - ASSERT_OK(TryReopen(options)); - ASSERT_EQ(Get("k"), "v"); + for (bool osl : {false, true}) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + options.avoid_flush_during_shutdown = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "v")); + const auto registered = ListLeftovers(options, dbname_); + ASSERT_EQ(registered.size(), 2U); + auto* cfd = dbfull()->GetVersionSet()->GetColumnFamilySet()->GetDefault(); + for (auto* mem : {cfd->mem(), cfd->PeekPrecreatedMemtable()}) { + ASSERT_NE(mem, nullptr); + ASSERT_TRUE(mem->IsFileRegistered()); + const auto path = MakeTableFileName(dbname_, mem->GetFileNumber()); + ASSERT_EQ(std::count(registered.begin(), registered.end(), path), 1); + } + Close(); + ASSERT_EQ(ListLeftovers(options, dbname_), registered); + for (const auto& path : registered) ASSERT_OK(env_->FileExists(path)); + ASSERT_OK(env_->FileExists(CrashSafePubSeqFileName(dbname_))); + options.avoid_flush_during_shutdown = false; + std::atomic converted{0}; + SyncPoint::GetInstance()->SetCallBack( + "MemTableRep::ConvertToSST:After", [&](void*) { ++converted; }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(options)); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_EQ(converted.load(), 1); + ASSERT_EQ(Get("k"), "v"); + ASSERT_EQ(CountL0(db_), 1); + Close(); + for (const auto& path : ListLeftovers(options, dbname_)) + ASSERT_OK(env_->FileExists(path)); + } } TEST_F(DBCsppCrashSafeTest, AvoidFlushCloseReopenDoesNotProbeWal) { @@ -1138,6 +2147,12 @@ TEST_F(DBCsppCrashSafeTest, FreshSidecarProbesOnlyOlderWal) { ASSERT_EQ(probed.size(), 1U); ASSERT_OK(Put("b", "2")); Close(); + // A normal close retains registered sources, enabling prefix conversion. + // Force full WAL replay here to exercise mixed old/new WAL format probing: + // only WALs older than the fresh sidecar's kind boundary need inspection. + const auto registered = ListLeftovers(on, dbname_); + ASSERT_FALSE(registered.empty()); + ASSERT_OK(env_->DeleteFile(registered.front())); probed.clear(); ASSERT_OK(TryReopen(on)); SyncPoint::GetInstance()->DisableProcessing(); @@ -1181,7 +2196,7 @@ TEST_F(DBCsppCrashSafeTest, ClassicWalToLogIndexWithoutSidecarIsNotSupported) { ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); ASSERT_EQ(rec.wal_offset_kind, 2U); ASSERT_EQ(rec.kind_since_wal, 0U); - ASSERT_EQ(rec.generation & 1, 1U); + ASSERT_EQ(rec.generation & 1, 0U); } // Failed Open must not force KindPrep with the unestablished log-index @@ -1227,17 +2242,34 @@ TEST_F(DBCsppCrashSafeTest, LogIndexWalWithoutSidecarRecoversAsClassic) { TEST_F(DBCsppCrashSafeTest, RecoverOffDeletesStaleSidecar) { Close(); - Options on = BaseCrashSafeOptions(dbname_, true, false); - on.avoid_flush_during_shutdown = true; - Destroy(on); - ASSERT_OK(TryReopen(on)); - ASSERT_OK(Put("k", "v")); - Close(); - ASSERT_OK(env_->FileExists(CrashSafePubSeqFileName(dbname_))); - Options off = BaseCrashSafeOptions(dbname_, false, false); - ASSERT_OK(TryReopen(off)); - ASSERT_TRUE(env_->FileExists(CrashSafePubSeqFileName(dbname_)).IsNotFound()); - ASSERT_EQ(Get("k"), "v"); + for (bool osl : {false, true}) { + SCOPED_TRACE(osl); + Options on = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&on, true); + on.avoid_flush_during_shutdown = true; + Destroy(on); + ASSERT_OK(TryReopen(on)); + ASSERT_OK(Put("k", "v")); + Close(); + ASSERT_OK(env_->FileExists(CrashSafePubSeqFileName(dbname_))); + ASSERT_FALSE(ListLeftovers(on, dbname_).empty()); + Options off = on; + off.memtable_crash_safe_recover = false; + ASSERT_OK(TryReopen(off)); + ASSERT_TRUE(env_->FileExists(CrashSafePubSeqFileName(dbname_)).IsNotFound()); + for (auto* cfd : *dbfull()->GetVersionSet()->GetColumnFamilySet()) { + ASSERT_TRUE(cfd->GetMemTableFiles().empty()); + } + ASSERT_TRUE(ListLeftovers(off, dbname_).empty()); + ASSERT_EQ(Get("k"), "v"); + Close(); + ASSERT_OK(TryReopen(off)); + for (auto* cfd : *dbfull()->GetVersionSet()->GetColumnFamilySet()) { + ASSERT_TRUE(cfd->GetMemTableFiles().empty()); + } + ASSERT_EQ(Get("k"), "v"); + Close(); + } } TEST_F(DBCsppCrashSafeTest, OddGenerationKindSwitchUsesKindPrep) { @@ -1359,7 +2391,9 @@ TEST_F(DBCsppCrashSafeTest, AtomicFlushCloseConverts) { ASSERT_OK(TryReopen(options)); ASSERT_OK(Put("k", "v")); Close(); - ASSERT_TRUE(ListLeftovers(options, dbname_).empty()); + ASSERT_EQ(ListLeftovers(options, dbname_).size(), 2U); + for (const auto& path : ListLeftovers(options, dbname_)) + ASSERT_OK(env_->FileExists(path)); ASSERT_OK(env_->FileExists(CrashSafePubSeqFileName(dbname_))); ASSERT_OK(TryReopen(options)); ASSERT_EQ(Get("k"), "v"); @@ -1784,9 +2818,10 @@ TEST_F(DBCsppCrashSafeTest, LogIndexHeaderHasWalRefWithoutRecover) { Destroy(options); ASSERT_OK(TryReopen(options)); ASSERT_OK(Put("k", std::string(64, 'v'))); - auto leftovers = ListLeftovers(options, dbname_); - ASSERT_FALSE(leftovers.empty()); - const int fd = ::open(leftovers[0].c_str(), O_RDONLY); + ASSERT_TRUE(ListLeftovers(options, dbname_).empty()); + const uint64_t number = dbfull()->GetVersionSet()->GetColumnFamilySet() + ->GetDefault()->mem()->GetFileNumber(); + const int fd = ::open(MakeTableFileName(dbname_, number).c_str(), O_RDONLY); ASSERT_GE(fd, 0); terark::DFA_MmapHeader hdr{}; ASSERT_EQ(::pread(fd, &hdr, sizeof(hdr), 0), @@ -2080,6 +3115,12 @@ TEST_F(CrashChild, DISABLED_LeftoverOnDbPathNotCfPaths0) { ASSERT_OK(DB::Open(options, dbname_, &child_db)); ASSERT_OK(child_db->Put(WriteOptions(), "pad", "p")); ASSERT_OK(child_db->Put(WriteOptions(), "x", "1")); + auto* cfd = static_cast(child_db)->GetVersionSet() + ->GetColumnFamilySet()->GetDefault(); + ASSERT_OK(WriteStringToFile( + options.env, TableFileName(cfd->ioptions()->cf_paths, + cfd->mem()->GetFileNumber(), 0), + dbname_ + "/active-file")); ::_exit(1); } #endif @@ -2095,18 +3136,27 @@ TEST_F(DBCsppCrashSafeTest, LeftoverOnDbPathNotCfPaths0) { ASSERT_OK(TryReopen(options)); ASSERT_OK(Put("k", "0")); Close(); - ASSERT_TRUE(ListLeftovers(options, dbname_).empty()); + ASSERT_EQ(ListLeftovers(options, dbname_).size(), 2U); + for (const auto& path : ListLeftovers(options, dbname_)) + ASSERT_OK(env_->FileExists(path)); #if !defined(OS_WIN) ASSERT_EQ(RunCrashChild(dbname_, "LeftoverOnDbPathNotCfPaths0"), 1); - auto leftovers_l0 = ListLeftovers(options, l0); - ASSERT_FALSE(leftovers_l0.empty()); + auto leftovers_l0 = ListLeftovers(options, dbname_); + ASSERT_EQ(leftovers_l0.size(), 2U); + std::string active; + ASSERT_OK(ReadFileToString(env_, dbname_ + "/active-file", &active)); + ASSERT_EQ(std::count(leftovers_l0.begin(), leftovers_l0.end(), active), 1); ASSERT_OK(TryReopen(options)); ASSERT_EQ(Get("k"), "0"); ASSERT_EQ(Get("pad"), "p"); ASSERT_EQ(Get("x"), "1"); - for (const auto& leftover_path : leftovers_l0) { - ASSERT_TRUE(env_->FileExists(leftover_path).IsNotFound()); - } + ASSERT_OK(env_->FileExists(active)); + std::vector files; + db_->GetLiveFilesMetaData(&files); + const auto matching = std::count_if(files.begin(), files.end(), [&](const auto& file) { + return MakeTableFileName(file.db_path, file.file_number) == active; + }); + ASSERT_EQ(matching, 1); #endif } @@ -2589,37 +3639,38 @@ TEST_F(DBCsppCrashSafeTest, LogIndexOnRecoverOffUsesWal) { ASSERT_EQ(Get("k"), "v"); } -TEST_F(DBCsppCrashSafeTest, ListLeftoversAdvancesFileNumber) { +TEST_F(DBCsppCrashSafeTest, ManifestRegistryIgnoresUnregisteredFiles) { for (bool osl : {false, true}) { SCOPED_TRACE(osl ? "OSL" : "CSPP"); Close(); - Options options = BaseCrashSafeOptions(dbname_, false, false); + Options options = BaseCrashSafeOptions(dbname_, true, false); if (osl) { SetupOsl(&options, true); } Destroy(options); ASSERT_OK(env_->CreateDirIfMissing(dbname_)); - const std::string prefix = dbname_ + (osl ? "/OffsetSkipList-" : "/cspp-"); - const std::string high = prefix + "000100.memtab-0"; - const std::string low = prefix + "000010.memtab-0"; + const std::string high = MakeTableFileName(dbname_, 100); + const std::string low = MakeTableFileName(dbname_, 10); ASSERT_OK(WriteStringToFile(env_, "", high)); ASSERT_OK(WriteStringToFile(env_, "", low)); - ASSERT_EQ(ListLeftovers(options, dbname_).size(), 2U); + ASSERT_TRUE(ListLeftovers(options, dbname_).empty()); // Remove both so collision checks cannot hide a stale counter. ASSERT_OK(env_->DeleteFile(high)); ASSERT_OK(env_->DeleteFile(low)); ASSERT_OK(TryReopen(options)); - ASSERT_EQ(ListLeftovers(options, dbname_), - std::vector{prefix + "000101.memtab-0"}); + ASSERT_EQ(ListLeftovers(options, dbname_).size(), 2U); + for (const auto& path : ListLeftovers(options, dbname_)) + ASSERT_OK(env_->FileExists(path)); Close(); - ASSERT_TRUE(ListLeftovers(options, dbname_).empty()); - // Empty and lower-number scans must not move the counter backwards. + const auto before = ListLeftovers(options, dbname_); + // The manifest remains authoritative when unrelated files appear. ASSERT_OK(WriteStringToFile(env_, "", low)); - ASSERT_EQ(ListLeftovers(options, dbname_).size(), 1U); + ASSERT_EQ(ListLeftovers(options, dbname_), before); ASSERT_OK(env_->DeleteFile(low)); ASSERT_OK(TryReopen(options)); - ASSERT_EQ(ListLeftovers(options, dbname_), - std::vector{prefix + "000102.memtab-0"}); + ASSERT_EQ(ListLeftovers(options, dbname_).size(), 2U); + for (const auto& path : ListLeftovers(options, dbname_)) + ASSERT_OK(env_->FileExists(path)); Close(); Destroy(options); } @@ -2632,15 +3683,19 @@ TEST_F(DBCsppCrashSafeTest, CreateMemTableRepDoesNotOverwriteLeftover) { ASSERT_OK(TryReopen(options)); ASSERT_OK(Put("keep", "me")); auto first = ListLeftovers(options, dbname_); - ASSERT_EQ(first.size(), 1U); + ASSERT_EQ(first.size(), 2U); + auto* cfd = dbfull()->GetVersionSet()->GetColumnFamilySet()->GetDefault(); + const auto active = MakeTableFileName(dbname_, cfd->mem()->GetFileNumber()); + ASSERT_EQ(std::count(first.begin(), first.end(), active), 1); std::string before; - ASSERT_OK(ReadFileToString(env_, first[0], &before)); + ASSERT_OK(ReadFileToString(env_, active, &before)); ASSERT_OK(dbfull()->TEST_SwitchMemtable()); ASSERT_OK(Put("other", "x")); auto after_list = ListLeftovers(options, dbname_); - ASSERT_GE(after_list.size(), 2U); + ASSERT_EQ(after_list, first); + ASSERT_NE(MakeTableFileName(dbname_, cfd->mem()->GetFileNumber()), active); std::string after; - ASSERT_OK(ReadFileToString(env_, first[0], &after)); + ASSERT_OK(ReadFileToString(env_, active, &after)); ASSERT_EQ(before, after); } @@ -2653,7 +3708,9 @@ TEST_F(DBCsppCrashSafeTest, Allow2pcAloneStillConverts) { ASSERT_OK(Put("k", "v")); ASSERT_OK(dbfull()->TEST_SwitchMemtable()); Close(); - ASSERT_TRUE(ListLeftovers(options, dbname_).empty()); + ASSERT_EQ(ListLeftovers(options, dbname_).size(), 2U); + for (const auto& path : ListLeftovers(options, dbname_)) + ASSERT_OK(env_->FileExists(path)); ASSERT_OK(TryReopen(options)); ASSERT_EQ(Get("k"), "v"); } @@ -2720,21 +3777,29 @@ TEST_F(CrashChild, DISABLED_LeftoverNoMagicFallsBackToWal) { TEST_F(DBCsppCrashSafeTest, LeftoverNoMagicFallsBackToWal) { Close(); Options options = BaseCrashSafeOptions(dbname_, true, false); - Destroy(options); - ASSERT_EQ(RunCrashChild(dbname_, "LeftoverNoMagicFallsBackToWal"), 1); - auto leftovers = ListLeftovers(options, dbname_); - ASSERT_FALSE(leftovers.empty()); - const int fd = ::open(leftovers[0].c_str(), O_RDWR); - ASSERT_GE(fd, 0); - terark::DFA_MmapHeader hdr{}; - ASSERT_EQ(::pread(fd, &hdr, sizeof(hdr), 0), - static_cast(sizeof(hdr))); - std::memset(hdr.reserved, 0, sizeof(hdr.reserved)); - ASSERT_EQ(::pwrite(fd, &hdr, sizeof(hdr), 0), - static_cast(sizeof(hdr))); - ::close(fd); - ASSERT_OK(TryReopen(options)); - ASSERT_EQ(Get("bad"), "hdr"); + for (const std::string damage : {"empty", "short-header", "missing-magic"}) { + SCOPED_TRACE(damage); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "LeftoverNoMagicFallsBackToWal"), 1); + auto leftovers = ListLeftovers(options, dbname_); + ASSERT_FALSE(leftovers.empty()); + const int fd = ::open(leftovers[0].c_str(), O_RDWR); + ASSERT_GE(fd, 0); + terark::DFA_MmapHeader hdr{}; + if (damage == "missing-magic") { + ASSERT_EQ(::pread(fd, &hdr, sizeof(hdr), 0), + static_cast(sizeof(hdr))); + std::memset(hdr.reserved, 0, sizeof(hdr.reserved)); + ASSERT_EQ(::pwrite(fd, &hdr, sizeof(hdr), 0), + static_cast(sizeof(hdr))); + } else { + ASSERT_EQ(::ftruncate(fd, damage == "empty" ? 0 : sizeof(hdr) - 1), 0); + } + ::close(fd); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("bad"), "hdr"); + Close(); + } } TEST_F(CrashChild, DISABLED_DualLeftoverSecondConvertFails) { @@ -2795,79 +3860,21 @@ TEST_F(DBCsppCrashSafeTest, TruncateInjectFailureFallsBackToWal) { ASSERT_EQ(RunCrashChild(dbname_, "TruncateInjectFailureFallsBackToWal"), 1); SyncPoint::GetInstance()->EnableProcessing(); SyncPoint::GetInstance()->ClearAllCallBacks(); + bool truncate_called = false; SyncPoint::GetInstance()->SetCallBack( - "CrashSafeRecover::Truncate:InjectStatus", [](void* arg) { - *static_cast(arg) = Status::IOError("inject truncate"); + "MemTableRep::ConvertToSST:Truncate", [&truncate_called](void* arg) { + truncate_called = true; + *static_cast(arg) = IOStatus::IOError("inject truncate"); }); ASSERT_OK(TryReopen(options)); + ASSERT_TRUE(truncate_called); ASSERT_EQ(Get("tr"), "ok"); SyncPoint::GetInstance()->DisableProcessing(); SyncPoint::GetInstance()->ClearAllCallBacks(); } -TEST_F(CrashChild, DISABLED_LinkFileInjectFailureFallsBackToWal) { - Options options = BaseCrashSafeOptions(dbname_, true, true); - SyncPoint::GetInstance()->SetCallBack( - "DBImpl::PersistPublishedSequence:AfterCommit", - [](void*) { ::_exit(1); }); - SyncPoint::GetInstance()->EnableProcessing(); - DB* child_db = nullptr; - ASSERT_OK(DB::Open(options, dbname_, &child_db)); - ASSERT_OK(child_db->Put(WriteOptions(), "lk", "ok")); - ::_exit(0); -} - -TEST_F(DBCsppCrashSafeTest, LinkFileInjectFailureFallsBackToWal) { - Close(); - Options options = BaseCrashSafeOptions(dbname_, true, true); - Destroy(options); - ASSERT_EQ(RunCrashChild(dbname_, "LinkFileInjectFailureFallsBackToWal"), 1); - SyncPoint::GetInstance()->EnableProcessing(); - SyncPoint::GetInstance()->ClearAllCallBacks(); - SyncPoint::GetInstance()->SetCallBack( - "CrashSafeRecover::LinkFile:InjectStatus", [](void* arg) { - *static_cast(arg) = Status::IOError("inject link"); - }); - ASSERT_OK(TryReopen(options)); - ASSERT_EQ(Get("lk"), "ok"); - SyncPoint::GetInstance()->DisableProcessing(); - SyncPoint::GetInstance()->ClearAllCallBacks(); -} - -TEST_F(CrashChild, DISABLED_AfterRenameBeforeAddFileFallsBackToWal) { - Options options = BaseCrashSafeOptions(dbname_, true, false); - SyncPoint::GetInstance()->SetCallBack( - "DBImpl::PersistPublishedSequence:AfterCommit", - [](void*) { ::_exit(1); }); - SyncPoint::GetInstance()->EnableProcessing(); - DB* child_db = nullptr; - ASSERT_OK(DB::Open(options, dbname_, &child_db)); - ASSERT_OK(child_db->Put(WriteOptions(), "or", "phan")); - ::_exit(0); -} -TEST_F(CrashChild, DISABLED_AfterRenameBeforeAddFileFallsBackToWalRecover) { - Options options = BaseCrashSafeOptions(dbname_, true, false); - SyncPoint::GetInstance()->SetCallBack( - "CrashSafeRecover::AfterRenameBeforeAddFile", - [](void*) { ::_exit(1); }); - SyncPoint::GetInstance()->EnableProcessing(); - DB* recover_db = nullptr; - DB::Open(options, dbname_, &recover_db); - ::_exit(0); -} -TEST_F(DBCsppCrashSafeTest, AfterRenameBeforeAddFileFallsBackToWal) { - Close(); - Options options = BaseCrashSafeOptions(dbname_, true, false); - Destroy(options); - ASSERT_EQ(RunCrashChild(dbname_, "AfterRenameBeforeAddFileFallsBackToWal"), 1); - SyncPoint::GetInstance()->EnableProcessing(); - SyncPoint::GetInstance()->ClearAllCallBacks(); - ASSERT_EQ(RunCrashChild(dbname_, "AfterRenameBeforeAddFileFallsBackToWalRecover"), 1); - ASSERT_OK(TryReopen(options)); - ASSERT_EQ(Get("or"), "phan"); -} TEST_F(DBCsppCrashSafeTest, CrashSafeOrLogIndexDisablesWalCompression) { for (bool crash_safe : {false, true}) { @@ -3133,7 +4140,7 @@ TEST_F(DBCsppCrashSafeTest, WritePreparedFallsBackToWal) { delete txn_db; } -TEST_F(DBCsppCrashSafeTest, AfterRenameCloseSecondFlushInject) { +TEST_F(DBCsppCrashSafeTest, AfterConvertCloseSecondFlushInject) { Close(); Options options = BaseCrashSafeOptions(dbname_, true, false); Destroy(options); @@ -3235,6 +4242,8 @@ TEST_F(DBCsppCrashSafeTest, AtomicFlushDualCfLeftoverConvertsBoth) { ASSERT_GE(CountL0(db_, "one"), 1); } + + TEST_F(CrashChild, DISABLED_DroppedCfLeftoverSkipped) { Options options = BaseCrashSafeOptions(dbname_, true, false); std::atomic pubs{0}; @@ -3268,13 +4277,10 @@ TEST_F(DBCsppCrashSafeTest, DroppedCfLeftoverSkipped) { ASSERT_EQ(Get(0, "keep"), "1"); ASSERT_EQ(Get(1, "drop"), "2"); ASSERT_OK(Flush(0)); - std::string left1; - for (const auto& p : ListLeftovers(options, dbname_)) { - if (p.find(".memtab-1") != std::string::npos) { - left1 = p; - break; - } - } + const auto registered = dbfull()->GetVersionSet()->GetColumnFamilySet() + ->GetColumnFamily(1)->GetMemTableFiles(); + ASSERT_FALSE(registered.empty()); + const std::string left1 = MakeTableFileName(dbname_, *registered.begin()); ASSERT_FALSE(left1.empty()); const std::string bak = left1 + ".bak"; CopyFile(left1, bak); @@ -3287,13 +4293,10 @@ TEST_F(DBCsppCrashSafeTest, DroppedCfLeftoverSkipped) { } ASSERT_OK(TryReopen(options)); ASSERT_EQ(Get("keep"), "1"); - bool dropped_left = false; - for (const auto& p : ListLeftovers(options, dbname_)) { - if (p.find(".memtab-1") != std::string::npos) { - dropped_left = true; - } - } - ASSERT_TRUE(dropped_left); + ASSERT_EQ(dbfull()->GetVersionSet()->GetColumnFamilySet() + ->GetColumnFamily(1), nullptr); + const auto leftovers = ListLeftovers(options, dbname_); + ASSERT_EQ(std::count(leftovers.begin(), leftovers.end(), left1), 0); } TEST_F(CrashChild, DISABLED_MultiChunkAfterCommitStillReadable) { @@ -3493,16 +4496,19 @@ TEST_F(DBCsppCrashSafeTest, InFlightFlushThenClose) { ASSERT_OK(dbfull()->TEST_SwitchMemtable()); ASSERT_OK(Put("tail", "2")); test::SleepingBackgroundTask sleeping_task; - env_->Schedule(&test::SleepingBackgroundTask::DoSleepTask, &sleeping_task, - Env::Priority::HIGH); - sleeping_task.WaitUntilSleeping(); + SyncPoint::GetInstance()->SetCallBack( + "FlushJob::WriteLevel0Table", [&](void*) { sleeping_task.DoSleep(); }); + SyncPoint::GetInstance()->EnableProcessing(); FlushOptions fo; fo.wait = false; - ASSERT_OK(db_->Flush(fo)); + const Status flush_status = db_->Flush(fo); + const bool paused = !sleeping_task.TimedWaitUntilSleeping(10 * 1000000); + if (!flush_status.ok() || !paused) sleeping_task.WakeUp(); + ASSERT_OK(flush_status); + ASSERT_TRUE(paused); SyncPoint::GetInstance()->SetCallBack( "DBImpl::FlushMemTable:AfterScheduleFlush", [&](void*) { sleeping_task.WakeUp(); }); - SyncPoint::GetInstance()->EnableProcessing(); std::thread closer([&] { Close(); }); sleeping_task.WaitUntilDone(); closer.join(); diff --git a/db/db_impl/db_impl.cc b/db/db_impl/db_impl.cc index 4b139aab1..53e0c0e52 100644 --- a/db/db_impl/db_impl.cc +++ b/db/db_impl/db_impl.cc @@ -4075,6 +4075,10 @@ Status DBImpl::CreateColumnFamilyImpl(const ColumnFamilyOptions& cf_options, // LogAndApply will both write the creation in MANIFEST and create // ColumnFamilyData object + auto pending_memtable = pending_outputs_.end(); + if (cf_options.memtable_factory->SupportCrashSafe()) { + pending_memtable = CaptureCurrentFileNumberInPendingOutputs(); + } { // write thread WriteThread::Writer w; write_thread_.EnterUnbatched(&w, &mutex_); @@ -4091,6 +4095,22 @@ Status DBImpl::CreateColumnFamilyImpl(const ColumnFamilyOptions& cf_options, assert(cfd != nullptr); std::map> dummy_created_dirs; s = cfd->AddDirectories(&dummy_created_dirs); + if (s.ok() && immutable_db_options_.memtable_crash_safe_recover) { + cfd->PrepareNewMemtableInBackground(*cfd->GetLatestMutableCFOptions()); + s = RegisterMemTableFiles(cfd); + if (!s.ok()) { + // CF creation is already committed. Keep its in-memory state valid, + // but do not return a writable handle after registration fails. + InstallSuperVersionAndScheduleWork(cfd, &sv_context, + *cfd->GetLatestMutableCFOptions()); + cfd->set_initialized(); + error_handler_.SetBGError(s, BackgroundErrorReason::kManifestWrite) + .PermitUncheckedError(); + } + } + } + if (pending_memtable != pending_outputs_.end()) { + pending_outputs_.erase(pending_memtable); } if (s.ok()) { auto* cfd = diff --git a/db/db_impl/db_impl.h b/db/db_impl/db_impl.h index c62f559ee..ff8531ba7 100644 --- a/db/db_impl/db_impl.h +++ b/db/db_impl/db_impl.h @@ -1437,8 +1437,6 @@ class DBImpl : public DB { // such a file's absolute path to its parent directory. std::unordered_map files_to_delete_; bool is_new_db_ = false; - // Restore the saved WAL cursor only after recovery edits are installed. - bool restore_published_seq_ = false; // WAL tail cursor for this Recover. RecoverLogFiles reads it from here. bool crash_safe_wal_tail_replay_ = false; uint8_t crash_safe_wal_offset_kind_ = 0; @@ -1966,12 +1964,10 @@ class DBImpl : public DB { void StagePublishedWal(SequenceNumber seq, uint64_t wal_number, uint64_t wal_offset); void AccountPendingMemtableWrites(size_t n); + Status RegisterMemTableFiles(ColumnFamilyData* cfd); bool CanConvertLeftoverForCrashSafeRecover( - const std::vector& leftover_snapshot, - SequenceNumber mmap_pubseq, uint64_t mmap_wal_number, - std::string* fail_reason); + SequenceNumber mmap_pubseq, std::string* fail_reason); Status ConvertLeftoverMemtables( - const std::vector& leftover_snapshot, SequenceNumber max_visible_seq, RecoveryContext* recovery_ctx); // The following two methods are used to flush a memtable to diff --git a/db/db_impl/db_impl_files.cc b/db/db_impl/db_impl_files.cc index 47901f4d2..e3659ced5 100644 --- a/db/db_impl/db_impl_files.cc +++ b/db/db_impl/db_impl_files.cc @@ -175,6 +175,14 @@ void DBImpl::FindObsoleteFiles(JobContext* job_context, bool force, if (doing_the_full_scan) { versions_->AddLiveFiles(&job_context->sst_live, &job_context->blob_live); + // These share the SST namespace, but are still mutable. Protect them from + // GC without exposing them as immutable SSTs to checkpoints and backups. + for (auto* cfd : *versions_->GetColumnFamilySet()) { + const auto& files = cfd->GetMemTableFiles(); + job_context->sst_live.insert(job_context->sst_live.end(), + files.begin(), files.end()); + cfd->AddMemTableFileNumbers(&job_context->sst_live); + } InfoLogPrefix info_log_prefix(!immutable_db_options_.db_log_dir.empty(), dbname_); std::set paths; diff --git a/db/db_impl/db_impl_open.cc b/db/db_impl/db_impl_open.cc index 8beab8bbd..edaecd6bd 100644 --- a/db/db_impl/db_impl_open.cc +++ b/db/db_impl/db_impl_open.cc @@ -543,20 +543,6 @@ bool PeekPublishedWalKind(const std::string& dbname, uint64_t* kind) { return ok; } -uint32_t ParseLeftoverCfId(const std::string& path) { - const auto pos = path.rfind(".memtab-"); - if (pos == std::string::npos) { - return std::numeric_limits::max(); - } - const char* p = path.c_str() + pos + 8; - char* end = nullptr; - const unsigned long id = std::strtoul(p, &end, 10); - if (end == p) { - return std::numeric_limits::max(); - } - return static_cast(id); -} - void DestroyKindPrepDb(DB* db, std::vector* handles) { for (auto* h : *handles) { delete h; @@ -820,15 +806,33 @@ void DBImpl::AccountPendingMemtableWrites(size_t n) { } } +Status DBImpl::RegisterMemTableFiles(ColumnFamilyData* cfd) { + mutex_.AssertHeld(); + VersionEdit edit; + cfd->AddMemTableFileEdits(&edit); + edit.SetColumnFamily(cfd->GetID()); + TEST_SYNC_POINT("DBImpl::RegisterMemTableFile:BeforeLogAndApply"); + Status s = versions_->LogAndApply(cfd, *cfd->GetLatestMutableCFOptions(), + ReadOptions(), &edit, &mutex_, + directories_.GetDbDir()); + TEST_SYNC_POINT_CALLBACK("DBImpl::RegisterMemTableFile:AfterLogAndApply", &s); + if (s.ok()) { + cfd->PublishRegisteredMemTables(); + TEST_SYNC_POINT("DBImpl::RegisterMemTableFile:BeforeInstall"); + } + return s; +} + bool DBImpl::CanConvertLeftoverForCrashSafeRecover( - const std::vector& leftover_snapshot, - SequenceNumber mmap_pubseq, uint64_t mmap_wal_number, - std::string* fail_reason) { + SequenceNumber mmap_pubseq, std::string* fail_reason) { mutex_.AssertHeld(); auto fail = [&](const std::string& reason) { *fail_reason = reason; return false; }; + if (!versions_->HasMemTableFileTracking()) { + return fail("MANIFEST predates MemTable file tracking"); + } if (immutable_db_options_.wal_filter != nullptr) { return fail("wal_filter requires full WAL replay"); } @@ -838,87 +842,10 @@ bool DBImpl::CanConvertLeftoverForCrashSafeRecover( if (immutable_db_options_.best_efforts_recovery) { return fail("best_efforts_recovery"); } - for (auto* cfd : *versions_->GetColumnFamilySet()) { - if (cfd->IsDropped()) { - continue; - } - auto* fac = cfd->ioptions()->memtable_factory.get(); - if (fac == nullptr || !fac->SupportCrashSafe()) { - return fail("CF " + cfd->GetName() + " factory !SupportCrashSafe"); - } - if (cfd->mem() != nullptr && !cfd->mem()->SupportConvertToSST()) { - return fail("CF " + cfd->GetName() + " mem !SupportConvertToSST"); - } - if (!cfd->imm()->UnflushedMemtablesSupportConvertToSST()) { - return fail("CF " + cfd->GetName() + " imm !SupportConvertToSST"); - } - } - - // Other tiers may be slow and contain many files; memtables use cf_paths[0]. - std::vector paths; - for (const auto* cfd : *versions_->GetColumnFamilySet()) { - if (!cfd->IsDropped()) { - paths.push_back( - NormalizePath(cfd->ioptions()->cf_paths[0].path + - std::string(1, kFilePathSeparator))); - } - } - std::sort(paths.begin(), paths.end()); - paths.erase(std::unique(paths.begin(), paths.end()), paths.end()); - const uint64_t next_file_number = versions_->current_next_file_number(); - for (const auto& path : paths) { - std::vector files; - const Status ls = env_->GetChildren(path, &files); - if (!ls.ok()) { - continue; - } - for (const auto& fname : files) { - uint64_t number = 0; - FileType type; - if (!ParseFileName(fname, &number, &type)) { - continue; - } - // Only numbers still ahead of MANIFEST next_file (crash mid-Open - // before LogAndApply). A Convert that failed after this Open already - // advanced next_file is invisible here; ConvertLeftover deletes those. - if (type == kTableFile && number >= next_file_number) { - return fail("orphan SST " + path + fname); - } - } - } - - std::unordered_set leftover_cfs; - for (const auto& leftover_path : leftover_snapshot) { - const uint32_t cf_id = ParseLeftoverCfId(leftover_path); - ColumnFamilyData* cfd = - versions_->GetColumnFamilySet()->GetColumnFamily(cf_id); - if (cfd == nullptr) { - ROCKS_LOG_WARN(immutable_db_options_.info_log, - "Crash-safe leftover %s belongs to dropped CF %u, skip", - leftover_path.c_str(), cf_id); - continue; - } - auto* fac = cfd->ioptions()->memtable_factory.get(); - const Status ps = - fac->ProbeCrashSafeLeftover(leftover_path, - immutable_db_options_.GetWalDir()); - if (!ps.ok()) { - return fail("probe " + leftover_path + ": " + ps.ToString()); - } - leftover_cfs.insert(cf_id); - } - if (mmap_pubseq > versions_->LastSequence() && leftover_cfs.empty()) { - return fail("CSPUBSEQ > MANIFEST LastSequence and no leftover"); - } - if (!leftover_cfs.empty() && mmap_pubseq == 0) { - return fail("CSPUBSEQ pubseq 0 with leftover present"); - } - if (mmap_wal_number != 0) { - for (const auto* cfd : *versions_->GetColumnFamilySet()) { - // Only CFs persisted past this WAL can omit their leftover safely. - if (!cfd->IsDropped() && cfd->GetLogNumber() <= mmap_wal_number && - leftover_cfs.count(cfd->GetID()) == 0) { - return fail("CF " + cfd->GetName() + " has no leftover before WAL cursor"); + if (mmap_pubseq == 0) { + for (auto* cfd : *versions_->GetColumnFamilySet()) { + if (!cfd->GetMemTableFiles().empty()) { + return fail("CSPUBSEQ pubseq 0 with leftover present"); } } } @@ -926,32 +853,30 @@ bool DBImpl::CanConvertLeftoverForCrashSafeRecover( } Status DBImpl::ConvertLeftoverMemtables( - const std::vector& leftover_snapshot, SequenceNumber max_visible_seq, RecoveryContext* recovery_ctx) { mutex_.AssertHeld(); - if (leftover_snapshot.empty()) { - return Status::OK(); - } - struct ConvertedLeftover { ColumnFamilyData* cfd; FileMetaData meta; std::vector blobs; }; std::vector converted; + // MANIFEST is the complete inventory, including empty, precreated tables. + // A directory scan cannot detect one missing file among several in a CF. + for (auto* cfd : *versions_->GetColumnFamilySet()) { + for (uint64_t file_num : cfd->GetMemTableFiles()) { + auto& one = converted.emplace_back(); + one.cfd = cfd; + one.meta.fd = FileDescriptor(file_num, 0, 0); + } + } + if (converted.empty()) return Status::OK(); Status s; mutex_.Unlock(); - for (const auto& leftover_path : leftover_snapshot) { - const uint32_t cf_id = ParseLeftoverCfId(leftover_path); - ColumnFamilyData* cfd = - versions_->GetColumnFamilySet()->GetColumnFamily(cf_id); - if (cfd == nullptr) { - continue; - } - ConvertedLeftover one; - one.cfd = cfd; - const uint64_t file_num = versions_->NewFileNumber(); - one.meta.fd = FileDescriptor(file_num, 0, 0); + for (auto& one : converted) { + auto* cfd = one.cfd; + const uint64_t file_num = one.meta.fd.GetNumber(); + const auto leftover_path = TableFileName(cfd->ioptions()->cf_paths, file_num, 0); one.meta.fd.smallest_seqno = 0; one.meta.fd.largest_seqno = max_visible_seq; one.meta.epoch_number = cfd->NewEpochNumber(); @@ -968,31 +893,16 @@ Status DBImpl::ConvertLeftoverMemtables( tboptions.add_blob_file = [&one](BlobFileAddition b) { one.blobs.push_back(std::move(b)); }; - Status one_s = - cfd->ioptions()->memtable_factory->RecoverCrashSafeMemTableToSST( - leftover_path, &one.meta, tboptions); - if (!one_s.ok()) { - s = one_s; - // Inject / rename-then-fail still leaves an SST. Include it so the - // cleanup below can unlink it; leftover is already gone. - converted.push_back(std::move(one)); + s = cfd->ioptions()->memtable_factory->RecoverCrashSafeMemTableToSST( + leftover_path, &one.meta, tboptions); + if (!s.ok()) { break; } - converted.push_back(std::move(one)); } mutex_.Lock(); if (!s.ok()) { - // NewFileNumber already ran. After this Open finishes, next_file is - // past these numbers, so DeleteUnreferencedSstFiles (only >= next_file, - // and it runs before Convert) will not collect them. + // Do not retire registered sources until full WAL recovery commits. for (auto& one : converted) { - if (one.meta.fd.GetNumber() == 0) { - continue; - } - env_->DeleteFile(TableFileName(one.cfd->ioptions()->cf_paths, - one.meta.fd.GetNumber(), - one.meta.fd.GetPathId())) - .PermitUncheckedError(); for (const auto& blob : one.blobs) { env_->DeleteFile(BlobFileName(one.cfd->ioptions()->cf_paths.front().path, blob.GetBlobFileNumber())) @@ -1002,13 +912,13 @@ Status DBImpl::ConvertLeftoverMemtables( return s; } for (auto& one : converted) { - if (one.meta.fd.GetFileSize() == 0) { - continue; - } VersionEdit edit; edit.SetColumnFamily(one.cfd->GetID()); - one.meta.marked_for_compaction = true; - edit.AddFile(0, one.meta); + edit.DeleteMemTableFile(one.meta.fd.GetNumber()); + if (one.meta.fd.GetFileSize() != 0) { + one.meta.marked_for_compaction = true; + edit.AddFile(0, one.meta); + } for (const auto& blob : one.blobs) { edit.AddBlobFile(blob); } @@ -1031,26 +941,6 @@ Status DBImpl::Recover( } } - std::vector leftover_snapshot; - if (immutable_db_options_.memtable_crash_safe_recover && !read_only) { - for (const auto& desc : column_families) { - if (!desc.options.memtable_factory) { - continue; - } - const auto& paths = desc.options.cf_paths.empty() - ? immutable_db_options_.db_paths - : desc.options.cf_paths; - desc.options.memtable_factory->ListCrashSafeLeftovers( - paths[0].path, &leftover_snapshot); - } - if (!leftover_snapshot.empty()) { - std::sort(leftover_snapshot.begin(), leftover_snapshot.end()); - leftover_snapshot.erase( - std::unique(leftover_snapshot.begin(), leftover_snapshot.end()), - leftover_snapshot.end()); - } - } - bool tmp_is_new_db = false; bool& is_new_db = recovery_ctx ? recovery_ctx->is_new_db_ : tmp_is_new_db; assert(db_lock_ == nullptr); @@ -1297,7 +1187,6 @@ Status DBImpl::Recover( s = SetupDBId(read_only, recovery_ctx); ROCKS_LOG_INFO(immutable_db_options_.info_log, "DB ID: %s\n", db_id_.c_str()); bool crash_safe_convert = false; - bool recovered_wals_ok = false; PublishedSeqRecord mmap_rec; bool mmap_valid = false; if (s.ok() && !read_only && @@ -1306,7 +1195,7 @@ Status DBImpl::Recover( std::string fail_reason; if (mmap_valid) { crash_safe_convert = CanConvertLeftoverForCrashSafeRecover( - leftover_snapshot, mmap_rec.pubseq, mmap_rec.wal_number, &fail_reason); + mmap_rec.pubseq, &fail_reason); } else { fail_reason = "invalid CSPUBSEQ generation"; } @@ -1315,26 +1204,23 @@ Status DBImpl::Recover( "Crash-safe recover check failed (%s), fallback to full " "WAL RecoverLogFiles", fail_reason.c_str()); - for (const auto& leftover_path : leftover_snapshot) { - ROCKS_LOG_WARN(immutable_db_options_.info_log, - "Crash-safe leftover not converted: %s", - leftover_path.c_str()); - } } } if (s.ok() && !read_only) { - if (pubseq_mmap_ != nullptr) { - // Recovery can consume only part of the leftover set before failing. - // Keep the cursor invalid across a second crash, including one after - // orphan/failed-conversion cleanup has removed the failure evidence. - pubseq_mmap_->generation |= 1; - recovery_ctx->restore_published_seq_ = mmap_valid; - } s = DeleteUnreferencedSstFiles(recovery_ctx); } + if (s.ok()) { // CrashSafe needs to create an sst with next_file_number_. + // next_file_number_ must be restored before creating MemTables. + // WAL replay may contain sequences below LastSequence(). + for (auto* cfd : *versions_->GetColumnFamilySet()) { + if (cfd->mem() == nullptr) { + assert(cfd->ioptions()->memtable_factory->SupportCrashSafe()); + cfd->CreateNewMemtable(*cfd->GetLatestMutableCFOptions(), 0); + } + } + } if (s.ok() && crash_safe_convert) { - const Status cs = ConvertLeftoverMemtables(leftover_snapshot, mmap_rec.pubseq, - recovery_ctx); + const Status cs = ConvertLeftoverMemtables(mmap_rec.pubseq, recovery_ctx); if (!cs.ok()) { ROCKS_LOG_WARN(immutable_db_options_.info_log, "Crash-safe leftover Convert failed (%s), fallback to " @@ -1476,7 +1362,6 @@ Status DBImpl::Recover( bool corrupted_wal_found = false; s = RecoverLogFiles(wals, &next_sequence, read_only, &corrupted_wal_found, recovery_ctx); - recovered_wals_ok = s.ok(); if (corrupted_wal_found && recovered_seq != nullptr) { *recovered_seq = next_sequence; } @@ -1490,13 +1375,18 @@ Status DBImpl::Recover( } } - if (s.ok() && recovered_wals_ok && !crash_safe_convert && - !leftover_snapshot.empty()) { - for (const auto& leftover_path : leftover_snapshot) { - ROCKS_LOG_INFO(immutable_db_options_.info_log, - "Crash-safe leftover deleted after WAL fallback: %s", - leftover_path.c_str()); - env_->DeleteFile(leftover_path).PermitUncheckedError(); + if (s.ok() && !read_only && !crash_safe_convert) { + // Retire the old inventory only together with the successful WAL recovery. + // Missing files are allowed here: their absence is what forced the replay. + for (auto* cfd : *versions_->GetColumnFamilySet()) { + const auto& files = cfd->GetMemTableFiles(); + if (files.empty()) continue; + VersionEdit edit; + edit.SetColumnFamily(cfd->GetID()); + for (uint64_t number : files) { + edit.DeleteMemTableFile(number); + } + recovery_ctx->UpdateVersionEdits(cfd, edit); } } @@ -1662,9 +1552,38 @@ Status DBImpl::LogAndApplyForRecovery(const RecoveryContext& recovery_ctx) { mutex_.AssertHeld(); assert(versions_->descriptor_log_ == nullptr); const ReadOptions read_options(Env::IOActivity::kDBOpen); + // Recovery retires the old inventory and installs all recovered CFs at once. + // A truncated MANIFEST tail must not install just part of that transition. + uint32_t remaining = 0; + bool tracks_memtable_files = false; + for (const auto& edits : recovery_ctx.edit_lists_) { + remaining += static_cast(edits.size()); + for (const auto* edit : edits) { + tracks_memtable_files |= edit->HasMemTableFileTracking() || + !edit->GetMemTableFileAdditions().empty() || + !edit->GetMemTableFileDeletions().empty(); + } + } + if (tracks_memtable_files && remaining > 1) { + for (const auto& edits : recovery_ctx.edit_lists_) { + for (auto* edit : edits) { + edit->MarkAtomicGroup(--remaining); + } + } + } + TEST_SYNC_POINT("DBImpl::RegisterMemTableFile:BeforeLogAndApply"); Status s = versions_->LogAndApply( recovery_ctx.cfds_, recovery_ctx.mutable_cf_opts_, read_options, recovery_ctx.edit_lists_, &mutex_, directories_.GetDbDir()); + TEST_SYNC_POINT_CALLBACK("DBImpl::RegisterMemTableFile:AfterLogAndApply", &s); + if (s.ok()) { + if (immutable_db_options_.memtable_crash_safe_recover) { + for (auto* cfd : *versions_->GetColumnFamilySet()) { + cfd->PublishRegisteredMemTables(); + } + } + TEST_SYNC_POINT("DBImpl::RegisterMemTableFile:BeforeInstall"); + } if (s.ok() && !(recovery_ctx.files_to_delete_.empty())) { mutex_.Unlock(); for (const auto& stale_sst_file : recovery_ctx.files_to_delete_) { @@ -1678,9 +1597,6 @@ Status DBImpl::LogAndApplyForRecovery(const RecoveryContext& recovery_ctx) { } mutex_.Lock(); } - if (s.ok() && recovery_ctx.restore_published_seq_) { - pubseq_mmap_->generation += 1; - } if (s.ok() && pubseq_mmap_ != nullptr && (pubseq_mmap_->generation & 1)) { // Discard a torn cursor before enabling publication. pubseq_mmap_->rec = PublishedSeqRecord{}; @@ -2932,6 +2848,15 @@ Status DBImpl::Open(const DBOptions& db_options, const std::string& dbname, impl->pubseq_mmap_->kind_since_wal = impl->logfile_number_; } if (s.ok()) { + if (impl->immutable_db_options_.memtable_crash_safe_recover) { + for (auto* cfd : *impl->versions_->GetColumnFamilySet()) { + VersionEdit edit; + edit.SetColumnFamily(cfd->GetID()); + cfd->PrepareNewMemtableInBackground(*cfd->GetLatestMutableCFOptions()); + cfd->AddMemTableFileEdits(&edit); + recovery_ctx.UpdateVersionEdits(cfd, edit); + } + } s = impl->LogAndApplyForRecovery(recovery_ctx); } diff --git a/db/db_impl/db_impl_write.cc b/db/db_impl/db_impl_write.cc index 2211106a5..1675d3096 100644 --- a/db/db_impl/db_impl_write.cc +++ b/db/db_impl/db_impl_write.cc @@ -2322,6 +2322,22 @@ Status DBImpl::SwitchMemtable(ColumnFamilyData* cfd, WriteContext* context) { return s; } + auto pending_memtable = pending_outputs_.end(); + if (cfd->ioptions()->memtable_factory->SupportCrashSafe()) { + auto* cached = cfd->PeekPrecreatedMemtable(); + uint64_t number = cached ? cached->GetFileNumber() + : versions_->current_next_file_number(); + // A background-created file may enter the cache while the DB mutex is free. + if (!pending_outputs_.empty()) { + terark::minimize(number, pending_outputs_.front()); + } + pending_memtable = pending_outputs_.insert(pending_outputs_.begin(), number); + } + ROCKSDB_SCOPE_EXIT(if (pending_memtable != pending_outputs_.end()) { + mutex_.AssertHeld(); + pending_outputs_.erase(pending_memtable); + }); + // Attempt to switch to a new memtable and trigger flush of old. // Do this without holding the dbmutex lock. assert(versions_->prev_log_number() == 0); @@ -2368,6 +2384,7 @@ Status DBImpl::SwitchMemtable(ColumnFamilyData* cfd, WriteContext* context) { new_mem = cfd->ConstructNewMemtable(mutable_cf_options, seq); context->superversion_context.NewSuperVersion(); } + TEST_SYNC_POINT("DBImpl::SwitchMemtable:BeforeInstallMemTable"); ROCKS_LOG_INFO(immutable_db_options_.info_log, "[%s] New memtable created with log file: #%" PRIu64 ". Immutable memtables: %d.\n", @@ -2385,6 +2402,39 @@ Status DBImpl::SwitchMemtable(ColumnFamilyData* cfd, WriteContext* context) { assert(log_recycle_files_.front() == recycle_log_number); log_recycle_files_.pop_front(); } + if (s.ok() && immutable_db_options_.memtable_crash_safe_recover && + !new_mem->IsFileRegistered()) { + TEST_SYNC_POINT_CALLBACK("DBImpl::SwitchMemtable:MemTableCacheMiss", new_mem); + s = error_handler_.GetBGError(); + // A flush may have registered this file after it left the cache. + if (s.ok() && !cfd->GetMemTableFiles().count(new_mem->GetFileNumber())) { + VersionEdit edit; + edit.SetColumnFamily(cfd->GetID()); + edit.AddMemTableFile(new_mem->GetFileNumber()); + edit.SetMemTableFileTracking(); + TEST_SYNC_POINT("DBImpl::RegisterMemTableFile:BeforeLogAndApply"); + s = versions_->LogAndApply(cfd, mutable_cf_options, ReadOptions(), &edit, + &mutex_, directories_.GetDbDir()); + TEST_SYNC_POINT_CALLBACK("DBImpl::RegisterMemTableFile:AfterLogAndApply", &s); + if (!s.ok()) { + error_handler_.SetBGError(s, BackgroundErrorReason::kManifestWrite) + .PermitUncheckedError(); + } else { + s = error_handler_.GetBGError(); + } + } + if (s.ok() && shutting_down_.load(std::memory_order_acquire)) { + s = Status::ShutdownInProgress(); + } + if (!s.ok()) { + delete new_mem; + delete new_log; + context->superversion_context.new_superversion.reset(); + return s; + } + new_mem->MarkFileRegistered(); + TEST_SYNC_POINT("DBImpl::RegisterMemTableFile:BeforeInstall"); + } if (s.ok() && creating_new_log) { InstrumentedMutexLock l(&log_write_mutex_); assert(new_log != nullptr); diff --git a/db/db_memtable_convert_test.cc b/db/db_memtable_convert_test.cc index 85b60976d..5b31be622 100644 --- a/db/db_memtable_convert_test.cc +++ b/db/db_memtable_convert_test.cc @@ -2,6 +2,8 @@ // Flush/Close conversion, independently of crash-safe recovery. #include +#include +#include #include #include @@ -10,8 +12,14 @@ #include "db/db_impl/db_impl.h" #include "db/db_test_util.h" +#include "db/memtable.h" +#include "db/version_set.h" +#include "env/env_chroot.h" +#include "file/filename.h" +#include "memory/arena.h" #include "port/stack_trace.h" #include "test_util/sync_point.h" +#include "table/table_builder.h" namespace ROCKSDB_NAMESPACE { @@ -154,6 +162,154 @@ TEST_P(DBMemtableConvertTest, NonConversionKeepsMergeThreshold) { } } +TEST_P(DBMemtableConvertTest, FileMmapConversionKeepsFileNumberAndPath) { + if (std::string(std::get<1>(GetParam())) != "kFileMmap") return; + Options options = ConvertOptions(); + DestroyAndReopen(options); + ASSERT_OK(Put("same-file", "value")); + for (auto* cfd : *dbfull()->GetVersionSet()->GetColumnFamilySet()) { + ASSERT_TRUE(cfd->GetMemTableFiles().empty()); + } + const uint64_t number = dbfull()->GetVersionSet()->GetColumnFamilySet() + ->GetDefault()->mem()->GetFileNumber(); + const std::string path = MakeTableFileName(dbname_, number); + ASSERT_OK(env_->FileExists(path)); + ObserveConversion(); + ASSERT_OK(db_->Flush(FlushOptions())); + std::vector files; + db_->GetLiveFilesMetaData(&files); + ASSERT_EQ(files.size(), 1U); + ASSERT_EQ(files[0].file_number, number); + ASSERT_OK(env_->FileExists(path)); + for (auto* cfd : *dbfull()->GetVersionSet()->GetColumnFamilySet()) { + ASSERT_TRUE(cfd->GetMemTableFiles().empty()); + } + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("same-file"), "value"); +} + +#if !defined(OS_WIN) +TEST_P(DBMemtableConvertTest, FileMmapExclusiveCreationAndChrootConversion) { + if (std::string(std::get<1>(GetParam())) != "kFileMmap") return; + Options options = ConvertOptions(); + DestroyAndReopen(options); + Close(); + const std::string logical_path = MakeTableFileName("", 900000); + const std::string physical_path = dbname_ + logical_path; + auto factory = PluginFactorySP::AcquirePlugin( + std::get<0>(GetParam()) ? "OffsetSkipList" : "CSPPMemTab", + {{"mem_cap", 16777216}, {"convert_to_sst", "kFileMmap"}, + {"chroot_dir", dbname_}}, repo_); + InternalKeyComparator icmp(options.comparator); + MemTable::KeyComparator cmp(icmp); + MutableCFOptions moptions(options); + moptions.write_buffer_size = 1 << 20; + Arena arena; + if (std::get<0>(GetParam())) { + const std::string sentinel = "existing SST must remain intact"; + ASSERT_OK(WriteStringToFile(env_, sentinel, physical_path)); + ASSERT_THROW({ + std::unique_ptr collision(factory->CreateMemTableRep( + logical_path, moptions, cmp, &arena, nullptr, nullptr, 0)); + }, std::runtime_error); + std::string after; + ASSERT_OK(ReadFileToString(env_, physical_path, &after)); + ASSERT_EQ(after, sentinel); + ASSERT_OK(env_->DeleteFile(physical_path)); + } + + std::unique_ptr chroot_env(NewChrootEnv(env_, dbname_)); + options.env = chroot_env.get(); + options.cf_paths = {{"", 0}}; + ImmutableOptions ioptions(options); + IntTblPropCollectorFactories collectors; + const std::string cf_name = "default"; + TableBuilderOptions tbo(ioptions, moptions, icmp, &collectors, + options.compression, options.compression_opts, 0, + cf_name, 0); + std::unique_ptr rep(factory->CreateMemTableRep( + logical_path, moptions, cmp, &arena, nullptr, nullptr, 0)); + ASSERT_TRUE(rep->SupportCrashSafe()); + rep->InitSetMemTableAsLogIndex(false); + ASSERT_TRUE(rep->InsertKeyValue(PackSequenceAndType(1, kTypeValue), "k", "v")); + rep->MarkReadOnly(); + FileMetaData meta; + meta.fd = FileDescriptor(900001, 0, 0); + meta.num_entries = 1; + ASSERT_TRUE(rep->ConvertToSST(&meta, tbo).IsInvalidArgument()); + rep.reset(); + ASSERT_OK(env_->FileExists(physical_path)); + meta.fd = FileDescriptor(900000, 0, 0); + meta.fd.largest_seqno = kMaxSequenceNumber; + ASSERT_OK(factory->RecoverCrashSafeMemTableToSST(logical_path, &meta, tbo)); + ASSERT_GT(meta.fd.GetFileSize(), 0U); + ASSERT_OK(env_->FileExists(physical_path)); + ASSERT_OK(env_->DeleteFile(physical_path)); + + // Retry conversion on the same live rep after a completed footer exists. + rep.reset(factory->CreateMemTableRep( + logical_path, moptions, cmp, &arena, nullptr, nullptr, 0)); + rep->InitSetMemTableAsLogIndex(false); + ASSERT_TRUE(rep->InsertKeyValue(PackSequenceAndType(1, kTypeValue), "k", "v")); + rep->MarkReadOnly(); + meta.fd = FileDescriptor(900000, 0, 0); + ASSERT_OK(rep->ConvertToSST(&meta, tbo)); + const uint64_t first_size = meta.fd.GetFileSize(); + ASSERT_GT(first_size, 0U); + ASSERT_OK(rep->ConvertToSST(&meta, tbo)); + ASSERT_EQ(meta.fd.GetFileSize(), first_size); + uint64_t actual_size = 0; + ASSERT_OK(env_->GetFileSize(physical_path, &actual_size)); + ASSERT_EQ(actual_size, first_size); + std::string value; + rep->GetPIK(ReadOptions(), ParsedInternalKey("k", 1, kTypeValue), &value, + [](void* arg, const MemTableRep::KeyValuePair& kv) { + *static_cast(arg) = kv.value.ToString(); + return false; + }); + ASSERT_EQ(value, "v"); + rep.reset(); + meta.fd = FileDescriptor(900000, 0, 0); + meta.fd.largest_seqno = kMaxSequenceNumber; + ASSERT_OK(factory->RecoverCrashSafeMemTableToSST(logical_path, &meta, tbo)); + ASSERT_OK(env_->DeleteFile(physical_path)); +} +#endif + +TEST_P(DBMemtableConvertTest, FileMmapCloseKeepsEveryFileNumber) { + if (std::string(std::get<1>(GetParam())) != "kFileMmap") return; + Options options = ConvertOptions(); + DestroyAndReopen(options); + ASSERT_OK(Put("head", "1")); + auto* cfd = dbfull()->GetVersionSet()->GetColumnFamilySet()->GetDefault(); + const uint64_t head_number = cfd->mem()->GetFileNumber(); + ASSERT_OK(dbfull()->TEST_SwitchMemtable()); + ASSERT_OK(Put("tail", "2")); + const uint64_t tail_number = cfd->mem()->GetFileNumber(); + const std::set numbers{head_number, tail_number}; + ASSERT_EQ(numbers.size(), 2U); + for (auto* cfd : *dbfull()->GetVersionSet()->GetColumnFamilySet()) { + ASSERT_TRUE(cfd->GetMemTableFiles().empty()); + } + ObserveConversion(); + ASSERT_OK(db_->Close()); + Close(); + ASSERT_EQ(converts_.load(), 2); + for (uint64_t number : numbers) { + ASSERT_OK(env_->FileExists(MakeTableFileName(dbname_, number))); + } + ASSERT_OK(TryReopen(options)); + std::vector files; + db_->GetLiveFilesMetaData(&files); + ASSERT_EQ(files.size(), 2U); + for (const auto& file : files) { + ASSERT_EQ(numbers.count(file.file_number), 1U); + } + ASSERT_EQ(Get("head"), "1"); + ASSERT_EQ(Get("tail"), "2"); +} + TEST_P(DBMemtableConvertTest, CloseConvertsAllMemtables) { CheckClose(false); } diff --git a/db/flush_job.cc b/db/flush_job.cc index 0dce27e1f..9294e4cc6 100644 --- a/db/flush_job.cc +++ b/db/flush_job.cc @@ -209,7 +209,15 @@ void FlushJob::PickMemTable() { edit_->SetColumnFamily(cfd_->GetID()); // path 0 for level 0 file. - meta_.fd = FileDescriptor(versions_->NewFileNumber(), 0, 0); + const uint64_t file_number = [&]() { + if (m->SupportCrashSafe()) { + ROCKSDB_ASSERT_EQ(mems_.size(), 1); + return m->GetFileNumber(); + } else { + return versions_->NewFileNumber(); + } + }(); + meta_.fd = FileDescriptor(file_number, 0, 0); meta_.epoch_number = cfd_->NewEpochNumber(); base_ = cfd_->current(); @@ -1078,6 +1086,11 @@ Status FlushJob::WriteLevel0Table() { // should not be added to the manifest. const bool has_output = meta_.fd.GetFileSize() > 0; + if (s.ok() && mems_.front()->IsFileRegistered()) { + ROCKSDB_ASSERT_EQ(mems_.size(), 1); + edit_->DeleteMemTableFile(mems_.front()->GetFileNumber()); + } + if (s.ok() && has_output) { TEST_SYNC_POINT("DBImpl::FlushJob:SSTFileCreated"); // if we have more than 1 background thread, then we cannot diff --git a/db/memtable.cc b/db/memtable.cc index f8e7545bd..028854950 100644 --- a/db/memtable.cc +++ b/db/memtable.cc @@ -22,6 +22,7 @@ #include "db/range_tombstone_fragmenter.h" #include "db/read_callback.h" #include "db/wide/wide_column_serialization.h" +#include "file/filename.h" #include "logging/logging.h" #include "memory/arena.h" #include "memory/memory_usage.h" @@ -72,7 +73,8 @@ MemTable::MemTable(const InternalKeyComparator& cmp, const ImmutableOptions& ioptions, const MutableCFOptions& mutable_cf_options, WriteBufferManager* write_buffer_manager, - SequenceNumber latest_seq, uint32_t column_family_id) + SequenceNumber latest_seq, uint32_t column_family_id, + uint64_t file_number) : comparator_(cmp), moptions_(ioptions, mutable_cf_options), refs_(0), @@ -86,7 +88,9 @@ MemTable::MemTable(const InternalKeyComparator& cmp, : nullptr, mutable_cf_options.memtable_huge_page_size), table_(ioptions.memtable_factory->CreateMemTableRep( - ioptions.cf_paths[0].path, // level0_dir + file_number == 0 + ? std::string() + : TableFileName(ioptions.cf_paths, file_number, 0), mutable_cf_options, comparator_, &arena_, mutable_cf_options.prefix_extractor.get(), ioptions.logger, column_family_id)), @@ -105,7 +109,8 @@ MemTable::MemTable(const InternalKeyComparator& cmp, write_buffer_size_(mutable_cf_options.write_buffer_size), flush_in_progress_(false), flush_completed_(false), - file_number_(0), + file_registered_(false), + file_number_(file_number), first_seqno_(0), earliest_seqno_(latest_seq), creation_seq_(latest_seq), diff --git a/db/memtable.h b/db/memtable.h index aaa19e326..d291fa46a 100644 --- a/db/memtable.h +++ b/db/memtable.h @@ -154,7 +154,8 @@ class MemTable : public CacheAlignedNewDelete { const ImmutableOptions& ioptions, const MutableCFOptions& mutable_cf_options, WriteBufferManager* write_buffer_manager, - SequenceNumber earliest_seq, uint32_t column_family_id); + SequenceNumber earliest_seq, uint32_t column_family_id, + uint64_t file_number = 0); // No copying allowed MemTable(const MemTable&) = delete; MemTable& operator=(const MemTable&) = delete; @@ -585,6 +586,9 @@ class MemTable : public CacheAlignedNewDelete { void SetFlushCompleted(bool completed) { flush_completed_ = completed; } uint64_t GetFileNumber() const { return file_number_; } + bool SupportCrashSafe() const { return table_->SupportCrashSafe(); } + bool IsFileRegistered() const { return file_registered_; } + void MarkFileRegistered() { file_registered_ = true; } void SetFileNumber(uint64_t file_num) { file_number_ = file_num; } @@ -661,10 +665,11 @@ class MemTable : public CacheAlignedNewDelete { // These are used to manage memtable flushes to storage bool flush_in_progress_; // started the flush bool flush_completed_; // finished the flush + bool file_registered_; bool needs_user_key_cmp_in_get_; bool support_convert_to_sst_; bool reject_memtable_as_log_index_; - uint64_t file_number_; // filled up after flush is complete + uint64_t file_number_; // FileMmap backing file or flush output // The updates to be applied to the transaction log when this // memtable is flushed to storage. diff --git a/db/memtable_list.cc b/db/memtable_list.cc index c02bc5f9e..7a62ef2f4 100644 --- a/db/memtable_list.cc +++ b/db/memtable_list.cc @@ -453,7 +453,7 @@ void MemTableList::RollbackMemtableFlush(const autovector& mems, #ifndef NDEBUG for (MemTable* m : mems) { assert(m->flush_in_progress_); - assert(m->file_number_ == 0); + assert(m->file_number_ == 0 || m->SupportCrashSafe()); } #endif @@ -475,7 +475,9 @@ void MemTableList::RollbackMemtableFlush(const autovector& mems, m->flush_in_progress_ = false; m->flush_completed_ = false; m->edit_.Clear(); - m->file_number_ = 0; + if (!m->SupportCrashSafe()) { + m->file_number_ = 0; + } num_flush_not_started_++; ++it; } else { @@ -486,8 +488,7 @@ void MemTableList::RollbackMemtableFlush(const autovector& mems, for (MemTable* m : mems) { if (m->flush_in_progress_) { - assert(m->file_number_ == 0); - m->file_number_ = 0; + assert(m->file_number_ == 0 || m->SupportCrashSafe()); m->flush_in_progress_ = false; m->flush_completed_ = false; m->edit_.Clear(); @@ -621,10 +622,16 @@ Status MemTableList::TryInstallMemtableFlushResults( const auto manifest_write_cb = [this, cfd, batch_count, log_buffer, to_delete, mu](const Status& status) { + if (status.ok() && !cfd->IsDropped()) { + cfd->PublishRegisteredMemTables(); + TEST_SYNC_POINT("FlushJob::AfterManifest"); + } RemoveMemTablesOrRestoreFlags(status, cfd, batch_count, log_buffer, to_delete, mu); }; if (write_edits) { + cfd->AddMemTableFileEdits(edit_list.front()); + TEST_SYNC_POINT("FlushJob::BeforeManifest"); // this can release and reacquire the mutex. s = vset->LogAndApply(cfd, mutable_cf_options, read_options, edit_list, mu, db_directory, /*new_descriptor_log=*/false, @@ -697,6 +704,12 @@ bool MemTableList::UnflushedMemtablesSupportConvertToSST() const { return true; } +void MemTableList::AddMemTableFileNumbers(std::vector* live) const { + for (MemTable* mem : current_->memlist_) { + live->push_back(mem->GetFileNumber()); + } +} + size_t MemTableList::ApproximateMemoryUsage() { return current_memory_usage_; } size_t MemTableList::MemoryAllocatedBytesExcludingLast() const { @@ -813,7 +826,9 @@ void MemTableList::RemoveMemTablesOrRestoreFlags( m->flush_in_progress_ = false; m->edit_.Clear(); num_flush_not_started_++; - m->file_number_ = 0; + if (!m->SupportCrashSafe()) { + m->file_number_ = 0; + } imm_flush_needed.store(true, std::memory_order_release); ++mem_id; } @@ -887,6 +902,7 @@ Status InstallMemtableAtomicFlushResults( (*mems_list[k])[0]->ReleaseFlushJobInfo(); committed_flush_jobs_info[k]->push_back(std::move(flush_job_info)); } + cfds[k]->AddMemTableFileEdits((*mems_list[k])[0]->GetEdits()); } Status s; @@ -933,11 +949,16 @@ Status InstallMemtableAtomicFlushResults( assert(0 == num_entries); } + TEST_SYNC_POINT("FlushJob::BeforeManifest"); // this can release and reacquire the mutex. s = vset->LogAndApply(cfds, mutable_cf_options_list, read_options, edit_lists, mu, db_directory); for (size_t k = 0; k != cfds.size(); ++k) { + if (s.ok() && !cfds[k]->IsDropped()) { + cfds[k]->PublishRegisteredMemTables(); + TEST_SYNC_POINT("FlushJob::AfterManifest"); + } auto* imm = (imm_lists == nullptr) ? cfds[k]->imm() : imm_lists->at(k); imm->InstallNewVersion(); } @@ -1002,7 +1023,9 @@ Status InstallMemtableAtomicFlushResults( m->SetFlushCompleted(false); m->SetFlushInProgress(false); m->GetEdits()->Clear(); - m->SetFileNumber(0); + if (!m->SupportCrashSafe()) { + m->SetFileNumber(0); + } imm->num_flush_not_started_++; } imm->imm_flush_needed.store(true, std::memory_order_release); diff --git a/db/memtable_list.h b/db/memtable_list.h index 598694b31..2e66b446b 100644 --- a/db/memtable_list.h +++ b/db/memtable_list.h @@ -333,6 +333,7 @@ class MemTableList { size_t ApproximateUnflushedMemTablesMemoryUsage(); bool UnflushedMemtablesSupportConvertToSST() const; + void AddMemTableFileNumbers(std::vector* live) const; // Returns an estimate of the timestamp of the earliest key. uint64_t ApproximateOldestKeyTime() const; diff --git a/db/version_set.cc b/db/version_set.cc index b462fa626..ceb3b7ebc 100644 --- a/db/version_set.cc +++ b/db/version_set.cc @@ -5506,6 +5506,7 @@ void VersionSet::Reset() { obsolete_manifests_.clear(); wals_.Reset(); has_memtable_file_tracking_ = false; + replaying_manifest_ = true; } void VersionSet::AppendVersion(ColumnFamilyData* column_family_data, @@ -6382,6 +6383,7 @@ Status VersionSet::Recover( if (s.ok()) { manifest_file_size_ = current_manifest_file_size; + replaying_manifest_ = false; ROCKS_LOG_INFO( db_options_->info_log, "Recovered from manifest file:%s succeeded," @@ -6551,6 +6553,7 @@ Status VersionSet::TryRecoverFromOneManifest( s = handler_pit.status(); if (s.ok()) { RecoverEpochNumbers(); + replaying_manifest_ = false; } return s; } @@ -7643,10 +7646,12 @@ ColumnFamilyData* VersionSet::CreateColumnFamily( update_stats); AppendVersion(new_cfd, v); - // GetLatestMutableCFOptions() is safe here without mutex since the - // cfd is not available to client - new_cfd->CreateNewMemtable(*new_cfd->GetLatestMutableCFOptions(), - LastSequence()); + if (!cf_options.memtable_factory->SupportCrashSafe() || !replaying_manifest_) { + // GetLatestMutableCFOptions() is safe here without mutex since the + // cfd is not available to client + new_cfd->CreateNewMemtable(*new_cfd->GetLatestMutableCFOptions(), + LastSequence()); + } new_cfd->SetLogNumber(edit->GetLogNumber()); return new_cfd; } diff --git a/db/version_set.h b/db/version_set.h index 5aa84e0bb..0ea12bd89 100644 --- a/db/version_set.h +++ b/db/version_set.h @@ -1683,6 +1683,7 @@ class VersionSet { // Protected by DB mutex. WalSet wals_; bool has_memtable_file_tracking_ = false; + bool replaying_manifest_ = true; std::unique_ptr column_family_set_; Cache* table_cache_; diff --git a/include/rocksdb/memtablerep.h b/include/rocksdb/memtablerep.h index 41297e6ea..32e87d1df 100644 --- a/include/rocksdb/memtablerep.h +++ b/include/rocksdb/memtablerep.h @@ -372,8 +372,10 @@ class MemTableRepFactory : public Customizable { uint32_t /* column_family_id */) { return CreateMemTableRep(key_cmp, allocator, slice_transform, logger); } + // The DB supplies the final SST path for a file-backed memtable. + // The DB owns the backing file; the Rep must not delete it. virtual MemTableRep* CreateMemTableRep( - const std::string& /*level0_dir*/, + const std::string& /*memtable_file_path*/, const MutableCFOptions&, const MemTableRep::KeyComparator& key_cmp, Allocator* allocator, const SliceTransform* slice_transform, Logger* logger, @@ -405,19 +407,6 @@ class MemTableRepFactory : public Customizable { return convert_to_sst == ConvertKind::kFileMmap; } - // Append leftover crash-safe mmap paths under cf_dir (plus factory chroot). - // Default: no leftovers. - virtual void ListCrashSafeLeftovers(const std::string& /*cf_dir*/, - std::vector* /*leftovers*/) { - } - - // Read-only leftover probe for Recover check. Must not truncate/rename. - // wal_dir is used to verify each WAL fileno can still be opened. - virtual Status ProbeCrashSafeLeftover(const std::string& /*path*/, - const std::string& /*wal_dir*/) const { - return Status::OK(); - } - // Load leftover, cap visible entries, truncate, ConvertToSST. The caller // supplies the visibility bound in meta->fd.largest_seqno; preserve it for // subsequent table readers. From 1abbaa167ffc83ffaa3c600ab7e6d5b6a69124b8 Mon Sep 17 00:00:00 2001 From: leipeng Date: Mon, 5 Oct 2026 09:04:05 +0800 Subject: [PATCH 38/58] Test MemTable registration during cache handoff and GC A cached table can leave the queue while its registration is still in flight. Tests must exercise those interleavings, cache misses, and registration failures so an uncommitted file is neither exposed for writes nor removed by garbage collection. --- db/db_cspp_crash_safe_test.cc | 478 ++++++++++++++++++++++++++++++++++ 1 file changed, 478 insertions(+) diff --git a/db/db_cspp_crash_safe_test.cc b/db/db_cspp_crash_safe_test.cc index 55e220bc7..3c5603522 100644 --- a/db/db_cspp_crash_safe_test.cc +++ b/db/db_cspp_crash_safe_test.cc @@ -900,9 +900,487 @@ TEST_F(DBCsppCrashSafeTest, OpenAndCreateColumnFamilyRegisterNextMemTable) { } } +TEST_F(DBCsppCrashSafeTest, SwitchWaitsForPendingCacheRegistration) { + Close(); + for (bool osl : {false, true}) { + for (bool atomic : {false, true}) { + for (int mode : {0, 1, 2, 3}) { // Success, failure, shutdown, already committed. + const bool fail = mode == 1; + const bool shutdown = mode == 2; + SCOPED_TRACE(osl); + SCOPED_TRACE(atomic); + SCOPED_TRACE(mode); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.atomic_flush = atomic; + options.max_bgerror_resume_count = 0; + options.avoid_flush_during_shutdown = true; + if (osl) SetupOsl(&options, true); + Destroy(options); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(options)); + auto* cfd = dbfull()->GetVersionSet()->GetColumnFamilySet() + ->GetColumnFamily(0); + ASSERT_OK(Put("first", "1")); + + std::mutex mu; + std::condition_variable cv; + bool paused = false, release = false, waiting = false; + bool flush_done = false, switch_done = false, pause_timeout = false; + std::atomic frontend_registrations{0}; + const auto caller = std::this_thread::get_id(); + std::atomic arm{false}, inject{false}; + SyncPoint::GetInstance()->SetCallBack( + "FlushJob::BeforeManifest", [&](void*) { arm.store(true); }); + SyncPoint::GetInstance()->SetCallBack( + "VersionSet::LogAndApply:WriteManifestStart", [&](void*) { + if (!arm.exchange(false)) return; + std::unique_lock lk(mu); + paused = true; + inject.store(fail); + cv.notify_all(); + // Do not strand the flush thread if an assertion misses the hook. + pause_timeout = !cv.wait_for(lk, std::chrono::seconds(30), + [&] { return release; }); + }); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::SwitchMemtable:BeforeInstallMemTable", [&](void*) { + if (mode != 3) return; + std::unique_lock lk(mu); + if (!paused) return; + waiting = true; + cv.notify_all(); + if (!cv.wait_for(lk, std::chrono::seconds(30), + [&] { return release && flush_done; })) + std::abort(); + }); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::SwitchMemtable:MemTableCacheMiss", [&](void* p) { + EXPECT_NE(static_cast(p)->GetFileNumber(), 0U); + if (mode == 3) return; + std::lock_guard lk(mu); + waiting = true; + cv.notify_all(); + }); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::RegisterMemTableFile:BeforeLogAndApply", [&](void*) { + EXPECT_NE(std::this_thread::get_id(), caller); + ++frontend_registrations; + }); + SyncPoint::GetInstance()->SetCallBack( + "VersionSet::ProcessManifestWrites:AfterSyncManifest", [&](void* p) { + if (inject.exchange(false)) { + *static_cast(p) = IOStatus::IOError("cache wait injection"); + } + }); + Status flush_status, switch_status; + std::thread flush_thread([&] { + flush_status = Flush(); + std::lock_guard lk(mu); + flush_done = true; + cv.notify_all(); + }); + bool reached_pause; + { + std::unique_lock lk(mu); + reached_pause = cv.wait_for(lk, std::chrono::seconds(10), + [&] { return paused || flush_done; }) && paused; + } + std::vector cache_files; + std::vector before_switch, after_switch; + uint64_t active_file = 0; + if (reached_pause) { + dbfull()->TEST_LockMutex(); + if (auto* head = cfd->PeekPrecreatedMemtable()) { + cache_files.push_back(head->GetFileNumber()); + } + active_file = cfd->mem()->GetFileNumber(); + dbfull()->TEST_UnlockMutex(); + EXPECT_OK(env_->GetChildren(dbname_, &before_switch)); + EXPECT_OK(Put("active", "2")); + } + std::thread switch_thread([&] { + switch_status = dbfull()->TEST_SwitchMemtable(); + std::lock_guard lk(mu); + switch_done = true; + cv.notify_all(); + }); + bool reached_wait; + { + std::unique_lock lk(mu); + reached_wait = cv.wait_for(lk, std::chrono::seconds(10), + [&] { return waiting || switch_done; }) && waiting; + } + if (reached_pause && reached_wait) { + // The popped head remains protected by the switch's pending output. + std::vector still_cached; + dbfull()->TEST_LockMutex(); + if (auto* head = cfd->PeekPrecreatedMemtable()) { + still_cached.push_back(head->GetFileNumber()); + } + dbfull()->TEST_UnlockMutex(); + EXPECT_TRUE(still_cached.empty()); + EXPECT_OK(db_->DisableFileDeletions()); + EXPECT_OK(db_->EnableFileDeletions(true)); + for (uint64_t number : cache_files) { + EXPECT_OK(env_->FileExists(MakeTableFileName(dbname_, number))); + } + EXPECT_OK(env_->GetChildren(dbname_, &after_switch)); + auto only_ssts = [](const std::vector& children) { + std::set files; + for (const auto& child : children) { + if (child.size() >= 4 && child.compare(child.size() - 4, 4, ".sst") == 0) + files.insert(child); + } + return files; + }; + EXPECT_EQ(only_ssts(before_switch), only_ssts(after_switch)); + std::lock_guard lk(mu); + EXPECT_FALSE(switch_done); + } + if (shutdown && reached_wait) { + CancelAllBackgroundWork(db_, false); + } + { + std::unique_lock lk(mu); + release = true; + cv.notify_all(); + if (!cv.wait_for(lk, std::chrono::seconds(30), + [&] { return flush_done && switch_done; })) { + // A missed wakeup must fail this test instead of hanging in join. + std::abort(); + } + } + flush_thread.join(); + switch_thread.join(); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_TRUE(reached_pause); + ASSERT_TRUE(reached_wait); + ASSERT_FALSE(pause_timeout); + ASSERT_EQ(cache_files.size(), 1U); + ASSERT_EQ(frontend_registrations.load(), mode == 3 ? 0 : 1); + if (shutdown) { + ASSERT_TRUE(switch_status.IsShutdownInProgress()); + ASSERT_TRUE(flush_status.ok() || flush_status.IsShutdownInProgress()); + ASSERT_EQ(cfd->mem()->GetFileNumber(), active_file); + } else if (fail) { + ASSERT_TRUE(flush_status.IsIOError()); + ASSERT_TRUE(switch_status.IsIOError()); + ASSERT_TRUE(dbfull()->TEST_GetBGError().IsIOError()); + ASSERT_EQ(cfd->mem()->GetFileNumber(), active_file); + ASSERT_OK(db_->DisableFileDeletions()); + ASSERT_OK(db_->EnableFileDeletions(true)); + ASSERT_OK(env_->FileExists(MakeTableFileName(dbname_, active_file))); + } else { + ASSERT_OK(flush_status); + ASSERT_OK(switch_status); + ASSERT_EQ(cfd->mem()->GetFileNumber(), cache_files.front()); + ASSERT_TRUE(cfd->mem()->IsFileRegistered()); + ASSERT_OK(Put("switched", "3")); + } + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("first"), "1"); + ASSERT_EQ(Get("active"), "2"); + if (!fail && !shutdown) { + ASSERT_EQ(Get("switched"), "3"); + } + Close(); + } + } + } +} +TEST_F(DBCsppCrashSafeTest, CacheMissRegistersWhileFlushIsPaused) { + Close(); + for (bool osl : {false, true}) { + for (bool atomic : {false, true}) { + for (bool fail : {false, true}) { + SCOPED_TRACE(osl); + SCOPED_TRACE(atomic); + SCOPED_TRACE(fail); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.atomic_flush = atomic; + options.paranoid_checks = !fail; + options.avoid_flush_during_shutdown = true; + if (osl) SetupOsl(&options, true); + Destroy(options); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(options)); + auto* cfd = dbfull()->GetVersionSet()->GetColumnFamilySet() + ->GetColumnFamily(0); + ASSERT_OK(Put("first", "1")); + if (atomic) { + CreateColumnFamilies({"aux"}, options); + ASSERT_EQ(handles_.size(), 1U); + } + auto* aux = atomic ? handles_.back() : nullptr; + if (atomic) { + ASSERT_OK(db_->Put(WriteOptions(), aux, "aux-first", "3")); + } + std::mutex mu; + std::condition_variable cv; + bool paused = false, release = false, done = false, waiting = false; + bool pause_timeout = false; + std::atomic first_convert{true}; + std::atomic conversions{0}; + const auto caller = std::this_thread::get_id(); + std::thread::id switch_id; + std::atomic registrations{0}; + std::atomic foreground_registrations{0}; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::SwitchMemtable:MemTableCacheMiss", [&](void*) { + std::lock_guard lk(mu); + waiting = true; + cv.notify_all(); + }); + SyncPoint::GetInstance()->SetCallBack( + "MemTableRep::ConvertToSST:Before", [&](void*) { + if (!first_convert.exchange(false)) return; + std::unique_lock lk(mu); + paused = true; + cv.notify_all(); + pause_timeout = !cv.wait_for(lk, std::chrono::seconds(30), + [&] { return release; }); + }); + SyncPoint::GetInstance()->SetCallBack( + "FlushJob::ConvertToSST:Status", [&](void* p) { + EXPECT_TRUE(static_cast(p)->ok()); + // Atomic flush runs aux first, then default; fail only the last. + if (++conversions == (atomic ? 2 : 1) && fail) + *static_cast(p) = Status::IOError("ignored table flush injection"); + }); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::RegisterMemTableFile:BeforeLogAndApply", [&](void*) { + std::lock_guard lk(mu); + EXPECT_NE(std::this_thread::get_id(), caller); + if (std::this_thread::get_id() == switch_id) + ++foreground_registrations; + else + ++registrations; + }); + FlushOptions flush_options; + flush_options.wait = false; + const Status flush_status = atomic + ? db_->Flush(flush_options, {db_->DefaultColumnFamily(), aux}) + : db_->Flush(flush_options); + bool reached_pause; + { + std::unique_lock lk(mu); + reached_pause = cv.wait_for(lk, std::chrono::seconds(10), + [&] { return paused; }); + } + dbfull()->TEST_LockMutex(); + const bool cache_empty = cfd->PeekPrecreatedMemtable() == nullptr; + dbfull()->TEST_UnlockMutex(); + EXPECT_TRUE(cache_empty); + EXPECT_OK(Put("second", "2")); + Status switch_status; + std::thread switch_thread([&] { + { + std::lock_guard lk(mu); + switch_id = std::this_thread::get_id(); + } + switch_status = dbfull()->TEST_SwitchMemtable(); + std::lock_guard lk(mu); + done = true; + cv.notify_all(); + }); + bool reached_wait; + bool completed_before_release; + { + std::unique_lock lk(mu); + reached_wait = cv.wait_for( + lk, std::chrono::seconds(10), [&] { return waiting || done; }) && + waiting; + } + { + std::unique_lock lk(mu); + completed_before_release = cv.wait_for( + lk, std::chrono::seconds(10), [&] { return done; }); + release = true; + cv.notify_all(); + if (!cv.wait_for(lk, std::chrono::seconds(30), [&] { return done; })) + std::abort(); + } + switch_thread.join(); + ASSERT_OK(dbfull()->TEST_WaitForBackgroundWork()); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_TRUE(reached_pause); + ASSERT_TRUE(reached_wait); + ASSERT_TRUE(completed_before_release); + ASSERT_FALSE(pause_timeout); + ASSERT_OK(flush_status); + ASSERT_OK(switch_status); + ASSERT_OK(dbfull()->TEST_GetBGError()); + ASSERT_TRUE(cfd->mem()->IsFileRegistered()); + ASSERT_EQ(foreground_registrations.load(), 1); + ASSERT_EQ(conversions.load(), atomic ? 2 : 1); + ASSERT_EQ(registrations.load(), 0); + if (atomic) { + auto* aux_cfd = dbfull()->GetVersionSet()->GetColumnFamilySet() + ->GetColumnFamily(aux->GetID()); + dbfull()->TEST_LockMutex(); + auto* aux_head = aux_cfd->PeekPrecreatedMemtable(); + const uint64_t aux_cache_number = aux_head ? aux_head->GetFileNumber() : 0; + dbfull()->TEST_UnlockMutex(); + ASSERT_NE(aux_cache_number, 0U); + ASSERT_EQ(dbfull()->GetVersionSet()->GetColumnFamilySet() + ->GetColumnFamily(aux->GetID())->GetMemTableFiles() + .count(aux_cache_number), fail ? 0U : 1U); + if (fail) { + ASSERT_OK(dbfull()->TEST_SwitchMemtable(aux_cfd)); + ASSERT_EQ(aux_cfd->mem()->GetFileNumber(), aux_cache_number); + ASSERT_TRUE(aux_cfd->mem()->IsFileRegistered()); + ASSERT_EQ(aux_cfd->GetMemTableFiles().count(aux_cache_number), 1U); + } + } + Close(); + if (atomic) { + ASSERT_OK(TryReopenWithColumnFamilies({"default", "aux"}, options)); + ASSERT_EQ(Get(0, "first"), "1"); + ASSERT_EQ(Get(0, "second"), "2"); + ASSERT_EQ(Get(1, "aux-first"), "3"); + } else { + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("first"), "1"); + ASSERT_EQ(Get("second"), "2"); + } + Close(); + } + } + } +} +TEST_F(DBCsppCrashSafeTest, FrontendCacheMissRegistrationFailure) { + Close(); + for (bool osl : {false, true}) { + for (bool after_sync : {false, true}) { + SCOPED_TRACE(osl); + SCOPED_TRACE(after_sync); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.max_bgerror_resume_count = 0; + options.avoid_flush_during_shutdown = true; + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("first", "1")); + ASSERT_OK(dbfull()->TEST_SwitchMemtable()); + ASSERT_OK(Put("active", "2")); + auto* cfd = dbfull()->GetVersionSet()->GetColumnFamilySet()->GetDefault(); + const uint64_t active = cfd->mem()->GetFileNumber(); + ASSERT_EQ(cfd->PeekPrecreatedMemtable(), nullptr); + const auto caller = std::this_thread::get_id(); + int injected = 0; + uint64_t candidate = 0; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::SwitchMemtable:MemTableCacheMiss", [&](void* p) { + candidate = static_cast(p)->GetFileNumber(); + EXPECT_NE(candidate, active); + EXPECT_EQ(cfd->PeekPrecreatedMemtable(), nullptr); + }); + SyncPoint::GetInstance()->SetCallBack( + after_sync ? "VersionSet::ProcessManifestWrites:AfterSyncManifest" + : "DBImpl::RegisterMemTableFile:AfterLogAndApply", + [&](void* p) { + EXPECT_EQ(std::this_thread::get_id(), caller); + ++injected; + if (after_sync) { + *static_cast(p) = IOStatus::IOError("frontend register injection"); + } else { + *static_cast(p) = Status::IOError("frontend register injection"); + } + }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_TRUE(dbfull()->TEST_SwitchMemtable().IsIOError()); + ASSERT_EQ(injected, 1); + ASSERT_TRUE(dbfull()->TEST_GetBGError().IsIOError()); + ASSERT_EQ(cfd->mem()->GetFileNumber(), active); + ASSERT_NE(candidate, 0U); + ASSERT_EQ(cfd->GetMemTableFiles().count(candidate), after_sync ? 0U : 1U); + ASSERT_EQ(cfd->PeekPrecreatedMemtable(), nullptr); + ASSERT_OK(env_->FileExists(MakeTableFileName(dbname_, candidate))); + ASSERT_NOK(Put("rejected", "3")); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_OK(db_->DisableFileDeletions()); + ASSERT_OK(db_->EnableFileDeletions(true)); + ASSERT_OK(env_->FileExists(MakeTableFileName(dbname_, active))); + ASSERT_EQ(Get("first"), "1"); + ASSERT_EQ(Get("active"), "2"); + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("first"), "1"); + ASSERT_EQ(Get("active"), "2"); + ASSERT_EQ(Get("rejected"), "NOT_FOUND"); + Close(); + } + } +} +TEST_F(DBCsppCrashSafeTest, FailedCacheRegistrationSurvivesGcAndReopen) { + Close(); + for (bool osl : {false, true}) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.max_bgerror_resume_count = 0; + if (osl) SetupOsl(&options, true); + Destroy(options); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("retry", "preserved")); + std::vector pending; + std::atomic pending_commit{false}; + SyncPoint::GetInstance()->SetCallBack("FlushJob::BeforeManifest", [&](void*) { + std::vector children; + ASSERT_OK(env_->GetChildren(dbname_, &children)); + for (const auto& child : children) { + if (child.size() >= 4 && child.compare(child.size() - 4, 4, ".sst") == 0) + pending.push_back(dbname_ + "/" + child); + } + pending_commit.store(true); + }); + std::atomic injected{false}; + std::atomic published{0}; + SyncPoint::GetInstance()->SetCallBack( + "VersionSet::ProcessManifestWrites:AfterSyncManifest", [&](void* p) { + if (pending_commit.load() && !injected.exchange(true)) + *static_cast(p) = IOStatus::IOError("cache register injection"); + }); + SyncPoint::GetInstance()->SetCallBack( + "FlushJob::MemTableCache:AfterPublish", [&](void*) { ++published; }); + ASSERT_NOK(Flush()); + ASSERT_TRUE(injected.load()); + ASSERT_EQ(published.load(), 0); + ASSERT_GE(pending.size(), 3U); // converting input, active, pending cache + ASSERT_OK(db_->DisableFileDeletions()); + ASSERT_OK(db_->EnableFileDeletions(true)); + for (const auto& path : pending) ASSERT_OK(env_->FileExists(path)); + SyncPoint::GetInstance()->ClearCallBack("FlushJob::BeforeManifest"); + // Plain MANIFEST IOError is fatal under the existing error policy. + // Resume preserves that error; reopening is the supported recovery path. + const Status resumed = db_->Resume(); + ASSERT_TRUE(resumed.IsIOError()); + ASSERT_EQ(Get("retry"), "preserved"); + Close(); + SyncPoint::GetInstance()->ClearCallBack( + "VersionSet::ProcessManifestWrites:AfterSyncManifest"); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("retry"), "preserved"); + published.store(0); + ASSERT_OK(Put("after-reopen", "committed")); + ASSERT_OK(Flush()); + ASSERT_GT(published.load(), 0); + ASSERT_EQ(Get("retry"), "preserved"); + Close(); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("retry"), "preserved"); + ASSERT_EQ(Get("after-reopen"), "committed"); + Close(); + } +} #endif TEST_F(DBCsppCrashSafeTest, CrashSafeRequiresFileMmapFactories) { From 048c2c9f54da943cd2007c9003cfc0629f56e40c Mon Sep 17 00:00:00 2001 From: leipeng Date: Mon, 5 Oct 2026 09:04:07 +0800 Subject: [PATCH 39/58] Test crash windows around MemTable inventory commits A completed conversion is not yet a committed recovery transition. Crashes on either side of a MANIFEST commit must preserve a usable inventory, and an interrupted recovery must remain retryable. Cover these windows for registration, ordinary flush, atomic multi-CF flush, and repeated recovery rather than testing only clean reopen. --- db/db_cspp_crash_safe_test.cc | 305 ++++++++++++++++++++++++++++++++++ 1 file changed, 305 insertions(+) diff --git a/db/db_cspp_crash_safe_test.cc b/db/db_cspp_crash_safe_test.cc index 3c5603522..f246ab4b1 100644 --- a/db/db_cspp_crash_safe_test.cc +++ b/db/db_cspp_crash_safe_test.cc @@ -432,7 +432,60 @@ TEST_F(DBCsppCrashSafeTest, FailedInitialRegistrationClearsDbPointer) { } } +TEST_F(CrashChild, DISABLED_RegistrationCommitWindow) { + ASSERT_EQ(arg_.size(), 3U); + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (arg_[0] == '1') SetupOsl(&options, true); + const char* point = arg_[2] == '0' + ? (arg_[1] == '0' ? "DBImpl::RegisterMemTableFile:AfterLogAndApply" + : "DBImpl::RegisterMemTableFile:BeforeInstall") + : (arg_[1] == '0' ? "FlushJob::MemTableCache:BeforePublish" + : "FlushJob::MemTableCache:AfterPublish"); + const auto arm = [&] { + SyncPoint::GetInstance()->SetCallBack(point, [](void*) { ::_exit(42); }); + SyncPoint::GetInstance()->EnableProcessing(); + }; + if (arg_[2] == '0') arm(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "before-switch", "preserved")); + if (arg_[2] == '1') arm(); + ASSERT_OK(child_db->Flush(FlushOptions())); + ::_exit(1); +} +TEST_F(DBCsppCrashSafeTest, RegistrationCommitCrashKeepsManifestInventory) { + Close(); + for (bool osl : {false, true}) { + for (bool marked : {false, true}) { + for (bool switching : {false, true}) { + SCOPED_TRACE(osl); + SCOPED_TRACE(marked); + SCOPED_TRACE(switching); + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + Destroy(options); + const std::string arg = std::to_string(osl) + std::to_string(marked) + + std::to_string(switching); + ASSERT_EQ(RunCrashChild(dbname_, "RegistrationCommitWindow", arg), 42); + const auto registered = ListLeftovers(options, dbname_); + ASSERT_EQ(registered.size(), 2U); + for (const auto& path : registered) ASSERT_OK(env_->FileExists(path)); + ASSERT_OK(TryReopen(options)); + ASSERT_TRUE(dbfull()->GetVersionSet()->HasMemTableFileTracking()); + ASSERT_EQ(Get("before-switch"), switching ? "preserved" : "NOT_FOUND"); + ASSERT_OK(Put("after-crash", "committed")); + Close(); + for (const auto& path : ListLeftovers(options, dbname_)) + ASSERT_OK(env_->FileExists(path)); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("before-switch"), switching ? "preserved" : "NOT_FOUND"); + ASSERT_EQ(Get("after-crash"), "committed"); + Close(); + } + } + } +} TEST_F(DBCsppCrashSafeTest, FailedNewColumnFamilyRegistrationRemainsReopenable) { Close(); @@ -666,7 +719,64 @@ TEST_F(DBCsppCrashSafeTest, LegacyManifestWithoutTrackingUsesFullWal) { } } +TEST_F(CrashChild, DISABLED_FlushManifestWindow) { + ASSERT_EQ(arg_.size(), 2U); + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (arg_[0] == '1') SetupOsl(&options, true); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "converted", "durable")); + SyncPoint::GetInstance()->SetCallBack( + arg_[1] == '0' ? "FlushJob::BeforeManifest" + : "FlushJob::AfterManifest", + [](void*) { ::_exit(42); }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(child_db->Flush(FlushOptions())); + ::_exit(1); +} +TEST_F(DBCsppCrashSafeTest, ConversionCrashAcrossManifestCommit) { + Close(); + for (bool osl : {false, true}) { + for (bool committed : {false, true}) { + SCOPED_TRACE(osl); + SCOPED_TRACE(committed); + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + Destroy(options); + const std::string arg = std::string(osl ? "1" : "0") + + (committed ? "1" : "0"); + ASSERT_EQ(RunCrashChild(dbname_, "FlushManifestWindow", arg), 42); + const auto registered = ListLeftovers(options, dbname_); + ASSERT_FALSE(registered.empty()); + for (const auto& path : registered) ASSERT_OK(env_->FileExists(path)); + std::atomic converts{0}; + SyncPoint::GetInstance()->SetCallBack( + "MemTableRep::ConvertToSST:After", + [&](void*) { ++converts; }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(options)); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_EQ(converts.load(), committed ? 0 : 1); + ASSERT_EQ(Get("converted"), "durable"); + ASSERT_EQ(CountL0(db_), 1); + std::vector files; + db_->GetLiveFilesMetaData(&files); + ASSERT_EQ(files.size(), 1U); + if (!committed) { + const std::string path = MakeTableFileName(dbname_, files[0].file_number); + ASSERT_EQ(std::count(registered.begin(), registered.end(), path), 1); + ASSERT_OK(env_->FileExists(path)); + } + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("converted"), "durable"); + ASSERT_EQ(CountL0(db_), 1); + Close(); + } + } +} TEST_F(DBCsppCrashSafeTest, GarbageCollectionKeepsActiveAndCachedMemTables) { Close(); @@ -4351,8 +4461,131 @@ TEST_F(DBCsppCrashSafeTest, TruncateInjectFailureFallsBackToWal) { SyncPoint::GetInstance()->ClearAllCallBacks(); } +TEST_F(CrashChild, DISABLED_AfterConvertBeforeAddFileKeepsRegisteredFile) { + ASSERT_EQ(arg_.size(), 3U); + Options options = BaseCrashSafeOptions(dbname_, true, arg_[1] == '1'); + if (arg_[0] == '1') SetupOsl(&options, true); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", + [](void*) { ::_exit(1); }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + auto* cfd = static_cast(child_db)->GetVersionSet() + ->GetColumnFamilySet()->GetDefault(); + ASSERT_OK(WriteStringToFile( + options.env, MakeTableFileName(dbname_, cfd->mem()->GetFileNumber()), + dbname_ + "/active-file")); + ASSERT_OK(child_db->Put(WriteOptions(), "or", std::string(128, 'v'))); + ::_exit(0); +} +TEST_F(CrashChild, DISABLED_AfterConvertBeforeAddFileKeepsRegisteredFileRecover) { + ASSERT_EQ(arg_.size(), 3U); + Options options = BaseCrashSafeOptions(dbname_, true, arg_[1] == '1'); + if (arg_[0] == '1') SetupOsl(&options, true); + const char* points[] = { + "MemTableRep::ConvertToSST:Truncate", + "CrashSafeRecover::AfterConvertBeforeAddFile", + "CrashSafeRecover::AfterConvertBeforeAddFile", + "DBImpl::RegisterMemTableFile:AfterLogAndApply"}; + ASSERT_LT(arg_[2] - '0', 4); + SyncPoint::GetInstance()->SetCallBack( + points[arg_[2] - '0'], + [&](void*) { + if (arg_[2] == '1') { + // Model an incomplete SST tail, without claiming this callback runs + // in the middle of a write. The persisted trie remains intact. + const auto files = ListLeftovers(options, dbname_); + ASSERT_EQ(files.size(), 2U); + std::string active; + ASSERT_OK(ReadFileToString(options.env, dbname_ + "/active-file", &active)); + ASSERT_EQ(std::count(files.begin(), files.end(), active), 1); + const int fd = ::open(active.c_str(), O_RDWR); + ASSERT_GE(fd, 0); + uint64_t structure_size = 0; + if (arg_[0] == '1') { + terark::OSL_MmapHeader hdr{}; + ASSERT_EQ(::pread(fd, &hdr, sizeof(hdr), 0), + static_cast(sizeof(hdr))); + structure_size = hdr.mem_used; + } else { + terark::DFA_MmapHeader hdr{}; + ASSERT_EQ(::pread(fd, &hdr, sizeof(hdr), 0), + static_cast(sizeof(hdr))); + structure_size = hdr.file_size; + } + uint64_t size = 0; + ASSERT_OK(options.env->GetFileSize(active, &size)); + ASSERT_GT(size, structure_size); + ASSERT_EQ(::ftruncate(fd, size - 1), 0); + ::close(fd); + } + ::_exit(1); + }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* recover_db = nullptr; + DB::Open(options, dbname_, &recover_db); + ::_exit(0); +} +TEST_F(DBCsppCrashSafeTest, AfterConvertBeforeAddFileKeepsRegisteredFile) { + Close(); + for (bool osl : {false, true}) { + for (bool log_index : {false, true}) { + for (int window = 0; window < 4; ++window) { + const std::string config = std::to_string(osl) + + std::to_string(log_index) + std::to_string(window); + SCOPED_TRACE(config); + Options options = BaseCrashSafeOptions(dbname_, true, log_index); + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, + "AfterConvertBeforeAddFileKeepsRegisteredFile", config), 1); + const auto before = ListLeftovers(options, dbname_); + ASSERT_EQ(before.size(), 2U); + std::string active; + ASSERT_OK(ReadFileToString(env_, dbname_ + "/active-file", &active)); + ASSERT_EQ(std::count(before.begin(), before.end(), active), 1); + for (const auto& path : before) ASSERT_OK(env_->FileExists(path)); + // Before commit, interrupt the same file twice to exercise footer + // replacement, not just a one-time conversion of the original source. + for (int crash = 0; crash < (window == 3 ? 1 : 2); ++crash) { + ASSERT_EQ(RunCrashChild(dbname_, + "AfterConvertBeforeAddFileKeepsRegisteredFileRecover", config), 1); + ASSERT_OK(env_->FileExists(active)); + if (window != 3) { + ASSERT_EQ(ListLeftovers(options, dbname_), before); + for (const auto& path : before) ASSERT_OK(env_->FileExists(path)); + } + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.generation & 1, 0U); + } + int converted = 0; + SyncPoint::GetInstance()->SetCallBack( + "MemTableRep::ConvertToSST:After", [&](void*) { ++converted; }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(options)); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_EQ(converted, window == 3 ? 0 : 1); + ASSERT_EQ(Get("or"), std::string(128, 'v')); + std::vector files; + db_->GetLiveFilesMetaData(&files); + ASSERT_EQ(files.size(), 1U); + ASSERT_EQ(files.front().level, 0); + ASSERT_EQ(MakeTableFileName(dbname_, files.front().file_number), + active); + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("or"), std::string(128, 'v')); + ASSERT_EQ(CountL0(db_), 1); + Close(); + } + } + } +} TEST_F(DBCsppCrashSafeTest, CrashSafeOrLogIndexDisablesWalCompression) { for (bool crash_safe : {false, true}) { @@ -4720,7 +4953,79 @@ TEST_F(DBCsppCrashSafeTest, AtomicFlushDualCfLeftoverConvertsBoth) { ASSERT_GE(CountL0(db_, "one"), 1); } +TEST_F(CrashChild, DISABLED_AtomicFlushManifestWindow) { + ASSERT_EQ(arg_.size(), 2U); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.atomic_flush = true; + if (arg_[0] == '1') SetupOsl(&options, true); + DB* child_db = nullptr; + std::vector handles; + const std::vector cfs = { + {kDefaultColumnFamilyName, options}, {"one", options}}; + ASSERT_OK(DB::Open(options, dbname_, cfs, &handles, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), handles[0], "default-key", "one")); + ASSERT_OK(child_db->Put(WriteOptions(), handles[1], "other-key", "two")); + SyncPoint::GetInstance()->SetCallBack( + arg_[1] == '0' ? "FlushJob::BeforeManifest" : "FlushJob::AfterManifest", + [](void*) { ::_exit(42); }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(child_db->Flush(FlushOptions(), handles)); + ::_exit(1); +} +TEST_F(DBCsppCrashSafeTest, AtomicFlushCrashAcrossManifestCommit) { + Close(); + for (bool osl : {false, true}) { + for (bool committed : {false, true}) { + SCOPED_TRACE(osl); + SCOPED_TRACE(committed); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.atomic_flush = true; + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_OK(TryReopen(options)); + CreateAndReopenWithCF({"one"}, options); + Close(); + const std::string arg = std::string(osl ? "1" : "0") + + (committed ? "1" : "0"); + ASSERT_EQ(RunCrashChild(dbname_, "AtomicFlushManifestWindow", arg), 42); + const auto registered = ListLeftovers(options, dbname_); + ASSERT_GE(registered.size(), 2U); + for (const auto& path : registered) ASSERT_OK(env_->FileExists(path)); + std::atomic converted{0}; + SyncPoint::GetInstance()->SetCallBack( + "MemTableRep::ConvertToSST:After", + [&](void*) { ++converted; }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopenWithColumnFamilies( + {kDefaultColumnFamilyName, "one"}, options)); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_EQ(converted.load(), committed ? 0 : 2); + ASSERT_EQ(Get(0, "default-key"), "one"); + ASSERT_EQ(Get(1, "other-key"), "two"); + ASSERT_EQ(CountL0(db_, kDefaultColumnFamilyName), 1); + ASSERT_EQ(CountL0(db_, "one"), 1); + std::vector files; + db_->GetLiveFilesMetaData(&files); + ASSERT_EQ(files.size(), 2U); + for (const auto& file : files) { + const std::string path = MakeTableFileName(dbname_, file.file_number); + ASSERT_OK(env_->FileExists(path)); + ASSERT_EQ(std::count(registered.begin(), registered.end(), path), + committed ? 0 : 1); + } + Close(); + ASSERT_OK(TryReopenWithColumnFamilies( + {kDefaultColumnFamilyName, "one"}, options)); + ASSERT_EQ(Get(0, "default-key"), "one"); + ASSERT_EQ(Get(1, "other-key"), "two"); + ASSERT_EQ(CountL0(db_, kDefaultColumnFamilyName), 1); + ASSERT_EQ(CountL0(db_, "one"), 1); + Close(); + } + } +} TEST_F(CrashChild, DISABLED_DroppedCfLeftoverSkipped) { Options options = BaseCrashSafeOptions(dbname_, true, false); From 6a7b5507a152e5665a17161f0766bec27f65e61c Mon Sep 17 00:00:00 2001 From: leipeng Date: Tue, 6 Oct 2026 00:26:21 +0800 Subject: [PATCH 40/58] Test CSPP DumpMem header initialization A successful table reopen does not detect missing CSPP header metadata: the table reader can obtain its format information elsewhere. Validate the serialized header and its checksum so DumpMem cannot silently omit them. --- db/db_cspp_crash_safe_test.cc | 17 +++++++++++++++++ 1 file changed, 17 insertions(+) diff --git a/db/db_cspp_crash_safe_test.cc b/db/db_cspp_crash_safe_test.cc index f246ab4b1..b7028de05 100644 --- a/db/db_cspp_crash_safe_test.cc +++ b/db/db_cspp_crash_safe_test.cc @@ -27,6 +27,7 @@ #include #include #include +#include #include "db/column_family.h" #include "db/db_impl/db_impl.h" @@ -1982,6 +1983,22 @@ TEST_F(DBCsppCrashSafeTest, ConvertedTableVisibilityFilter) { mem.reset(); const std::string fname = TableFileName(options.cf_paths, 1, 0); + if (!osl && !file_mmap) { + const int fd = ::open(fname.c_str(), O_RDONLY); + ASSERT_GE(fd, 0); + terark::DFA_MmapHeader header{}; + const ssize_t read = ::pread(fd, &header, sizeof(header), 0); + ::close(fd); + ASSERT_EQ(read, static_cast(sizeof(header))); + uint32_t prefix[3]; + memcpy(prefix, header.reserved, sizeof(prefix)); + ASSERT_EQ(prefix[0], 0x50505343U); // CSPP crash-safe magic. + ASSERT_EQ(prefix[1], 0U); // No WAL references in this conversion. + ASSERT_EQ(prefix[2], 0U); + ASSERT_GE(header.crc32cLevel, 1U); + ASSERT_EQ(header.header_crc32, + terark::Crc32c_update(0, &header, sizeof(header) - 4)); + } const std::type_info* unfiltered_type = nullptr; for (SequenceNumber limit : {kMaxSequenceNumber, meta.fd.largest_seqno}) { std::unique_ptr file; From d2761bc845d22258c94e67df0fbebe300ee06f6c Mon Sep 17 00:00:00 2001 From: leipeng Date: Tue, 6 Oct 2026 00:27:12 +0800 Subject: [PATCH 41/58] Verify file-owned visibility of recovered SSTs FileDescriptor sequence bounds belong to a DB's version state, not to every reader of an SST. A recovered table must expose the same published prefix through DB readers, SstFileReader, and SstFileDumper regardless of caller-provided bounds. Ordinary converted tables must remain unfiltered. TableReader views must also distinguish finite pubseq values from kMaxSequenceNumber. --- db/db_cspp_crash_safe_test.cc | 262 ++++++++++++++++++++-------------- 1 file changed, 156 insertions(+), 106 deletions(-) diff --git a/db/db_cspp_crash_safe_test.cc b/db/db_cspp_crash_safe_test.cc index b7028de05..9af0ce942 100644 --- a/db/db_cspp_crash_safe_test.cc +++ b/db/db_cspp_crash_safe_test.cc @@ -46,12 +46,16 @@ #include "rocksdb/io_status.h" #include "rocksdb/convenience.h" #include "rocksdb/statistics.h" +#include "rocksdb/sst_file_reader.h" #include "rocksdb/utilities/checkpoint.h" #include "rocksdb/utilities/transaction_db.h" #include "rocksdb/wal_filter.h" +#include "table/format.h" #include "table/get_context.h" +#include "table/sst_file_dumper.h" #include "table/table_builder.h" #include "table/table_reader.h" +#include "table/top_table_reader.h" #include "utilities/merge_operators.h" #include "utilities/fault_injection_fs.h" #include "test_util/sync_point.h" @@ -1727,7 +1731,7 @@ TEST_F(DBCsppCrashSafeTest, RecoveredTableUsesFileSequenceBound) { const std::string fname = TableFileName(options.cf_paths, 2, 0); for (SequenceNumber limit : {meta.fd.largest_seqno, SequenceNumber(4), - SequenceNumber(0)}) { + SequenceNumber(0), kMaxSequenceNumber}) { SCOPED_TRACE(limit); std::unique_ptr file; ASSERT_OK(env_->GetFileSystem()->NewRandomAccessFile( @@ -1737,14 +1741,29 @@ TEST_F(DBCsppCrashSafeTest, RecoveredTableUsesFileSequenceBound) { EnvOptions env_options; TableReaderOptions tro(ioptions, options.prefix_extractor, env_options, icmp, 0); - // Reopen the same bytes with a different FileDescriptor visibility bound. + // The file header owns visibility, independently of the caller's bound. FileDescriptor fd = meta.fd; fd.largest_seqno = limit; tro.largest_seqno = fd.largest_seqno; + if (osl) { + const auto block = ReadMetaBlockE( + reader.get(), fd.GetFileSize(), 0x62546d654d4c534fULL, ioptions, + "OffsetSkipList"); + ASSERT_EQ(block.data.size(), 48U); + // Packed metadata has a 40-byte prefix followed by uint64_t pubseq. + uint64_t pubseq; + memcpy(&pubseq, block.data.data() + 40, sizeof(pubseq)); + ASSERT_EQ(pubseq, 2U); + } std::unique_ptr table; ASSERT_OK(options.table_factory->NewTableReader( ReadOptions(), tro, std::move(reader), fd.GetFileSize(), &table, true)); + const auto& compression = table->GetTableProperties()->compression_options; + ASSERT_EQ(compression.substr(0, compression.find(';', 1)), ";pubseq:2"); + const auto* view = dynamic_cast(table.get()); + ASSERT_NE(view, nullptr); + ASSERT_EQ(json::parse(view->ToWebViewString({{"html", false}}))["pubseq"], 2); for (const char* key : {"key", "ghost"}) { PinnableSlice value; GetContext get_context( @@ -1753,15 +1772,29 @@ TEST_F(DBCsppCrashSafeTest, RecoveredTableUsesFileSequenceBound) { InternalKey ikey(key, kMaxSequenceNumber, kTypeValue); ASSERT_OK(table->Get(ReadOptions(), ikey.Encode(), &get_context, nullptr)); - const bool visible = limit == 4 || (limit == 2 && key[0] == 'k'); + const bool visible = key[0] == 'k'; ASSERT_EQ(get_context.State(), visible ? GetContext::kFound : GetContext::kNotFound); if (visible) { - ASSERT_EQ(value.ToString(), key[0] == 'g' ? "unpublished" - : limit == 2 ? "old" : "new"); + ASSERT_EQ(value.ToString(), "old"); } } } + SstFileReader standalone(options); + ASSERT_OK(standalone.Open(fname)); + std::unique_ptr standalone_it(standalone.NewIterator(ReadOptions())); + standalone_it->SeekToFirst(); + ASSERT_TRUE(standalone_it->Valid()); + ASSERT_EQ(standalone_it->key().ToString(), "key"); + ASSERT_EQ(standalone_it->value().ToString(), "old"); + standalone_it->Next(); + ASSERT_FALSE(standalone_it->Valid()); + ASSERT_OK(standalone_it->status()); + SstFileDumper dumper(options, fname, Temperature::kUnknown, 0, + true, false, false, EnvOptions(), true); + ASSERT_OK(dumper.getStatus()); + ASSERT_OK(dumper.ReadSequential(false, 0, false, "", false, "")); + ASSERT_EQ(dumper.GetReadNumber(), 1U); } } @@ -1822,13 +1855,27 @@ TEST_F(DBCsppCrashSafeTest, RecoveredTableIteratorUsesFileSequenceBound) { TableReaderOptions tro(ioptions, options.prefix_extractor, env_options, icmp, 0); tro.largest_seqno = limit; + if (osl) { + const auto block = ReadMetaBlockE( + reader.get(), meta.fd.GetFileSize(), 0x62546d654d4c534fULL, + ioptions, "OffsetSkipList"); + ASSERT_EQ(block.data.size(), 48U); + uint64_t pubseq; + memcpy(&pubseq, block.data.data() + 40, sizeof(pubseq)); + ASSERT_EQ(pubseq, 4U); + } std::unique_ptr table; ASSERT_OK(options.table_factory->NewTableReader( ReadOptions(), tro, std::move(reader), meta.fd.GetFileSize(), &table, true)); + const auto& compression = table->GetTableProperties()->compression_options; + ASSERT_EQ(compression.substr(0, compression.find(';', 1)), ";pubseq:4"); + const auto* view = dynamic_cast(table.get()); + ASSERT_NE(view, nullptr); + ASSERT_EQ(json::parse(view->ToWebViewString({{"html", false}}))["pubseq"], 4); std::vector expected; for (const auto& key : physical) { - if (GetInternalKeySeqno(key) <= limit) expected.push_back(key); + if (GetInternalKeySeqno(key) <= 4) expected.push_back(key); } for (bool use_arena : {false, true}) { SCOPED_TRACE(use_arena); @@ -1912,7 +1959,7 @@ TEST_F(DBCsppCrashSafeTest, RecoveredTableIteratorUsesFileSequenceBound) { for (int i = 0; i < 20; ++i) { mem_it->RandomSeek(); if (it->Valid()) { - ASSERT_LE(GetInternalKeySeqno(it->key()), limit); + ASSERT_LE(GetInternalKeySeqno(it->key()), 4U); } } } @@ -1922,111 +1969,114 @@ TEST_F(DBCsppCrashSafeTest, RecoveredTableIteratorUsesFileSequenceBound) { } } -TEST_F(DBCsppCrashSafeTest, ConvertedTableVisibilityFilter) { +TEST_F(DBCsppCrashSafeTest, ConvertedTableUsesUnboundedHeaderSequence) { Close(); for (bool osl : {false, true}) { for (bool file_mmap : {false, true}) { - for (const auto& marker : { - std::make_pair("", false), {"VisFilter:0", false}, - {"VisFilter:10", false}, {"XVisFilter:1", false}, - {"Other:VisFilter:1", false}, {"VisFilter:1x", false}, - {"VisFilter:1", true}, {"Other:0;VisFilter:1", true}, - {"Other:0;VisFilter:1;Tail:0", true}}) { - SCOPED_TRACE(osl ? "OSL" : "CSPP"); - SCOPED_TRACE(file_mmap); - SCOPED_TRACE(marker.first); - Options options = BaseCrashSafeOptions(dbname_, true, false); - const char* js = file_mmap - ? R"({"mem_cap":16777216,"convert_to_sst":"kFileMmap"})" - : R"({"mem_cap":16777216,"convert_to_sst":"kDumpMem"})"; - if (osl) { - options.memtable_factory = EasyNewMemTableRep("OffsetSkipList", js); - } else { - options.memtable_factory.reset(NewCSPPMemTabForPlain(js)); - } - const SidePluginRepo repo; - options.table_factory = PluginFactorySP::AcquirePlugin( - osl ? "OffsetSkipListTable" : "CSPPMemTabTable", json::parse(js), repo); - Destroy(options); - ASSERT_OK(env_->CreateDirIfMissing(dbname_)); - options.cf_paths = {{dbname_, 0}}; - InternalKeyComparator icmp(options.comparator); - ImmutableOptions ioptions(options); - MutableCFOptions moptions(options); - WriteBufferManager wb(options.db_write_buffer_size); - auto mem = std::make_unique( - icmp, ioptions, moptions, &wb, kMaxSequenceNumber, 0, 1); - ASSERT_OK(mem->Add(1, kTypeValue, "a", "1", nullptr)); - ASSERT_OK(mem->Add(2, kTypeValue, "b", "2", nullptr)); - ASSERT_OK(mem->Add(3, kTypeValue, "b", "3", nullptr)); - mem->MarkImmutable(); - IntTblPropCollectorFactories collectors; - TableBuilderOptions tbo(ioptions, moptions, icmp, &collectors, - options.compression, options.compression_opts, 0, - kDefaultColumnFamilyName, 0); - FileMetaData meta; - meta.fd = FileDescriptor(1, 0, 0); - meta.fd.smallest_seqno = 1; - meta.fd.largest_seqno = 3; - SyncPoint::GetInstance()->SetCallBack( - "PropertyBlockBuilder::AddTableProperty:Start", [marker](void* p) { - auto& opts = static_cast(p)->compression_options; - ASSERT_EQ(opts.find("VisFilter:"), std::string::npos); - opts += marker.first; - }); - SyncPoint::GetInstance()->EnableProcessing(); - const Status converted = mem->ConvertToSST(&meta, tbo); - SyncPoint::GetInstance()->DisableProcessing(); - SyncPoint::GetInstance()->ClearAllCallBacks(); - ASSERT_OK(converted); - ASSERT_GT(meta.fd.GetFileSize(), 0U); - mem.reset(); + SCOPED_TRACE(osl ? "OSL" : "CSPP"); + SCOPED_TRACE(file_mmap); + Options options = BaseCrashSafeOptions(dbname_, true, false); + const char* js = file_mmap + ? R"({"mem_cap":16777216,"convert_to_sst":"kFileMmap"})" + : R"({"mem_cap":16777216,"convert_to_sst":"kDumpMem"})"; + if (osl) { + options.memtable_factory = EasyNewMemTableRep("OffsetSkipList", js); + } else { + options.memtable_factory.reset(NewCSPPMemTabForPlain(js)); + } + const SidePluginRepo repo; + options.table_factory = PluginFactorySP::AcquirePlugin( + osl ? "OffsetSkipListTable" : "CSPPMemTabTable", json::parse(js), repo); + Destroy(options); + ASSERT_OK(env_->CreateDirIfMissing(dbname_)); + options.cf_paths = {{dbname_, 0}}; + InternalKeyComparator icmp(options.comparator); + ImmutableOptions ioptions(options); + MutableCFOptions moptions(options); + WriteBufferManager wb(options.db_write_buffer_size); + auto mem = std::make_unique( + icmp, ioptions, moptions, &wb, kMaxSequenceNumber, 0, 1); + ASSERT_OK(mem->Add(1, kTypeValue, "a", "1", nullptr)); + ASSERT_OK(mem->Add(2, kTypeValue, "b", "2", nullptr)); + ASSERT_OK(mem->Add(3, kTypeValue, "b", "3", nullptr)); + mem->MarkImmutable(); + IntTblPropCollectorFactories collectors; + TableBuilderOptions tbo(ioptions, moptions, icmp, &collectors, + options.compression, options.compression_opts, 0, + kDefaultColumnFamilyName, 0); + FileMetaData meta; + meta.fd = FileDescriptor(1, 0, 0); + meta.fd.smallest_seqno = 1; + meta.fd.largest_seqno = 3; + ASSERT_OK(mem->ConvertToSST(&meta, tbo)); + ASSERT_GT(meta.fd.GetFileSize(), 0U); + mem.reset(); - const std::string fname = TableFileName(options.cf_paths, 1, 0); - if (!osl && !file_mmap) { - const int fd = ::open(fname.c_str(), O_RDONLY); - ASSERT_GE(fd, 0); - terark::DFA_MmapHeader header{}; - const ssize_t read = ::pread(fd, &header, sizeof(header), 0); - ::close(fd); - ASSERT_EQ(read, static_cast(sizeof(header))); - uint32_t prefix[3]; - memcpy(prefix, header.reserved, sizeof(prefix)); - ASSERT_EQ(prefix[0], 0x50505343U); // CSPP crash-safe magic. - ASSERT_EQ(prefix[1], 0U); // No WAL references in this conversion. - ASSERT_EQ(prefix[2], 0U); - ASSERT_GE(header.crc32cLevel, 1U); - ASSERT_EQ(header.header_crc32, - terark::Crc32c_update(0, &header, sizeof(header) - 4)); + const std::string fname = TableFileName(options.cf_paths, 1, 0); + if (!osl && !file_mmap) { + const int fd = ::open(fname.c_str(), O_RDONLY); + ASSERT_GE(fd, 0); + terark::DFA_MmapHeader header{}; + const ssize_t read = ::pread(fd, &header, sizeof(header), 0); + ::close(fd); + ASSERT_EQ(read, static_cast(sizeof(header))); + uint32_t prefix[3]; + memcpy(prefix, header.reserved, sizeof(prefix)); + ASSERT_EQ(prefix[0], 0x50505343U); // CSPP crash-safe magic. + ASSERT_EQ(prefix[1], 0U); // No WAL references in this conversion. + ASSERT_EQ(prefix[2], 0U); + // pubseq follows the 16-byte prefix and sixteen 24-byte WAL slots. + uint64_t pubseq; + memcpy(&pubseq, header.reserved + 400, sizeof(pubseq)); + ASSERT_EQ(pubseq, 0U); // Unbounded ordinary conversion. + ASSERT_GE(header.crc32cLevel, 1U); + ASSERT_EQ(header.header_crc32, + terark::Crc32c_update(0, &header, sizeof(header) - 4)); + } + const std::type_info* unfiltered_type = nullptr; + for (SequenceNumber limit : {kMaxSequenceNumber, SequenceNumber(0), + SequenceNumber(2), meta.fd.largest_seqno}) { + std::unique_ptr file; + ASSERT_OK(env_->GetFileSystem()->NewRandomAccessFile( + fname, FileOptions(), &file, nullptr)); + auto reader = std::make_unique(std::move(file), fname); + EnvOptions env_options; + TableReaderOptions tro(ioptions, options.prefix_extractor, env_options, + icmp, 0); + tro.largest_seqno = limit; + if (osl) { + const auto block = ReadMetaBlockE( + reader.get(), meta.fd.GetFileSize(), 0x62546d654d4c534fULL, + ioptions, "OffsetSkipList"); + ASSERT_EQ(block.data.size(), 48U); + uint64_t pubseq; + memcpy(&pubseq, block.data.data() + 40, sizeof(pubseq)); + ASSERT_EQ(pubseq, 0U); } - const std::type_info* unfiltered_type = nullptr; - for (SequenceNumber limit : {kMaxSequenceNumber, meta.fd.largest_seqno}) { - std::unique_ptr file; - ASSERT_OK(env_->GetFileSystem()->NewRandomAccessFile( - fname, FileOptions(), &file, nullptr)); - auto reader = std::make_unique(std::move(file), fname); - EnvOptions env_options; - TableReaderOptions tro(ioptions, options.prefix_extractor, env_options, - icmp, 0); - tro.largest_seqno = limit; - std::unique_ptr table; - ASSERT_OK(options.table_factory->NewTableReader( - ReadOptions(), tro, std::move(reader), meta.fd.GetFileSize(), - &table, true)); - std::unique_ptr it(table->NewIterator( - ReadOptions(), nullptr, nullptr, false, - TableReaderCaller::kUserIterator)); - if (limit == kMaxSequenceNumber) { - unfiltered_type = &typeid(*it); - } else { - ASSERT_NE(unfiltered_type, nullptr); - // A finite file maximum alone must not select VisibleIter. - ASSERT_EQ(typeid(*it) == *unfiltered_type, !marker.second); - } - size_t count = 0; - for (it->SeekToFirst(); it->Valid(); it->Next()) ++count; - ASSERT_EQ(count, 3U); + std::unique_ptr table; + ASSERT_OK(options.table_factory->NewTableReader( + ReadOptions(), tro, std::move(reader), meta.fd.GetFileSize(), + &table, true)); + const auto& compression = table->GetTableProperties()->compression_options; + ASSERT_EQ(compression.find("pubseq:"), std::string::npos); + ASSERT_EQ(compression.find("VisFilter:"), std::string::npos); + const auto* view = dynamic_cast(table.get()); + ASSERT_NE(view, nullptr); + ASSERT_EQ(json::parse(view->ToWebViewString({{"html", false}}))["pubseq"], + "kMaxSequenceNumber"); + std::unique_ptr it(table->NewIterator( + ReadOptions(), nullptr, nullptr, false, + TableReaderCaller::kUserIterator)); + if (limit == kMaxSequenceNumber) { + unfiltered_type = &typeid(*it); + } else { + ASSERT_NE(unfiltered_type, nullptr); + // A caller's finite maximum must not select VisibleIter. + ASSERT_EQ(typeid(*it), *unfiltered_type); } + size_t count = 0; + for (it->SeekToFirst(); it->Valid(); it->Next()) ++count; + ASSERT_EQ(count, 3U); } } } From 4fddef85204eac88371f6598976f1c3e2562469a Mon Sep 17 00:00:00 2001 From: leipeng Date: Tue, 6 Oct 2026 12:20:25 +0800 Subject: [PATCH 42/58] Document the 100/80-column wrapping rule Keep short statements together without allowing already-wrapped statements to become unnecessarily wide. Apply this rule only to changed code so formatting does not obscure the functional diff or disturb existing code. --- AGENTS.md | 5 +++++ 1 file changed, 5 insertions(+) create mode 100644 AGENTS.md diff --git a/AGENTS.md b/AGENTS.md new file mode 100644 index 000000000..398bb9094 --- /dev/null +++ b/AGENTS.md @@ -0,0 +1,5 @@ +# Formatting + +- Keep each single statement on one line when the complete line, including indentation, fits within 100 columns. This includes calls, declarations, assignments, returns, and `if` / `while` / `for` headers; do not collapse block bodies. +- If that line would exceed 100 columns, wrap it to 80 columns per line, including indentation. Preserve the surrounding continuation-indent style. +- Format only uncommitted added or modified code. Preserve untouched existing code; do not run whole-file formatting. From 11c0a73e3d334a3e6e1512f6d399cdf15d6dfbea Mon Sep 17 00:00:00 2001 From: leipeng Date: Tue, 6 Oct 2026 12:21:41 +0800 Subject: [PATCH 43/58] Verify persistent MemTable statistics across writer lifetimes Recovery statistics must not depend on whether a writer was still alive, had exited, or had reported its pending WAL bytes before the process died. Check both MemTable implementations against counts derived from the writes, including normal conversion and the cached memory estimate. Header aggregate counters are not the recovery source of truth. Clearing or inflating them must not affect totals rebuilt from cumulative writers, and repeating recovery must produce the same table and WAL statistics. --- db/db_cspp_crash_safe_test.cc | 224 ++++++++++++++++++++++++++++++++-- 1 file changed, 217 insertions(+), 7 deletions(-) diff --git a/db/db_cspp_crash_safe_test.cc b/db/db_cspp_crash_safe_test.cc index 9af0ce942..aa48939eb 100644 --- a/db/db_cspp_crash_safe_test.cc +++ b/db/db_cspp_crash_safe_test.cc @@ -2406,6 +2406,213 @@ static Options LogRefCrashOptions(const std::string& dbname, return options; } +static Options PersistentStatsOptions(const std::string& dbname, const std::string& config) { + Options options = BaseCrashSafeOptions(dbname, true, config[1] != 'N'); + options.cf_paths = {{dbname, 0}}; + const bool osl = config[0] == 'O'; + const json params = { + {"mem_cap", 16777216}, {"convert_to_sst", "kFileMmap"}, + {"enable_gc", config[2] == 'E'}, + {"log_ref_format", config[1] == 'S' ? "kShortLogRef" : "kPlainLogRef"}}; + options.memtable_factory = EasyNewMemTableRep( + osl ? "OffsetSkipList" : "CSPPMemTab", params.dump()); + const SidePluginRepo repo; + options.table_factory = PluginFactorySP::AcquirePlugin( + osl ? "OffsetSkipListTable" : "CSPPMemTabTable", params, repo); + options.merge_operator = MergeOperators::CreateStringAppendOperator(); + return options; +} + +TEST_F(CrashChild, DISABLED_PersistentMemTableStats) { + Options options = PersistentStatsOptions(dbname_, arg_); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + std::mutex gate; + std::condition_variable ready_cv; + size_t ready = 0; + const bool exited = arg_[2] == 'E'; + const bool window = arg_[2] == 'B' || arg_[2] == 'A'; + const size_t value_size = arg_[2] == 'T' || arg_[2] == 'F' ? 600 * 1024 : 128; + auto* cfd = static_cast(child_db)->GetVersionSet()->GetColumnFamilySet()->GetDefault(); + auto write = [&](int id) { + WriteOptions wo; + wo.memtable_insert_hint_per_batch = id == 1; + const std::string prefix = std::to_string(id); + const bool check_memory = arg_[2] == 'T' || arg_[2] == 'F'; + size_t memory_before = 0; + if (check_memory) { + // Warm this writer's TLS allocation before measuring WAL-cache growth. + ASSERT_OK(child_db->Put(wo, prefix + "dup", "")); + memory_before = cfd->mem()->ApproximateMemoryUsage(); + } + WriteBatch batch; + if (!check_memory) { + ASSERT_OK(batch.Put(prefix + "dup", "")); + } + ASSERT_OK(batch.Put(prefix + "dup", std::string(value_size, 'v'))); + ASSERT_OK(batch.Delete(prefix + "del")); + ASSERT_OK(batch.SingleDelete(prefix + "one")); + ASSERT_OK(batch.Put(prefix + "last", "v")); + ASSERT_OK(child_db->Write(wo, &batch)); + ASSERT_OK(child_db->Merge(wo, prefix + "mer", "m")); + if (check_memory) { + const size_t memory_after = cfd->mem()->ApproximateMemoryUsage(); + ASSERT_GE(memory_after, memory_before + value_size); + ASSERT_LT(memory_after, memory_before + value_size + 64 * 1024); + } + std::unique_lock lock(gate); + ready++; + ready_cv.notify_all(); + if (!exited) { + ready_cv.wait(lock, [] { return false; }); // TLS remains live at _exit. + } + }; + std::thread first(write, 0); + { + std::unique_lock lock(gate); + ready_cv.wait(lock, [&] { return ready == 1; }); + } + if (exited) first.join(); + const std::string active = MakeTableFileName(dbname_, cfd->mem()->GetFileNumber()); + ASSERT_OK(WriteStringToFile(options.env, active, dbname_ + "/stats-active")); + if (window) { + SyncPoint::GetInstance()->SetCallBack( + arg_[2] == 'B' ? "MemTableStats::BeforePublish" + : "MemTableStats::AfterPublish", + [](void*) { ::_exit(42); }); + SyncPoint::GetInstance()->EnableProcessing(); + std::thread next([&] { + ASSERT_OK(child_db->Put(WriteOptions(), "never-counted", "v")); + }); + next.join(); + ::_exit(1); // The first insertion must reach the publication hook. + } + std::thread second(write, 1); + { + std::unique_lock lock(gate); + ready_cv.wait(lock, [&] { return ready == 2; }); + } + if (exited) second.join(); + // Independent oracle: each writer has six entries, two deletions, one merge, + // 25 user-key bytes plus six tags, and value_size + 2 real value bytes. + ASSERT_EQ(cfd->mem()->num_entries(), 12U); + ASSERT_EQ(cfd->mem()->num_deletes(), 4U); + ASSERT_EQ(cfd->mem()->num_merges(), 2U); + ASSERT_EQ(cfd->mem()->raw_key_size(), 146U); + ASSERT_EQ(cfd->mem()->raw_value_size(), 2 * (value_size + 2)); + if (arg_[2] == 'F') { + FlushOptions flush; + flush.wait = true; + ASSERT_OK(child_db->Flush(flush)); + ColumnFamilyMetaData meta; + child_db->GetColumnFamilyMetaData(&meta); + ASSERT_EQ(meta.blob_files.size(), 1U); + ASSERT_EQ(meta.blob_files[0].total_blob_count, 2U); + ASSERT_EQ(meta.blob_files[0].total_blob_bytes, 2 * value_size); + } + ::_exit(42); +} + +TEST_F(DBCsppCrashSafeTest, PersistentWalStatsNormalConversion) { + Close(); + for (const char* config : {"CPF", "OPF"}) { + SCOPED_TRACE(config); + Options options = PersistentStatsOptions(dbname_, config); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "PersistentMemTableStats", config), 42); + } +} + +TEST_F(DBCsppCrashSafeTest, PersistentMemTableStatsSurviveWriterLifetime) { + Close(); + for (const char* config : {"CNL", "CNE", "CPL", "CPE", "CSL", "CSE", + "ONL", "ONE", "OPL", "OPE", "OSL", "OSE", + "CNB", "CNA", "ONB", "ONA", "CPT", "OPT"}) { + SCOPED_TRACE(config); + Options options = PersistentStatsOptions(dbname_, config); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "PersistentMemTableStats", config), 42); + std::string active; + ASSERT_OK(ReadFileToString(env_, dbname_ + "/stats-active", &active)); + const uint64_t batches = config[2] == 'B' || config[2] == 'A' ? 1 : 2; + const uint64_t value_size = config[2] == 'T' ? 600 * 1024 : 128; + uint64_t file_number; + FileType file_type; + ASSERT_TRUE(ParseFileName(active.substr(active.find_last_of('/') + 1), + &file_number, &file_type)); + ASSERT_EQ(file_type, kTableFile); + FileMetaData meta; + meta.fd = FileDescriptor(file_number, 0, 0); + meta.fd.largest_seqno = batches * 6; + InternalKeyComparator icmp(options.comparator); + ImmutableOptions ioptions(options); + MutableCFOptions moptions(options); + IntTblPropCollectorFactories collectors; + TableBuilderOptions tbo(ioptions, moptions, icmp, &collectors, + options.compression, options.compression_opts, 0, + kDefaultColumnFamilyName, 0); + uint64_t next_blob = 900000; + tbo.generate_file_no = [&] { return next_blob++; }; + std::vector blobs; + tbo.add_blob_file = [&](BlobFileAddition blob) { blobs.push_back(blob); }; + // Every conversion rebuilds header totals from durable writer counters. + for (int attempt = 0; attempt < 2; attempt++) { + SCOPED_TRACE(attempt); + blobs.clear(); + int recoveries = 0; + SyncPoint::GetInstance()->SetCallBack( + "MemTableStats::RecoverWals:AfterReset", + [&](void*) { recoveries++; }); + SyncPoint::GetInstance()->EnableProcessing(); + Status s = options.memtable_factory->RecoverCrashSafeMemTableToSST(active, &meta, tbo); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearCallBack("MemTableStats::RecoverWals:AfterReset"); + ASSERT_OK(s); + ASSERT_EQ(recoveries, 1); + ASSERT_EQ(meta.num_entries, batches * 6); + ASSERT_EQ(meta.num_deletions, batches * 2); + ASSERT_EQ(meta.num_merges, batches); + ASSERT_EQ(meta.raw_key_size, batches * 73); + ASSERT_EQ(meta.raw_value_size, batches * (value_size + 2)); + ASSERT_EQ(blobs.size(), config[1] == 'N' ? 0U : 1U); + for (const auto& blob : blobs) { + ASSERT_EQ(blob.GetTotalBlobCount(), batches); + ASSERT_EQ(blob.GetTotalBlobBytes(), batches * value_size); + } + if (config[1] != 'N') { + const int fd = ::open(active.c_str(), O_RDONLY); + ASSERT_GE(fd, 0); + const size_t offset = config[0] == 'O' + ? offsetof(terark::OSL_MmapHeader, reserved) + 2 * sizeof(uint32_t) + : offsetof(terark::DFA_MmapHeader, reserved) + 4 * sizeof(uint32_t); + uint64_t wal[3]; // fileno, cnt, bytes + const ssize_t n = ::pread(fd, wal, sizeof(wal), offset); + const int close_result = ::close(fd); + ASSERT_EQ(n, static_cast(sizeof(wal))); + ASSERT_EQ(close_result, 0); + ASSERT_EQ(wal[1], batches); + ASSERT_EQ(wal[2], batches * value_size); + } + SstFileReader reader(options); + ASSERT_OK(reader.Open(active)); + const auto properties = reader.GetTableProperties(); + ASSERT_EQ(properties->num_entries, batches * 6); + ASSERT_EQ(properties->num_deletions, batches * 2); + ASSERT_EQ(properties->num_merge_operands, batches); + ASSERT_EQ(properties->raw_key_size, batches * 73); + ASSERT_EQ(properties->raw_value_size, batches * (value_size + 2)); + if (config[1] != 'N') { + const auto& compression = properties->compression_options; + unsigned long long cnt, bytes; + ASSERT_EQ(sscanf(compression.c_str() + compression.find(';') + 1, + "%*u:%*u:%llu:%llu", &cnt, &bytes), 2); + ASSERT_EQ(cnt, batches); + ASSERT_EQ(bytes, batches * value_size); + } + } // Destroy the reader before recovering the same file again. + } +} + TEST_F(CrashChild, DISABLED_LogRefRecoveryIgnoresCounters) { Options options = LogRefCrashOptions(dbname_, arg_); DB* child_db = nullptr; @@ -2439,7 +2646,7 @@ TEST_F(DBCsppCrashSafeTest, LogRefRecoveryIgnoresCounters) { ASSERT_EQ(std::count(leftovers.begin(), leftovers.end(), active), 1); const int fd = ::open(active.c_str(), O_RDWR); ASSERT_GE(fd, 0); - // Header statistics are not a source of truth for WAL references. + // Recovery rebuilds header totals from durable writer counters. const size_t offset = config[0] == 'O' ? offsetof(terark::OSL_MmapHeader, reserved) + 2 * sizeof(uint32_t) : offsetof(terark::DFA_MmapHeader, reserved) + 4 * sizeof(uint32_t); @@ -2447,8 +2654,8 @@ TEST_F(DBCsppCrashSafeTest, LogRefRecoveryIgnoresCounters) { const ssize_t n = ::pread(fd, wal, sizeof(wal), offset); ASSERT_EQ(n, static_cast(sizeof(wal))); if (config[2] == 'S') { - // Simulate a crash before TLS statistics were flushed to the mapped header. - // A zero approximate count must not discard the real WAL references. + // Deliberately clear header counters. Recovery must retain the real WAL + // references and reconstruct totals from durable writer statistics. wal[1] = 0; wal[2] = 0; ASSERT_EQ(::pwrite(fd, wal, sizeof(wal), offset), @@ -2462,10 +2669,9 @@ TEST_F(DBCsppCrashSafeTest, LogRefRecoveryIgnoresCounters) { ColumnFamilyMetaData cf_meta; db_->GetColumnFamilyMetaData(&cf_meta); ASSERT_EQ(cf_meta.blob_files.size(), 1U); - ASSERT_EQ(cf_meta.blob_files[0].total_blob_count, - std::max(wal[1], 1)); - ASSERT_EQ(cf_meta.blob_files[0].total_blob_bytes, - std::max(wal[2], 1)); + const uint64_t count = config[2] == 'L' ? 5000 : 3; + ASSERT_EQ(cf_meta.blob_files[0].total_blob_count, count); + ASSERT_EQ(cf_meta.blob_files[0].total_blob_bytes, count * 128); ASSERT_EQ(Get("0"), std::string(128, 'v')); ASSERT_EQ(Get("1"), std::string(128, 'v')); ASSERT_EQ(Get("inline"), "v"); @@ -2524,6 +2730,10 @@ TEST_F(DBCsppCrashSafeTest, LogRefRecoveryMultipleWals) { ColumnFamilyMetaData meta; db_->GetColumnFamilyMetaData(&meta); ASSERT_EQ(meta.blob_files.size(), 3U); + for (const auto& blob : meta.blob_files) { + ASSERT_EQ(blob.total_blob_count, 1U); + ASSERT_EQ(blob.total_blob_bytes, 128U); + } std::unique_ptr it(db_->NewIterator(ReadOptions())); it->SeekToFirst(); for (int i = 0; i < 3; ++i) { From a2331cc7b853c80d8bbe6ab43293f16fa3bd9089 Mon Sep 17 00:00:00 2001 From: leipeng Date: Tue, 6 Oct 2026 12:21:52 +0800 Subject: [PATCH 44/58] Test persistent statistics with concurrent and duplicate inserts Per-writer statistics must include all successful concurrent insertions without including a rejected duplicate. Compare recovered table metadata with the MemTable wrapper and the known workload, covering both ordinary insertion and insertion hints in CSPP and OffsetSkipList. --- db/db_cspp_crash_safe_test.cc | 75 +++++++++++++++++++++++++++++++++++ 1 file changed, 75 insertions(+) diff --git a/db/db_cspp_crash_safe_test.cc b/db/db_cspp_crash_safe_test.cc index aa48939eb..75a5eb8ed 100644 --- a/db/db_cspp_crash_safe_test.cc +++ b/db/db_cspp_crash_safe_test.cc @@ -2613,6 +2613,81 @@ TEST_F(DBCsppCrashSafeTest, PersistentMemTableStatsSurviveWriterLifetime) { } } +TEST_F(DBCsppCrashSafeTest, PersistentMemTableStatsConcurrentAndDuplicateAdds) { + Close(); + for (const char* config : {"CNE", "ONE"}) { + SCOPED_TRACE(config); + Options options = PersistentStatsOptions(dbname_, config); + Destroy(options); + ASSERT_OK(env_->CreateDirIfMissing(dbname_)); + InternalKeyComparator icmp(options.comparator); + ImmutableOptions ioptions(options); + MutableCFOptions moptions(options); + WriteBufferManager wb(options.db_write_buffer_size); + auto mem = std::make_unique(icmp, ioptions, moptions, &wb, kMaxSequenceNumber, 0, 1); + constexpr size_t writers = 4, per_writer = 16; + std::atomic ready{0}; + std::atomic go{false}; + MemTablePostProcessInfo counters[writers]; + std::vector threads; + for (size_t id = 0; id < writers; id++) { + threads.emplace_back([&, id] { + ready++; + while (!go.load()) std::this_thread::yield(); + void* hint = nullptr; + for (size_t i = 0; i < per_writer; i++) { + std::string key(3, 'a'); + key[0] = char('a' + id); + key[1] = char('A' + i); + ASSERT_OK(mem->Add(id * per_writer + i + 1, kTypeValue, key, + std::string(128, 'v'), nullptr, true, + &counters[id], id % 2 ? &hint : nullptr)); + } + mem->FinishHint(hint); + }); + } + while (ready.load() != writers) std::this_thread::yield(); + go.store(true); + for (auto& thread : threads) thread.join(); + for (const auto& counter : counters) mem->BatchPostProcess(counter); + ASSERT_OK(mem->Add(1000, kTypeValue, "dup", "v", nullptr)); + ASSERT_TRUE(mem->Add(1000, kTypeValue, "dup", "v", nullptr).IsTryAgain()); + // Four writers add distinct three-byte keys; the rejected duplicate adds + // neither a record nor bytes to either the wrapper or persistent counters. + constexpr uint64_t entries = writers * per_writer + 1; + constexpr uint64_t key_bytes = entries * (3 + 8); + constexpr uint64_t value_bytes = writers * per_writer * 128 + 1; + ASSERT_EQ(mem->num_entries(), entries); + ASSERT_EQ(mem->raw_key_size(), key_bytes); + ASSERT_EQ(mem->raw_value_size(), value_bytes); + mem->MarkImmutable(); + const std::string leftover = MakeTableFileName(dbname_, 2); + CopyFile(MakeTableFileName(dbname_, 1), leftover); + mem.reset(); + IntTblPropCollectorFactories collectors; + TableBuilderOptions tbo(ioptions, moptions, icmp, &collectors, + options.compression, options.compression_opts, 0, + kDefaultColumnFamilyName, 0); + FileMetaData meta; + meta.fd = FileDescriptor(2, 0, 0); + meta.fd.largest_seqno = 1000; + ASSERT_OK(options.memtable_factory->RecoverCrashSafeMemTableToSST(leftover, &meta, tbo)); + ASSERT_EQ(meta.num_entries, entries); + ASSERT_EQ(meta.num_deletions, 0U); + ASSERT_EQ(meta.num_merges, 0U); + ASSERT_EQ(meta.raw_key_size, key_bytes); + ASSERT_EQ(meta.raw_value_size, value_bytes); + SstFileReader reader(options); + ASSERT_OK(reader.Open(leftover)); + const auto properties = reader.GetTableProperties(); + ASSERT_EQ(properties->num_entries, entries); + ASSERT_EQ(properties->num_deletions, 0U); + ASSERT_EQ(properties->num_merge_operands, 0U); + ASSERT_EQ(properties->raw_key_size, key_bytes); + ASSERT_EQ(properties->raw_value_size, value_bytes); + } +} + TEST_F(CrashChild, DISABLED_LogRefRecoveryIgnoresCounters) { Options options = LogRefCrashOptions(dbname_, arg_); DB* child_db = nullptr; From 8a561b665e38f0cebb4acaddbdcb8fc61b23854f Mon Sep 17 00:00:00 2001 From: leipeng Date: Tue, 6 Oct 2026 12:22:07 +0800 Subject: [PATCH 45/58] Test interrupted reconstruction of MemTable WAL statistics Recovery is not an atomic operation and can itself be interrupted. A later Open must be able to rebuild the same totals even if the mapped aggregates were left cleared or only partly accumulated. Exercise interruptions after reset, count, byte and node aggregation with live and exited writers and with multiple WALs. This protects the separation between permanent cumulative counters and replaceable recovery output. --- db/db_cspp_crash_safe_test.cc | 42 +++++++++++++++++++++++++++++++++++ 1 file changed, 42 insertions(+) diff --git a/db/db_cspp_crash_safe_test.cc b/db/db_cspp_crash_safe_test.cc index 75a5eb8ed..6799ca965 100644 --- a/db/db_cspp_crash_safe_test.cc +++ b/db/db_cspp_crash_safe_test.cc @@ -2523,6 +2523,36 @@ TEST_F(DBCsppCrashSafeTest, PersistentWalStatsNormalConversion) { } } +TEST_F(CrashChild, DISABLED_InterruptedWalStatsRecovery) { + Options options = PersistentStatsOptions(dbname_, arg_); + std::string active; + ASSERT_OK(ReadFileToString(options.env, dbname_ + "/stats-active", &active)); + uint64_t file_number; + FileType file_type; + ASSERT_TRUE(ParseFileName(active.substr(active.find_last_of('/') + 1), &file_number, &file_type)); + InternalKeyComparator icmp(options.comparator); + ImmutableOptions ioptions(options); + MutableCFOptions moptions(options); + IntTblPropCollectorFactories collectors; + TableBuilderOptions tbo(ioptions, moptions, icmp, &collectors, + options.compression, options.compression_opts, 0, + kDefaultColumnFamilyName, 0); + uint64_t next_blob = 800000; + tbo.generate_file_no = [&] { return next_blob++; }; + tbo.add_blob_file = [](BlobFileAddition) {}; + const char* hooks[] = {"MemTableStats::RecoverWals:AfterReset", + "MemTableStats::RecoverWals:AfterCount", + "MemTableStats::RecoverWals:AfterBytes", + "MemTableStats::RecoverWals:AfterNode"}; + SyncPoint::GetInstance()->SetCallBack(hooks[arg_[3] - '0'], [](void*) { ::_exit(42); }); + SyncPoint::GetInstance()->EnableProcessing(); + FileMetaData meta; + meta.fd = FileDescriptor(file_number, 0, 0); + meta.fd.largest_seqno = kMaxSequenceNumber; + ASSERT_OK(options.memtable_factory->RecoverCrashSafeMemTableToSST(active, &meta, tbo)); + ::_exit(1); // Each selected hook must interrupt actual WAL aggregation. +} + TEST_F(DBCsppCrashSafeTest, PersistentMemTableStatsSurviveWriterLifetime) { Close(); for (const char* config : {"CNL", "CNE", "CPL", "CPE", "CSL", "CSE", @@ -2534,6 +2564,13 @@ TEST_F(DBCsppCrashSafeTest, PersistentMemTableStatsSurviveWriterLifetime) { ASSERT_EQ(RunCrashChild(dbname_, "PersistentMemTableStats", config), 42); std::string active; ASSERT_OK(ReadFileToString(env_, dbname_ + "/stats-active", &active)); + if (config[1] == 'P' && (config[2] == 'T' || config[2] == 'E')) { + // Two writers: live after threshold reporting, or already exited. + for (char boundary : {'0', '1', '2', '3'}) { + ASSERT_EQ(RunCrashChild(dbname_, "InterruptedWalStatsRecovery", + std::string(config) + boundary), 42); + } + } const uint64_t batches = config[2] == 'B' || config[2] == 'A' ? 1 : 2; const uint64_t value_size = config[2] == 'T' ? 600 * 1024 : 128; uint64_t file_number; @@ -2788,6 +2825,11 @@ TEST_F(DBCsppCrashSafeTest, LogRefRecoveryMultipleWals) { // The empty rotating CF is now FileMmap too; its registered sources follow // the primary CF's earlier file number in the complete inventory. ASSERT_GE(leftovers.size(), 2U); + ASSERT_OK(WriteStringToFile(env_, leftovers.front(), dbname_ + "/stats-active")); + for (char boundary : {'0', '1', '2', '3'}) { + ASSERT_EQ(RunCrashChild(dbname_, "InterruptedWalStatsRecovery", + std::string(config) + boundary), 42); + } const int fd = ::open(leftovers.front().c_str(), O_RDONLY); ASSERT_GE(fd, 0); uint32_t num_wals = 0; From 042855a46f75771cd8de41bdbbfde3d167bb3cca Mon Sep 17 00:00:00 2001 From: leipeng Date: Tue, 6 Oct 2026 13:18:57 +0800 Subject: [PATCH 46/58] Allow compaction of tables with inexact visible entry counts Crash-safe recovery can retain physical records beyond pubseq. Recovered statistics may count those records, while iterators correctly hide them. Treating this expected difference as corruption prevents otherwise valid recovered tables from being compacted. Keep strict record-count validation for exact tables. A recovered reader can declare that num_entries is not exact for its visible contents; failure to obtain that declaration must not suppress a mismatch. --- db/compaction/compaction_job.cc | 41 ++++++++++- db/db_compaction_test.cc | 7 ++ db/db_cspp_crash_safe_test.cc | 119 ++++++++++++++++++++++++++++++++ table/table_reader.h | 3 + 4 files changed, 169 insertions(+), 1 deletion(-) diff --git a/db/compaction/compaction_job.cc b/db/compaction/compaction_job.cc index 9d0cc0690..25368c852 100644 --- a/db/compaction/compaction_job.cc +++ b/db/compaction/compaction_job.cc @@ -932,7 +932,46 @@ if (stats_) { uint64_t expected = compaction_stats_.stats.num_input_records - num_input_range_del; uint64_t actual = compaction_job_stats_->num_input_records; - if (expected != actual) { + auto can_verify_record_count = [&] { + auto* c = compact_->compaction; + auto* cfd = c->column_family_data(); + auto* tc = cfd->table_cache(); + const auto* cf_options = c->mutable_cf_options(); + const ReadOptions ro(Env::IOActivity::kCompaction); + for (const auto& input : *c->inputs()) { + for (const auto* file : input.files) { + bool supported; + if (auto* reader = file->fd.table_reader) { + TEST_SYNC_POINT("CompactionJob::VerifyRecordCount:PinnedReader"); + supported = reader->IsNumEntriesExact(); + } else { + TEST_SYNC_POINT("CompactionJob::VerifyRecordCount:FindTable"); + TableCache::TypedHandle* handle = nullptr; + Status s = tc->FindTable( + ro, file_options_, cfd->internal_comparator(), *file, + &handle, cf_options->block_protection_bytes_per_key, + cf_options->prefix_extractor); + supported = true; + if (s.ok()) { + supported = tc->GetTableReaderFromHandle(handle)->IsNumEntriesExact(); + tc->ReleaseHandle(handle); + } + TEST_SYNC_POINT_CALLBACK("CompactionJob::VerifyRecordCount:FindTableStatus", &s); + if (!s.ok()) { + return true; // Keep the original mismatch error. + } + } + if (!supported) { + TEST_SYNC_POINT("CompactionJob::VerifyRecordCount:Unsupported"); + return false; + } + } + } + TEST_SYNC_POINT("CompactionJob::VerifyRecordCount:Supported"); + return true; + }; + if (expected != actual && can_verify_record_count()) { + TEST_SYNC_POINT("CompactionJob::VerifyRecordCount:Mismatch"); std::string msg = "Total number of input records: " + std::to_string(expected) + ", but processed " + std::to_string(actual) + " records."; diff --git a/db/db_compaction_test.cc b/db/db_compaction_test.cc index 4975b1cef..d23232a86 100644 --- a/db/db_compaction_test.cc +++ b/db/db_compaction_test.cc @@ -10131,9 +10131,16 @@ TEST_F(DBCompactionTest, VerifyRecordCount) { *(bool*)stop_ptr = true; } }); + int supported = 0, mismatches = 0; + SyncPoint::GetInstance()->SetCallBack( + "CompactionJob::VerifyRecordCount:Supported", [&](void*) { supported++; }); + SyncPoint::GetInstance()->SetCallBack( + "CompactionJob::VerifyRecordCount:Mismatch", [&](void*) { mismatches++; }); SyncPoint::GetInstance()->EnableProcessing(); Status s = db_->CompactRange(CompactRangeOptions(), nullptr, nullptr); + ASSERT_GT(supported, 0); + ASSERT_GT(mismatches, 0); ASSERT_TRUE(s.IsCorruption()); const char* expect = "Compaction number of input keys does not match number of keys " diff --git a/db/db_cspp_crash_safe_test.cc b/db/db_cspp_crash_safe_test.cc index 6799ca965..ec0620633 100644 --- a/db/db_cspp_crash_safe_test.cc +++ b/db/db_cspp_crash_safe_test.cc @@ -23,6 +23,7 @@ #include #include +#include #include #include @@ -1798,6 +1799,124 @@ TEST_F(DBCsppCrashSafeTest, RecoveredTableUsesFileSequenceBound) { } } +TEST_F(CrashChild, DISABLED_RecoveredHiddenRecordsCompact) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (arg_ == "OSL") { + SetupOsl(&options, true); + } + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "keep", "old")); + auto* cfd = static_cast(child_db)->GetVersionSet()->GetColumnFamilySet()->GetDefault(); + // These physical entries are beyond the published sequence and absent in WAL. + ASSERT_OK(cfd->mem()->Add(2, kTypeValue, "keep", "ghost", nullptr)); + ASSERT_OK(cfd->mem()->Add(3, kTypeValue, "ghost", "unpublished", nullptr)); + ::_exit(42); +} + +TEST_F(DBCsppCrashSafeTest, RecoveredHiddenRecordsCompact) { + Close(); + for (const auto& config : + {std::make_pair(false, -1), {false, 12}, {true, -1}, {true, 12}, + {false, 0}}) { + const bool osl = config.first; + const bool fail_lookup = config.second == 0; + SCOPED_TRACE(osl); + SCOPED_TRACE(config.second); + Options options = BaseCrashSafeOptions(dbname_, true, false); + // Capacity is max_open_files - 10; capacity / 4 must be zero to avoid pinning. + options.max_open_files = fail_lookup ? 12 : config.second; + if (osl) { + SetupOsl(&options, true); + } + ASSERT_TRUE(options.compaction_verify_record_count); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "RecoveredHiddenRecordsCompact", osl ? "OSL" : "CSPP"), 42); + // Compaction writes ordinary tables while dispatch reads the recovered format. + SidePluginRepo repo; + repo.Put("default", Options().table_factory); + repo.Put("converted", options.table_factory); + options.table_factory = PluginFactorySP::AcquirePlugin( + "Dispatch", {{"default", "$default"}}, repo); + DispatcherTableBackPatch(options.table_factory.get(), repo); + ASSERT_OK(TryReopen(options)); + TablePropertiesCollection properties; + ASSERT_OK(db_->GetPropertiesOfAllTables(&properties)); + uint64_t physical_entries = 0; + for (const auto& property : properties) { + physical_entries += property.second->num_entries; + } + ASSERT_EQ(physical_entries, 3U); + auto check_visible = [&] { + std::unique_ptr it(db_->NewIterator(ReadOptions())); + it->SeekToFirst(); + ASSERT_TRUE(it->Valid()); + ASSERT_EQ(it->key().ToString(), "keep"); + ASSERT_EQ(it->value().ToString(), "old"); + it->Next(); + ASSERT_FALSE(it->Valid()); + ASSERT_OK(it->status()); + ASSERT_EQ(Get("ghost"), "NOT_FOUND"); + }; + check_visible(); + // Overlap prevents a trivial move, forcing the record-count verification path. + ASSERT_OK(Put("keep", "old")); + ASSERT_OK(Flush()); + ASSERT_EQ(CountL0(db_), 2); + auto* cfd = dbfull()->GetVersionSet()->GetColumnFamilySet()->GetDefault(); + for (const auto* file : cfd->current()->storage_info()->LevelFiles(0)) { + ASSERT_EQ(file->fd.table_reader != nullptr, config.second == -1); + if (config.second != -1) { + TableCache::Evict(dbfull()->TEST_table_cache(), file->fd.GetNumber()); + } + } + std::atomic unsupported{0}, find_table{0}, pinned_reader{0}, mismatch{0}; + auto* sync = SyncPoint::GetInstance(); + sync->SetCallBack("CompactionJob::VerifyRecordCount:Unsupported", + [&](void*) { unsupported++; }); + sync->SetCallBack("CompactionJob::VerifyRecordCount:FindTable", + [&](void*) { find_table++; }); + sync->SetCallBack("CompactionJob::VerifyRecordCount:PinnedReader", + [&](void*) { pinned_reader++; }); + sync->SetCallBack("CompactionJob::VerifyRecordCount:Mismatch", [&](void*) { mismatch++; }); + if (fail_lookup) { + sync->SetCallBack("CompactionJob::VerifyRecordCount:FindTableStatus", + [](void* arg) { + *static_cast(arg) = Status::IOError("injected reader lookup failure"); + }); + } + sync->EnableProcessing(); + Status compact = db_->CompactRange(CompactRangeOptions(), nullptr, nullptr); + sync->DisableProcessing(); + sync->ClearCallBack("CompactionJob::VerifyRecordCount:Unsupported"); + sync->ClearCallBack("CompactionJob::VerifyRecordCount:FindTable"); + sync->ClearCallBack("CompactionJob::VerifyRecordCount:PinnedReader"); + sync->ClearCallBack("CompactionJob::VerifyRecordCount:FindTableStatus"); + sync->ClearCallBack("CompactionJob::VerifyRecordCount:Mismatch"); + if (fail_lookup) { + ASSERT_TRUE(compact.IsCorruption()); + ASSERT_NE(std::strstr(compact.getState(), "Compaction number of input keys"), nullptr); + ASSERT_EQ(unsupported.load(), 0); + ASSERT_GT(mismatch.load(), 0); + ASSERT_GT(find_table.load(), 0); + Close(); + continue; + } + ASSERT_OK(compact); + ASSERT_GT(unsupported.load(), 0); + ASSERT_EQ(mismatch.load(), 0); + if (config.second == -1) { + ASSERT_EQ(find_table.load(), 0); + ASSERT_GT(pinned_reader.load(), 0); + } else { + ASSERT_GT(find_table.load(), 0); + } + ASSERT_EQ(CountL0(db_), 0); + check_visible(); + Close(); + } +} + TEST_F(DBCsppCrashSafeTest, RecoveredTableIteratorUsesFileSequenceBound) { Close(); for (bool osl : {false, true}) { diff --git a/table/table_reader.h b/table/table_reader.h index 8db7ade1d..3884a2e13 100644 --- a/table/table_reader.h +++ b/table/table_reader.h @@ -111,6 +111,9 @@ class TableReader : public CacheAlignedNewDelete { virtual std::shared_ptr GetTableProperties() const = 0; + // Whether num_entries is exact for the table's visible contents. + virtual bool IsNumEntriesExact() const { return true; } + // Prepare work that can be done before the real Get() virtual void Prepare(const Slice& /*target*/) {} virtual void PreparePIK(const ParsedInternalKey& pik) { From 0668b031bbc71beb5c6b5757622dcc075f751784 Mon Sep 17 00:00:00 2001 From: rockeet Date: Wed, 7 Oct 2026 17:23:27 +0800 Subject: [PATCH 47/58] Test repeated avoid-flush close cycles after crash-safe recovery Crash-safe recovery converts the leftovers that a previous avoid-flush close retained, and retires their inventory entries. A DB configured with avoid_flush_during_shutdown therefore repeats that cycle on every restart, and the second cycle depends on recovery leaving a freshly registered MemTable behind. Extend the existing avoid-flush case with a second close and reopen. A recovered MemTable left unregistered would silently degrade that open to a full WAL replay: no error, no data loss, just the end of crash-safe recovery. --- db/db_cspp_crash_safe_test.cc | 33 +++++++++++++++++++++++++++++++++ 1 file changed, 33 insertions(+) diff --git a/db/db_cspp_crash_safe_test.cc b/db/db_cspp_crash_safe_test.cc index ec0620633..fd5fede28 100644 --- a/db/db_cspp_crash_safe_test.cc +++ b/db/db_cspp_crash_safe_test.cc @@ -3205,6 +3205,39 @@ TEST_F(DBCsppCrashSafeTest, AvoidFlushDuringShutdownKeepsRegisteredMemTable) { Close(); for (const auto& path : ListLeftovers(options, dbname_)) ASSERT_OK(env_->FileExists(path)); + + // Second cycle: the reopen above converted the leftovers into L0 and + // retired their inventory entries. Closing again with + // avoid_flush_during_shutdown must leave a freshly registered MemTable + // behind, so that this open converts it too. Were the retired inventory or + // the post-recovery MemTable left unregistered, the cycle would silently + // degrade to a full WAL replay: no error, no data loss, just the end of + // crash-safe recovery for this DB. + options.avoid_flush_during_shutdown = true; + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k2", "v2")); + auto* cfd2 = dbfull()->GetVersionSet()->GetColumnFamilySet()->GetDefault(); + ASSERT_NE(cfd2->mem(), nullptr); + ASSERT_TRUE(cfd2->mem()->IsFileRegistered()); + const auto before_second_close = ListLeftovers(options, dbname_); + ASSERT_EQ(before_second_close.size(), registered.size()); + Close(); + ASSERT_EQ(ListLeftovers(options, dbname_), before_second_close); + for (const auto& path : before_second_close) + ASSERT_OK(env_->FileExists(path)); + ASSERT_OK(env_->FileExists(CrashSafePubSeqFileName(dbname_))); + converted.store(0); + SyncPoint::GetInstance()->SetCallBack( + "MemTableRep::ConvertToSST:After", [&](void*) { ++converted; }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(options)); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_EQ(converted.load(), 1); + ASSERT_EQ(Get("k"), "v"); + ASSERT_EQ(Get("k2"), "v2"); + ASSERT_EQ(CountL0(db_), 2); + Close(); } } From dd3030d05bc5f725112363912a8c30fd185c87c2 Mon Sep 17 00:00:00 2001 From: rockeet Date: Wed, 7 Oct 2026 17:30:14 +0800 Subject: [PATCH 48/58] Add file-backed MemTable reps to memtablerep_bench 1. A cspp rep configured with convert_to_sst=kFileMmap is created through the overload that takes the memtable file path; the plain overload leaves the path empty and cannot build one. Add a flag to select that overload, with the path resolved against the factory's chroot_dir. 2. Report ApproximateMemoryUsage after each benchmark, next to the arena's, so the mmap-backed working set is visible without guessing from bytes per key. The rep is not null there: a benchmark list starting with readrandom or readseq dereferences it inside Run() first. --- memtable/memtablerep_bench.cc | 24 ++++++++++++++++++++++++ 1 file changed, 24 insertions(+) diff --git a/memtable/memtablerep_bench.cc b/memtable/memtablerep_bench.cc index 72680b2a9..2f880ef35 100644 --- a/memtable/memtablerep_bench.cc +++ b/memtable/memtablerep_bench.cc @@ -68,6 +68,15 @@ DEFINE_string(memtablerep, "skiplist", "\t cspp:{\"mem_cap\":\"16G\"} or\n" "\t OffsetSkipList:{\"mem_cap\":\"16G\"}"); +// A file-backed memtable (cspp with convert_to_sst=kFileMmap in the json +// above) is created through the overload that takes the memtable file path, +// which is chroot_dir prefixed by the factory, so the json must set +// chroot_dir to an existing directory. LogRef modes are not supported: this +// bench has no WAL, so m_ref_to_wal stays kNoLogRef. +DEFINE_string(memtable_file_path, "", + "If non-empty, create the memtable rep through the file-backed " + "form, the path is relative to the json's chroot_dir"); + DEFINE_int64(bucket_count, 1000000, "bucket_count parameter to pass into NewHashSkiplistRepFactory or " "NewHashLinkListRepFactory"); @@ -720,6 +729,13 @@ int main(int argc, char** argv) { uint64_t sequence; auto createMemtableRep = [&] { sequence = 0; + if (!FLAGS_memtable_file_path.empty()) { + ROCKSDB_NAMESPACE::MutableCFOptions mcfopt(options); + return factory->CreateMemTableRep(FLAGS_memtable_file_path, mcfopt, + key_comp, &arena, + options.prefix_extractor.get(), + options.info_log.get(), 0); + } return factory->CreateMemTableRep(key_comp, &arena, options.prefix_extractor.get(), options.info_log.get()); @@ -784,6 +800,14 @@ int main(int argc, char** argv) { } std::cout << "Running " << name.ToString() << std::endl; benchmark->Run(); + // approximate memory of the rep itself (arena/mmap), i.e. the working + // set the benchmark just walked + size_t usage = memtablerep->ApproximateMemoryUsage(); + std::cout << "ApproximateMemoryUsage: " << usage << " bytes (" + << usage / 1048576.0 << " MiB)" + << ", arena: " << arena.ApproximateMemoryUsage() << " bytes (" + << arena.ApproximateMemoryUsage() / 1048576.0 << " MiB)" + << std::endl; } return 0; From ac3d063555302fddca00dee51d7c7161cbcaa196 Mon Sep 17 00:00:00 2001 From: leipeng Date: Wed, 7 Oct 2026 19:08:55 +0800 Subject: [PATCH 49/58] Improve MemTableRep benchmark key encoding and random-read measurements Random key generation can obscure the cost of MemTable lookups, and sampling with replacement does not visit every key. Keep the sampled workload while adding an independently shuffled full-key workload, so lookup cost and full-range coverage can both be measured. Use big-endian integer keys so bytewise key order matches numeric order. --- memtable/memtablerep_bench.cc | 51 ++++++++++++++++++++++++++++++----- 1 file changed, 45 insertions(+), 6 deletions(-) diff --git a/memtable/memtablerep_bench.cc b/memtable/memtablerep_bench.cc index 2f880ef35..c6569a325 100644 --- a/memtable/memtablerep_bench.cc +++ b/memtable/memtablerep_bench.cc @@ -48,6 +48,7 @@ DEFINE_string(benchmarks, "fillrandom", "\tfillrandom -- write N random values\n" "\tfillseq -- write N values in sequential order\n" "\treadrandom -- read N values in random order\n" + "\treaduniqrand -- read all keys in pre-shuffled order\n" "\treadseq -- scan the DB\n" "\treadwrite -- 1 thread writes while N - 1 threads " "do random\n" @@ -148,6 +149,10 @@ bool g_is_topling_memtab = false; namespace ROCKSDB_NAMESPACE { namespace { +inline void EncodeBigEndian64(char* dst, uint64_t value) { + unaligned_save(dst, NativeOfBigEndian64(value)); +} + struct CallbackVerifyArgs { bool found; bool needs_user_key_cmp; @@ -213,6 +218,9 @@ class KeyGenerator { return std::numeric_limits::max(); } + uint64_t NextUniqRand(uint64_t& next) { return values_[next++]; } + bool IsUniqRand() const { return mode_ == UNIQUE_RANDOM; } + private: Random64* rand_; WriteMode mode_; @@ -262,7 +270,7 @@ class FillBenchmarkThread : public BenchmarkThread { auto internal_key_size = 16; uint64_t key = key_gen_->Next(); char key_buf[8]; // user key - EncodeFixed64(key_buf, key); + EncodeBigEndian64(key_buf, key); uint64_t tag = ++(*sequence_); Slice ukey(key_buf, sizeof(key_buf)); Slice value = generator_.Generate(FLAGS_item_size); @@ -292,7 +300,7 @@ class FillBenchmarkThread : public BenchmarkThread { assert(buf != nullptr); char* p = EncodeVarint32(buf, internal_key_size); auto key = key_gen_->Next(); - EncodeFixed64(p, key); + EncodeBigEndian64(p, key); p += 8; EncodeFixed64(p, ++(*sequence_)); p += 8; @@ -358,13 +366,18 @@ class ConcurrentFillBenchmarkThread : public FillBenchmarkThread { class ReadBenchmarkThread : public BenchmarkThread { ReadOptions read_opt_; + uint64_t uniq_idx_; + bool is_uniqrand_; bool needs_user_key_cmp_; public: ReadBenchmarkThread(MemTableRep* table, KeyGenerator* key_gen, uint64_t* bytes_written, uint64_t* bytes_read, + uint64_t uniq_idx, uint64_t* sequence, uint64_t num_ops, uint64_t* read_hits) : BenchmarkThread(table, key_gen, bytes_written, bytes_read, sequence, num_ops, read_hits) { + uniq_idx_ = uniq_idx; + is_uniqrand_ = key_gen->IsUniqRand(); if (FLAGS_enable_zero_copy) { read_opt_.StartPin(); } @@ -394,8 +407,8 @@ class ReadBenchmarkThread : public BenchmarkThread { void ReadOne() { char user_key[sizeof(uint64_t)]; - auto key = key_gen_->Next(); - EncodeFixed64(user_key, key); + auto key = is_uniqrand_ ? key_gen_->NextUniqRand(uniq_idx_) : key_gen_->Next(); + EncodeBigEndian64(user_key, key); LookupKey lookup_key(Slice(user_key, sizeof(user_key)), *sequence_); InternalKeyComparator internal_key_comp(BytewiseComparator()); CallbackVerifyArgs verify_args; @@ -461,7 +474,7 @@ class ConcurrentReadBenchmarkThread : public ReadBenchmarkThread { uint64_t* sequence, uint64_t num_ops, uint64_t* read_hits, std::atomic_int* threads_done) - : ReadBenchmarkThread(table, key_gen, bytes_written, bytes_read, sequence, + : ReadBenchmarkThread(table, key_gen, bytes_written, bytes_read, 0, sequence, num_ops, read_hits) { threads_done_ = threads_done; } @@ -502,6 +515,8 @@ class SeqConcurrentReadBenchmarkThread : public SeqReadBenchmarkThread { class Benchmark { public: + double random_time = 0; + explicit Benchmark(MemTableRep* table, KeyGenerator* key_gen, uint64_t* sequence, uint32_t num_threads) : table_(table), @@ -519,6 +534,7 @@ class Benchmark { StopWatchNano timer(SystemClock::Default().get(), true); RunThreads(&threads, &bytes_written, &bytes_read, true, &read_hits); auto elapsed_time = static_cast(timer.ElapsedNanos() / 1000); + elapsed_time -= random_time; std::cout << "Elapsed time: " << static_cast(elapsed_time) << " us" << std::endl; @@ -584,9 +600,15 @@ class ReadBenchmark : public Benchmark { uint64_t* bytes_read, bool /*write*/, uint64_t* read_hits) override { for (int i = 0; i < FLAGS_num_threads; ++i) { + uint64_t uniq_idx = i * num_read_ops_per_thread_; + uint64_t num_ops = num_read_ops_per_thread_; + if (i + 1 == FLAGS_num_threads) { + num_ops = FLAGS_num_operations - uniq_idx; + } threads->emplace_back( ReadBenchmarkThread(table_, key_gen_, bytes_written, bytes_read, - sequence_, num_read_ops_per_thread_, read_hits)); + uniq_idx, + sequence_, num_ops, read_hits)); } for (auto& thread : *threads) { thread.join(); @@ -774,6 +796,12 @@ int main(int argc, char** argv) { &rng, ROCKSDB_NAMESPACE::RANDOM, FLAGS_num_operations)); benchmark.reset(new ROCKSDB_NAMESPACE::ReadBenchmark( memtablerep.get(), key_gen.get(), &sequence)); + } else if (name == ROCKSDB_NAMESPACE::Slice("readuniqrand")) { + FLAGS_seed++; + key_gen.reset(new ROCKSDB_NAMESPACE::KeyGenerator( + &rng, ROCKSDB_NAMESPACE::UNIQUE_RANDOM, FLAGS_num_operations)); + benchmark.reset(new ROCKSDB_NAMESPACE::ReadBenchmark( + memtablerep.get(), key_gen.get(), &sequence)); } else if (name == ROCKSDB_NAMESPACE::Slice("readseq")) { key_gen.reset(new ROCKSDB_NAMESPACE::KeyGenerator( &rng, ROCKSDB_NAMESPACE::SEQUENTIAL, FLAGS_num_operations)); @@ -799,6 +827,17 @@ int main(int argc, char** argv) { continue; } std::cout << "Running " << name.ToString() << std::endl; + if (name == ROCKSDB_NAMESPACE::Slice("readrandom")) { + auto num_ops = FLAGS_num_operations / FLAGS_num_threads; + volatile uint64_t sink = 0; + ROCKSDB_NAMESPACE::StopWatchNano timer(ROCKSDB_NAMESPACE::SystemClock::Default().get(), true); + for (int i = 0; i < num_ops; i++) { + sink = key_gen->Next(); + } + benchmark->random_time = static_cast(timer.ElapsedNanos() / 1000); + std::cout << "Random generation time: " << benchmark->random_time << " us" << std::endl; + std::cout << "Random generation sink: " << sink << std::endl; + } benchmark->Run(); // approximate memory of the rep itself (arena/mmap), i.e. the working // set the benchmark just walked From 7951012babd56ec04b61586e24783088ac98ce87 Mon Sep 17 00:00:00 2001 From: leipeng Date: Wed, 7 Oct 2026 19:32:54 +0800 Subject: [PATCH 50/58] Use GetPIK for native MemTableRep benchmark lookups Constructing LookupKey and a comparator for each lookup adds benchmark overhead that CSPP and OSL do not require. Exercise their native GetPIK path so the measurement better reflects lookup cost, while retaining the comparison-based path for other MemTable representations. --- memtable/memtablerep_bench.cc | 17 +++++++++++++++++ 1 file changed, 17 insertions(+) diff --git a/memtable/memtablerep_bench.cc b/memtable/memtablerep_bench.cc index c6569a325..196ed21e4 100644 --- a/memtable/memtablerep_bench.cc +++ b/memtable/memtablerep_bench.cc @@ -405,7 +405,24 @@ class ReadBenchmarkThread : public BenchmarkThread { return false; } + static bool callbackPIK(void* arg, const MemTableRep::KeyValuePair&) { + auto self = static_cast(arg); + (*self->bytes_read_) += VarintLength(16) + 16 + FLAGS_item_size; + (*self->read_hits_)++; + return false; + } + void ReadOnePIK() { + char user_key[sizeof(uint64_t)]; + auto key = is_uniqrand_ ? key_gen_->NextUniqRand(uniq_idx_) : key_gen_->Next(); + EncodeBigEndian64(user_key, key); + ParsedInternalKey pik(Slice(user_key, sizeof(user_key)), *sequence_, kValueTypeForSeek); + table_->GetPIK(read_opt_, pik, this, callbackPIK); + } + void ReadOne() { + if (!needs_user_key_cmp_) { + return ReadOnePIK(); + } char user_key[sizeof(uint64_t)]; auto key = is_uniqrand_ ? key_gen_->NextUniqRand(uniq_idx_) : key_gen_->Next(); EncodeBigEndian64(user_key, key); From e51047e1a932ef0c94caf150a60ccb3a1326fbc1 Mon Sep 17 00:00:00 2001 From: leipeng Date: Thu, 8 Oct 2026 00:19:41 +0800 Subject: [PATCH 51/58] Add readreverse to memtablerep_bench Forward and reverse traversal have different costs, especially for skip lists. A separate readreverse workload lets one benchmark invocation measure both directions against the same populated MemTable, avoiding another fill just to change the scan direction. Preserve the existing --reverse behavior. --- memtable/memtablerep_bench.cc | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/memtable/memtablerep_bench.cc b/memtable/memtablerep_bench.cc index 196ed21e4..312ae1361 100644 --- a/memtable/memtablerep_bench.cc +++ b/memtable/memtablerep_bench.cc @@ -50,6 +50,7 @@ DEFINE_string(benchmarks, "fillrandom", "\treadrandom -- read N values in random order\n" "\treaduniqrand -- read all keys in pre-shuffled order\n" "\treadseq -- scan the DB\n" + "\treadreverse -- scan the DB in reverse order\n" "\treadwrite -- 1 thread writes while N - 1 threads " "do random\n" "\t reads\n" @@ -819,7 +820,8 @@ int main(int argc, char** argv) { &rng, ROCKSDB_NAMESPACE::UNIQUE_RANDOM, FLAGS_num_operations)); benchmark.reset(new ROCKSDB_NAMESPACE::ReadBenchmark( memtablerep.get(), key_gen.get(), &sequence)); - } else if (name == ROCKSDB_NAMESPACE::Slice("readseq")) { + } else if (name == ROCKSDB_NAMESPACE::Slice("readseq") || + name == ROCKSDB_NAMESPACE::Slice("readreverse")) { key_gen.reset(new ROCKSDB_NAMESPACE::KeyGenerator( &rng, ROCKSDB_NAMESPACE::SEQUENTIAL, FLAGS_num_operations)); benchmark.reset(new ROCKSDB_NAMESPACE::SeqReadBenchmark(memtablerep.get(), @@ -855,7 +857,10 @@ int main(int argc, char** argv) { std::cout << "Random generation time: " << benchmark->random_time << " us" << std::endl; std::cout << "Random generation sink: " << sink << std::endl; } + const bool saved_reverse = FLAGS_reverse; + FLAGS_reverse = saved_reverse || name == ROCKSDB_NAMESPACE::Slice("readreverse"); benchmark->Run(); + FLAGS_reverse = saved_reverse; // approximate memory of the rep itself (arena/mmap), i.e. the working // set the benchmark just walked size_t usage = memtablerep->ApproximateMemoryUsage(); From e5f69d8e5567b71043824557d66d55f11f1d11d0 Mon Sep 17 00:00:00 2001 From: leipeng Date: Thu, 8 Oct 2026 00:22:24 +0800 Subject: [PATCH 52/58] Update rockside for the crash-safe enterprise benchmark sample Keep the checked-out sample aligned with its file-mapped MemTable setup so process restarts can reuse MemTable data through crash-safe recovery. --- sideplugin/rockside | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/sideplugin/rockside b/sideplugin/rockside index d066c938d..972f48eb7 160000 --- a/sideplugin/rockside +++ b/sideplugin/rockside @@ -1 +1 @@ -Subproject commit d066c938d64b2581407119d74104c10a5d0cc893 +Subproject commit 972f48eb7df345f9c5b46d593fa53197436eccee From ef838f45e13af58741c688a33a9d9ec312d685f1 Mon Sep 17 00:00:00 2001 From: rockeet Date: Fri, 9 Oct 2026 09:44:05 +0800 Subject: [PATCH 53/58] Inline FindFileInRangeTmpl into its callers The key-prefix-cache binary search was emitted as an out-of-line clone (nm showed two FindFileInRangeTmpl symbols) with a full prologue -- two pushes, endbr64, the stack-protector canary, an argument shuffle and a call/ret -- to search one level's 15 files in four integer compares. It runs once per point lookup, so that frame cost more than the search: in perf annotate, the prefix compare itself was 0.23% of the function while its prologue lines summed to ~4.8%. always_inline removes the symbol entirely and drops 30.1 instructions per point read (1457.7 -> 1427.6, identical across runs) on a 1e7-key ToplingZipTable + UintIndex readrandom. Forcing the caller FindFileInRange inline as well changes nothing further -- the compiler already inlined it. --- db/version_set.cc | 1 + 1 file changed, 1 insertion(+) diff --git a/db/version_set.cc b/db/version_set.cc index ceb3b7ebc..afdb0c4f7 100644 --- a/db/version_set.cc +++ b/db/version_set.cc @@ -129,6 +129,7 @@ inline uint64_t HostPrefixCache(const ParsedInternalKey& ikey) { } template +__attribute__((always_inline)) size_t FindFileInRangeTmpl(Cmp cmp, const LevelFilesBrief& brief, const ParsedInternalKey& key, size_t lo, size_t hi) { const uint64_t* pxcache = brief.prefix_cache; From f2c40c3dc9bd2f19e2394ced3665fd9c42fd7308 Mon Sep 17 00:00:00 2001 From: rockeet Date: Fri, 9 Oct 2026 22:11:33 +0800 Subject: [PATCH 54/58] Inline GetContext construction into its callers under LTO Constructing GetContext is a fixed per-Get cost: a 312-byte object, 13 call sites, and it stayed out of line because under LTO a call to a same-DSO symbol with default visibility is a PLT call. An inlining mandate is refused outright in that state: 'error: inlining failed in call to always_inline: function body can be overwritten at link time'. flatten (already on Version::GetInst) cannot help; it only sees bodies in its own TU. So the two attributes go together, and only where both are legal. Both the decision and its reasoning live in one macro, next to ROCKSDB_PRIszt in the platform port headers -- port/port.h is only the dispatcher: port_posix.h always_inline + visibility("hidden"), gated port_win.h empty: the build never sets TOPLINGDB_HAVE_LTO there, and visibility("hidden") has no meaning in PE/COFF #if defined(TOPLINGDB_HAVE_LTO) && !defined(ROCKSDB_UNIT_TEST) GCC defines no macro of its own for -flto, so the Makefile supplies TOPLINGDB_HAVE_LTO under USE_LTO=1. Without it a non-LTO build would not merely skip inlining, it would fail to compile ('function body not available'). The unit-test arm is excluded because six test files construct GetContext directly through the shared library, and MAKE_UNIT_TEST=1 defines ROCKSDB_UNIT_TEST for the whole test build, library objects included. Measured (1e7-key ToplingZipTable + UintIndex, db_bench, zero-copy, TOPLINGDB_GetContext_sampling=kNone, alternating paired rounds): 1427.6 -> 1419.7 insn/op and 621 -> 605 cyc/op (-2.7%, the two arms' ranges do not overlap). Version::GetInst drops from 1081 to 1025 instructions of text. The executed saving is small because inlining removes only the call shell -- call/ret, the prologue with canary, and the argument transport -- while the 56 member-assignment instructions stay: the object escapes to the reader in another TU, so no store is dead. --- Makefile | 5 +++++ port/port_posix.h | 27 +++++++++++++++++++++++++++ port/win/port_win.h | 5 +++++ table/get_context.h | 8 ++++++++ 4 files changed, 45 insertions(+) diff --git a/Makefile b/Makefile index 30f7ce819..830428c77 100644 --- a/Makefile +++ b/Makefile @@ -244,6 +244,11 @@ OPTION_lto := lto-0 ifeq ($(USE_LTO), 1) ifeq (${DEBUG_LEVEL},0) CXXFLAGS += -flto + # Lets headers hand the LTO inliner an explicit mandate where it is + # wanted (see table/get_context.h): GCC defines no macro of its own + # for -flto, and `always_inline` on a cross-TU body is a hard error + # without it. + CXXFLAGS += -DTOPLINGDB_HAVE_LTO LDFLAGS += -flto=auto -fuse-linker-plugin OPTION_lto := lto-$(if $(filter 1,${USE_LTO}),1,0) endif diff --git a/port/port_posix.h b/port/port_posix.h index ea0259a5c..31e61d484 100644 --- a/port/port_posix.h +++ b/port/port_posix.h @@ -24,6 +24,33 @@ #define __declspec(S) +// Force-inline a function whose body lives in another translation unit, and +// hide it, so that LTO may inline it into its callers. Both halves are needed +// and both have to be gated: +// +// * `visibility("hidden")` clears the semantic-interposition rule. Without +// it GCC refuses an inlining mandate outright -- "error: inlining failed +// in call to 'always_inline': function body can be overwritten at link +// time". Hidden on its own is only permission, not an order: measured to +// leave the call out of line, worth ~1 instruction. +// * `always_inline` is the order. It must be absent when the body is not +// reachable, so this macro is empty outside LTO -- without -flto GCC fails +// the build with "function body not available" rather than quietly not +// inlining. TOPLINGDB_HAVE_LTO is set by the Makefile under USE_LTO=1. +// * Unit-test builds are excluded. They construct such objects directly and +// link the shared library, so the symbol has to stay exported; +// MAKE_UNIT_TEST=1 defines ROCKSDB_UNIT_TEST for the whole test build, +// library objects included. +// +// `flatten` is not an alternative: it only flattens calls whose body is +// visible in the calling TU. +#if defined(TOPLINGDB_HAVE_LTO) && !defined(ROCKSDB_UNIT_TEST) +#define TOPLING_LTO_HIDDEN_INLINE \ + __attribute__((visibility("hidden"))) __attribute__((always_inline)) +#else +#define TOPLING_LTO_HIDDEN_INLINE +#endif + #undef PLATFORM_IS_LITTLE_ENDIAN #if defined(OS_MACOSX) #include diff --git a/port/win/port_win.h b/port/win/port_win.h index 387c08122..7d1cb5425 100644 --- a/port/win/port_win.h +++ b/port/win/port_win.h @@ -57,6 +57,11 @@ using ssize_t = SSIZE_T; #define ROCKSDB_PRIszt "Iu" #endif +// No LTO force-inlining here: the build never sets TOPLINGDB_HAVE_LTO on this +// platform, and `visibility("hidden")` has no meaning in PE/COFF. See +// port/port_posix.h for the POSIX definition and the reasoning. +#define TOPLING_LTO_HIDDEN_INLINE + #ifdef _MSC_VER #define __attribute__(A) diff --git a/table/get_context.h b/table/get_context.h index dbea262d6..1c1de41e1 100644 --- a/table/get_context.h +++ b/table/get_context.h @@ -101,6 +101,10 @@ class GetContext { // and false if all the merge operands associated with user_key has to be // returned. Id do_merge=false then all the merge operands are stored in // merge_context and they are never merged. The value pointer is untouched. + // Constructing this is a fixed per-Get cost. See TOPLING_LTO_HIDDEN_INLINE + // (port/port.h): the inlining mandate needs hiding to be legal, and both + // are absent outside LTO and in unit-test builds. + TOPLING_LTO_HIDDEN_INLINE GetContext(const Comparator* ucmp, const MergeOperator* merge_operator, Logger* logger, Statistics* statistics, GetState init_state, const Slice& user_key, PinnableSlice* value, @@ -111,6 +115,10 @@ class GetContext { PinnedIteratorsManager* _pinned_iters_mgr = nullptr, ReadCallback* callback = nullptr, bool* is_blob_index = nullptr, uint64_t tracing_get_id = 0, BlobFetcher* blob_fetcher = nullptr); + // Constructing this is a fixed per-Get cost. See TOPLING_LTO_HIDDEN_INLINE + // (port/port.h): the inlining mandate needs hiding to be legal, and both + // are absent outside LTO and in unit-test builds. + TOPLING_LTO_HIDDEN_INLINE GetContext(const Comparator* ucmp, const MergeOperator* merge_operator, Logger* logger, Statistics* statistics, GetState init_state, const Slice& user_key, PinnableSlice* value, From e4ff3d7696fdc6b57f91c5f5591bec39773fe09a Mon Sep 17 00:00:00 2001 From: rockeet Date: Fri, 9 Oct 2026 22:29:37 +0800 Subject: [PATCH 55/58] Cut two per-read steps out of db_bench's read loops SelectDBWithCfh draws a random number on every operation, but a single-DB run always resolves to db_ -- the draw only advances the per-thread RNG stream and adds an mt19937_64 step plus a 64-bit modulo to every read. The key generator draws from that same stream, so removing it does change which keys a run reads; that is deliberate here (the benchmark exists to measure the DB, not its own bookkeeping) and it means absolute numbers taken before this change cannot be mixed with numbers taken after. ReadRandom also asks for the default column family handle inside the loop: a virtual call that, from the shared library, goes through the PLT, for a handle that never changes. Cache it per DB instead. Measured on readrandom over a 1e7-key ToplingZipTable (zero-copy, TOPLINGDB_GetContext_sampling=kNone): 1396.7 -> 1356.5 instructions per read, -40.2 (-2.88%). The draw accounts for all of it -- the CF-handle cache alone measures exactly 0.0, kept only because it is correct and free. db_bench's share of the instruction profile drops 24.2% -> 21.3%. Both arms of the LookupKey A/B lose the same 40, so their difference is unchanged (+86.0, re-measured on all three configurations). --- tools/db_bench_tool.cc | 23 ++++++++++++++++++++++- 1 file changed, 22 insertions(+), 1 deletion(-) diff --git a/tools/db_bench_tool.cc b/tools/db_bench_tool.cc index 9970dc08b..ccb50267b 100644 --- a/tools/db_bench_tool.cc +++ b/tools/db_bench_tool.cc @@ -5191,9 +5191,30 @@ class Benchmark { DB* SelectDB(ThreadState* thread) { return SelectDBWithCfh(thread)->db; } DBWithColumnFamilies* SelectDBWithCfh(ThreadState* thread) { + // A single-DB run always resolves to db_, so the draw below would only + // advance the per-thread RNG stream (and add an mt19937_64 step plus a + // modulo to every read) for nothing. The key generator draws from the same + // stream, so this does change which keys a run reads -- that is the point: + // the benchmark exists to measure the DB, not its own bookkeeping. + if (LIKELY(db_.db != nullptr)) { + return &db_; + } return SelectDBWithCfh(thread->rand.Next()); } + // The default column family handle never changes for a given DB, but asking + // for it is a virtual call (and, in the shared library, goes through the + // PLT). Per-operation loops should not pay for that on every key. + ColumnFamilyHandle* DefaultCfh(DBWithColumnFamilies* d) { + if (UNLIKELY(d != cfh_cache_db_)) { + cfh_cache_db_ = d; + cfh_cache_ = d->db->DefaultColumnFamily(); + } + return cfh_cache_; + } + DBWithColumnFamilies* cfh_cache_db_ = nullptr; + ColumnFamilyHandle* cfh_cache_ = nullptr; + DBWithColumnFamilies* SelectDBWithCfh(uint64_t rand_int) { if (db_.db != nullptr) { return &db_; @@ -6308,7 +6329,7 @@ class Benchmark { if (FLAGS_num_column_families > 1) { cfh = db_with_cfh->GetCfh(key_rand); } else { - cfh = db_with_cfh->db->DefaultColumnFamily(); + cfh = DefaultCfh(db_with_cfh); } if (read_operands_) { for (size_t i = 0; i < pinnable_vals.size(); ++i) { From f6002d5550b3bfd2efb2b00650f21293e23404f2 Mon Sep 17 00:00:00 2001 From: rockeet Date: Sun, 11 Oct 2026 11:53:05 +0800 Subject: [PATCH 56/58] Fence the cold branches flatten was inlining into the hot functions ROCKSDB_FLATTEN is a transitive inlining root, and under -fno-semantic-interposition + LTO it pulls the entire callee tree in. On the point-get, seek and compaction entry points that tree is mostly cold, so the functions carried a blob of code that no workload executes: DBImpl::GetImpl 1,179,692 B CompactionIterator::PrepareOutput 797,070 DBIter::PrevWithKey 409,284 DBIter::Seek 68,795 Drop flatten from those four, and fence the other direction too: put terark_no_inline on the cold helpers that a hot function calls directly, so the same transitive pull cannot come in through that door. Only direct calls are worth fencing -- MemTable::Get calls NewRangeTombstoneIteratorInternal and MaxCoveringTombstoneSeqnum itself, and DBIter::Next calls ReverseToForward itself. A mark on something reached only through a vtable (MemTableRep::GetPIK, TableReader::GetPIK) or through an already out-of-line aggregator (FragmentedRangeTombstoneIterator::SplitBySnapshot) provably changes nothing, so those are not added. before after DBImpl::GetImpl 1,179,692 3,891 CompactionIterator::PrepareOutput 797,070 909 DBIter::PrevWithKey 409,284 86 DBIter::Seek 68,795 2,182 MemTable::Get 68,771 3,482 DBIter::Next 2,284 853 shared object 517,968,920 452,397,256 Measured against a HEAD build of this tree with the same flags (fsi + LTO, DEBUG_LEVEL=0), alternating paired rounds: readrandom on a 1e7-key ToplingZipTable with zero-copy and TOPLINGDB_GetContext_sampling=kNone goes 1358.55 -> 1327.60 instructions per read (-30.95, -2.28%) and 615 -> 565 cycles. readseq loses 414.13 -> 402.18 instructions, but its cycles go up about 5% against a flat instruction count, which the instruction count does not explain; left alone. --- db/compaction/compaction_iterator.cc | 1 - db/db_impl/db_impl.cc | 1 - db/db_iter.cc | 4 ++-- db/memtable.cc | 1 + db/range_tombstone_fragmenter.cc | 1 + 5 files changed, 4 insertions(+), 4 deletions(-) diff --git a/db/compaction/compaction_iterator.cc b/db/compaction/compaction_iterator.cc index aeaa7d3f3..66c80d434 100644 --- a/db/compaction/compaction_iterator.cc +++ b/db/compaction/compaction_iterator.cc @@ -1269,7 +1269,6 @@ void CompactionIterator::DecideOutputLevel() { } } -ROCKSDB_FLATTEN void CompactionIterator::PrepareOutput() { if (Valid()) { if (LIKELY(!is_range_del_)) { diff --git a/db/db_impl/db_impl.cc b/db/db_impl/db_impl.cc index 53e0c0e52..cca68b6be 100644 --- a/db/db_impl/db_impl.cc +++ b/db/db_impl/db_impl.cc @@ -2302,7 +2302,6 @@ bool DBImpl::ShouldReferenceSuperVersion(const MergeContext& merge_context) { merge_context.GetOperands().size(); } -ROCKSDB_FLATTEN Status DBImpl::GetImpl(const ReadOptions& read_options, const Slice& key, GetImplOptions& get_impl_options) { #if defined(ROCKSDB_UNIT_TEST) diff --git a/db/db_iter.cc b/db/db_iter.cc index 2df4c8348..8a3ba9983 100644 --- a/db/db_iter.cc +++ b/db/db_iter.cc @@ -307,7 +307,6 @@ Slice DBIter::NextWithKey() { return Slice(nullptr, 0); } -ROCKSDB_FLATTEN Slice DBIter::PrevWithKey() { return IterPrevWithKeyImpl(this); } bool DBIter::SetBlobValueIfNeeded(const Slice& user_key, @@ -1166,6 +1165,7 @@ void DBIter::Prev() { } } +terark_no_inline bool DBIter::ReverseToForward() { assert(iter_.status().ok()); @@ -1208,6 +1208,7 @@ bool DBIter::ReverseToForward() { } // Move iter_ to the key before saved_key_. +terark_no_inline bool DBIter::ReverseToBackward() { assert(iter_.status().ok()); @@ -1939,7 +1940,6 @@ void DBIter::SetSavedKeyToSeekForPrevTarget(const Slice& target) { } } -ROCKSDB_FLATTEN void DBIter::Seek(const Slice& target) { PERF_COUNTER_ADD(iter_seek_count, 1); PERF_CPU_TIMER_GUARD(iter_seek_cpu_nanos, clock_); diff --git a/db/memtable.cc b/db/memtable.cc index 028854950..4167c0102 100644 --- a/db/memtable.cc +++ b/db/memtable.cc @@ -610,6 +610,7 @@ FragmentedRangeTombstoneIterator* MemTable::NewRangeTombstoneIterator( immutable_memtable); } +terark_no_inline FragmentedRangeTombstoneIterator* MemTable::NewRangeTombstoneIteratorInternal( const ReadOptions& read_options, SequenceNumber read_seq, bool immutable_memtable) { diff --git a/db/range_tombstone_fragmenter.cc b/db/range_tombstone_fragmenter.cc index 7e7cedeca..07c257105 100644 --- a/db/range_tombstone_fragmenter.cc +++ b/db/range_tombstone_fragmenter.cc @@ -467,6 +467,7 @@ bool FragmentedRangeTombstoneIterator::Valid() const { return tombstones_ != nullptr && pos_ != tombstones_->end(); } +terark_no_inline SequenceNumber FragmentedRangeTombstoneIterator::MaxCoveringTombstoneSeqnum( const Slice& target_user_key) { SeekToCoveringTombstone(target_user_key); From 77be64d240a5068df5157d8c7fccdbf30936a4f9 Mon Sep 17 00:00:00 2001 From: rockeet Date: Sun, 11 Oct 2026 11:53:05 +0800 Subject: [PATCH 57/58] Take flatten off the two mutually recursive BaseDeltaIterator steps Advance and UpdateCurrent call each other, so flatten on both made each function inline the other's whole tree: both came out at ~291 KB, near duplicates of one another. BaseDeltaIterator::Advance 291,267 -> 265 BaseDeltaIterator::UpdateCurrent 291,000 -> 2,599 Sizes are from a HEAD build of this tree with the same flags (fsi + LTO, DEBUG_LEVEL=0). No db_bench workload exercises the WBWI/transaction delta iterator, so there is no instruction-count measurement for this one. --- .../write_batch_with_index/write_batch_with_index_internal.cc | 2 -- 1 file changed, 2 deletions(-) diff --git a/utilities/write_batch_with_index/write_batch_with_index_internal.cc b/utilities/write_batch_with_index/write_batch_with_index_internal.cc index dedc4b186..72411b8f4 100644 --- a/utilities/write_batch_with_index/write_batch_with_index_internal.cc +++ b/utilities/write_batch_with_index/write_batch_with_index_internal.cc @@ -293,7 +293,6 @@ void BaseDeltaIterator::AssertInvariants() { #endif } -ROCKSDB_FLATTEN void BaseDeltaIterator::Advance(bool const_forward) { if (UNLIKELY(equal_keys_)) { assert(BaseValid() && DeltaValid()); @@ -498,7 +497,6 @@ struct BDI_VirtualCmpNoTS { const Comparator* cmp; }; -ROCKSDB_FLATTEN void BaseDeltaIterator::UpdateCurrent(bool const_forward) { if (0 == opt_cmp_type_) UpdateCurrentTpl(const_forward, BDI_BytewiseCmpNoTS()); From a61476bfa8896eaf8abc1b32ef8e0adfc04e1ff0 Mon Sep 17 00:00:00 2001 From: rockeet Date: Sun, 11 Oct 2026 13:46:42 +0800 Subject: [PATCH 58/58] Make the build assume its own symbols are not interposable -fno-semantic-interposition was only ever arriving through the hand-edited, gitignored make_config.mk, so a fresh clone built a library whose every same-DSO call went through the PLT and whose LTO inliner could not assume anything about a same-DSO definition. Nothing in this DSO is meant to be interposed -- it is the implementation, not a layer something else overrides -- so the flag belongs in PLATFORM_CXXFLAGS unconditionally, next to the -fno-builtin-memcmp filter-out. Only the compile line matters: the flag rides in each object's recorded LTO options, which is what codegen reads at link time. A three-TU probe shows an LDFLAGS-only build still emits `call foo@plt`, while compile-line-only and both-sides builds are byte-identical. 3 alternating paired rounds, 2e7 reads, current tree: readrandom 1349.77 -> 1327.63 insn/op (-1.64%, ranges disjoint) and ~601 -> ~565 cyc/op. readseq instructions are flat (401.4 -> 402.2) but cycles rise 124 -> 135. This reverses the earlier "net loss" verdict, which was measured before the flatten removals: what amplified the flag then was the flatten explosions. --- Makefile | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/Makefile b/Makefile index 830428c77..9ab650052 100644 --- a/Makefile +++ b/Makefile @@ -148,6 +148,13 @@ endif include make_config.mk PLATFORM_CCFLAGS := $(filter-out -fno-builtin-memcmp, ${PLATFORM_CCFLAGS}) PLATFORM_CXXFLAGS := $(filter-out -fno-builtin-memcmp, ${PLATFORM_CXXFLAGS}) +# Nothing in this library is meant to be interposed -- it is the implementation, +# not a layer someone else overrides. Under default interposition semantics a +# call to a same-DSO definition still has to go through the PLT anyway, and +# under -flto it also keeps the inliner from using the whole-program view. The +# flag has to be on the compile line; adding it to LDFLAGS alone does nothing +# (measured). +PLATFORM_CXXFLAGS += -fno-semantic-interposition # defined in make_config.mk ROCKSDB_FULL_VERSION := ${ROCKSDB_MAJOR}.${ROCKSDB_MINOR}.${ROCKSDB_PATCH}