diff --git a/.github/scripts/bench_logs_to_pages.py b/.github/scripts/bench_logs_to_pages.py index 235294797e..e890d1d1c6 100644 --- a/.github/scripts/bench_logs_to_pages.py +++ b/.github/scripts/bench_logs_to_pages.py @@ -60,6 +60,7 @@ YAML_USED_NAMES = ( "db_bench-fillrandom.yaml", "db_bench-fillseq.yaml", + "db_bench-fillseq-cspp.yaml", ) ENGINE_LABELS = { "zipkeyonly": "ToplingDB zipkeyonly", @@ -137,7 +138,7 @@ def format_iec(num_bytes: int) -> str: return f"{n:.1f}{units[idx]}" -SHM_WORKLOADS = ("fillrandom", "fillseq") +SHM_WORKLOADS = ("fillrandom", "fillseq", "fillseq-cspp") SHM_WORKLOAD_LABELS = SHM_SUITE_LABELS @@ -160,7 +161,14 @@ def load_shm_usages(eng_dir: Path) -> Dict[str, Optional[Dict[str, int]]]: return out -RSS_WORKLOADS = ("fillrandom", "fillseq", "fillrandom-omit", "fillseq-omit") +RSS_WORKLOADS = ( + "fillrandom", + "fillseq", + "fillrandom-omit", + "fillseq-omit", + "fillseq-cspp", + "fillseq-cspp-omit", +) def parse_rss_usage(text: str) -> Optional[int]: @@ -240,6 +248,10 @@ def _bytes(eng: str, wl: str, key: str) -> Optional[int]: rows_html = [] for wl in SHM_WORKLOADS: + if wl == "fillseq-cspp" and all( + _bytes(e, wl, "allocated_bytes") is None for e in ENGINES + ): + continue cells = [f"{html.escape(SHM_WORKLOAD_LABELS.get(wl, wl))}"] for e in ENGINES: b = _bytes(e, wl, "allocated_bytes") @@ -633,6 +645,26 @@ def build_db_bench_compare( LAZY_ENGINES = ("zipkeyonly", "zipkeyvalue", "rocksdb-v8.10") +def _has_topling_fillseq_cspp(engines: Dict[str, Any]) -> bool: + return any( + bool((engines.get(e) or {}).get("db_bench_fillseq_cspp")) + for e in TOPLING_ENGINES + ) + + +def _db_bench_by_engine( + engines: Dict[str, Any], + *, + topling_key: str, + rocks_key: str = "db_bench", +) -> Dict[str, List[Dict[str, str]]]: + out: Dict[str, List[Dict[str, str]]] = {} + for e in ENGINES: + key = topling_key if e in TOPLING_ENGINES else rocks_key + out[e] = (engines.get(e) or {}).get(key) or [] + return out + + def _hl(text: str, kind: str) -> str: """Color a short phrase: kind is 'faster' (green) or 'slower' (red).""" return f'{html.escape(text)}' @@ -762,11 +794,51 @@ def _cost_ratio_cell(baseline: Optional[float], subject: Optional[float]) -> str ) _CSPP_METRICS_LOW = ( "Elapsed time", - "write us/op", - "read us/op", ) +def _memtablerep_elapsed_display( + mmap: Dict[str, str], bench: str, elapsed_raw: str +) -> str: + """Append write/read us/op onto the Elapsed time cell.""" + write_us = mmap.get(f"{bench}|write us/op") + read_us = mmap.get(f"{bench}|read us/op") + extras: List[str] = [] + if write_us and read_us: + extras.append(f"write {write_us} us/op") + extras.append(f"read {read_us} us/op") + elif write_us: + extras.append(f"{write_us} us/op") + elif read_us: + extras.append(f"{read_us} us/op") + if not extras: + return elapsed_raw or "—" + base = elapsed_raw or "—" + return f"{base} ({', '.join(extras)})" + + +def _fold_memtablerep_usop(rows: List[Dict[str, str]]) -> List[Dict[str, str]]: + """Drop standalone us/op rows; fold them into Elapsed time.""" + mmap = _metric_map(rows) + out: List[Dict[str, str]] = [] + for row in rows: + metric = row["metric"] + if metric in ("write us/op", "read us/op"): + continue + if metric == "Elapsed time": + out.append( + { + **row, + "value": _memtablerep_elapsed_display( + mmap, row["benchmark"], row["value"] + ), + } + ) + else: + out.append(row) + return out + + def build_memtablerep_compare( cspp_rows: List[Dict[str, str]], skiplist_topling: List[Dict[str, str]], @@ -806,6 +878,11 @@ def build_memtablerep_compare( _metric_number(c_raw), _metric_number(r_raw), ) + if metric == "Elapsed time": + o_raw = _memtablerep_elapsed_display(offset_skiplist, bench, o_raw) + c_raw = _memtablerep_elapsed_display(cspp, bench, c_raw) + t_raw = _memtablerep_elapsed_display(skip_t, bench, t_raw) + r_raw = _memtablerep_elapsed_display(skip_r, bench, r_raw) if metric in _CSPP_METRICS_HIGH: offset_skiplist_ratio = _throughput_ratio_cell(r_n, o_n) cspp_ratio = _throughput_ratio_cell(r_n, c_n) @@ -860,13 +937,22 @@ def _load_engine_logs(log_root: Path) -> Dict[str, Dict[str, Any]]: ) omit_fr_rows: List[Dict[str, str]] = [] omit_fs_rows: List[Dict[str, str]] = [] + omit_fs_cspp_rows: List[Dict[str, str]] = [] + fs_cspp_path = eng_dir / "db_bench-fillseq-cspp.log" + fs_cspp_rows: List[Dict[str, str]] = [] + if fs_cspp_path.is_file(): + fs_cspp_rows = parse_db_bench( + fs_cspp_path.read_text(encoding="utf-8", errors="replace") + ) if eng == "rocksdb-v8.10": # Reuse readseq×3 from the main fill* suites (no separate omit/scan pass). omit_fr_rows = _readseq_rows(fr_rows) omit_fs_rows = _readseq_rows(db_rows) + omit_fs_cspp_rows = omit_fs_rows else: omit_fr = eng_dir / "db_bench-fillrandom-omit.log" omit_fs = eng_dir / "db_bench-fillseq-omit.log" + omit_fs_cspp = eng_dir / "db_bench-fillseq-cspp-omit.log" if omit_fr.is_file(): omit_fr_rows = parse_db_bench( omit_fr.read_text(encoding="utf-8", errors="replace") @@ -875,6 +961,10 @@ def _load_engine_logs(log_root: Path) -> Dict[str, Dict[str, Any]]: omit_fs_rows = parse_db_bench( omit_fs.read_text(encoding="utf-8", errors="replace") ) + if omit_fs_cspp.is_file(): + omit_fs_cspp_rows = parse_db_bench( + omit_fs_cspp.read_text(encoding="utf-8", errors="replace") + ) skiplist_rows: List[Dict[str, str]] = [] cspp_rows: List[Dict[str, str]] = [] offset_skiplist_rows: List[Dict[str, str]] = [] @@ -893,8 +983,10 @@ def _load_engine_logs(log_root: Path) -> Dict[str, Dict[str, Any]]: result[eng] = { "db_bench": db_rows, "db_bench_fillrandom": fr_rows, + "db_bench_fillseq_cspp": fs_cspp_rows, "db_bench_omit_fillrandom": omit_fr_rows, "db_bench_omit_fillseq": omit_fs_rows, + "db_bench_omit_fillseq_cspp": omit_fs_cspp_rows, "memtablerep_skiplist": skiplist_rows, "memtablerep_cspp": cspp_rows, "memtablerep_OffsetSkipList": offset_skiplist_rows, @@ -1117,6 +1209,15 @@ def _build_per_engine_details(engines_data: Dict[str, Any]) -> str: detail_parts.append( _table(db_bench_detail_keys, data["db_bench"], db_bench_detail_keys) ) + if data.get("db_bench_fillseq_cspp"): + detail_parts.append("

db_bench (fillseq suite, CSPP)

") + detail_parts.append( + _table( + db_bench_detail_keys, + data["db_bench_fillseq_cspp"], + db_bench_detail_keys, + ) + ) if eng in TOPLING_ENGINES: if data.get("db_bench_omit_fillrandom"): detail_parts.append( @@ -1140,12 +1241,23 @@ def _build_per_engine_details(engines_data: Dict[str, Any]) -> str: db_bench_detail_keys, ) ) + if data.get("db_bench_omit_fillseq_cspp"): + detail_parts.append( + "

db_bench omit lazy-load (fillseq CSPP DB)

" + ) + detail_parts.append( + _table( + db_bench_detail_keys, + data["db_bench_omit_fillseq_cspp"], + db_bench_detail_keys, + ) + ) if data.get("memtablerep_skiplist"): detail_parts.append("

memtablerep_bench (skiplist)

") detail_parts.append( _table( ["benchmark", "metric", "value"], - data["memtablerep_skiplist"], + _fold_memtablerep_usop(data["memtablerep_skiplist"]), ["benchmark", "metric", "value"], ) ) @@ -1154,7 +1266,7 @@ def _build_per_engine_details(engines_data: Dict[str, Any]) -> str: detail_parts.append( _table( ["benchmark", "metric", "value"], - data["memtablerep_cspp"], + _fold_memtablerep_usop(data["memtablerep_cspp"]), ["benchmark", "metric", "value"], ) ) @@ -1165,7 +1277,7 @@ def _build_per_engine_details(engines_data: Dict[str, Any]) -> str: detail_parts.append( _table( ["benchmark", "metric", "value"], - data["memtablerep_OffsetSkipList"], + _fold_memtablerep_usop(data["memtablerep_OffsetSkipList"]), ["benchmark", "metric", "value"], ) ) @@ -1208,22 +1320,30 @@ def emit(args: argparse.Namespace) -> None: "db_bench-fillrandom.log", "db_bench-fillrandom-omit.log", "db_bench-fillseq-omit.log", + "db_bench-fillseq-cspp.log", + "db_bench-fillseq-cspp-omit.log", "memtablerep_bench-skiplist.log", "memtablerep_bench-cspp.log", "memtablerep_bench-OffsetSkipList.log", "shm_usage.txt", "shm_usage-fillrandom.txt", "shm_usage-fillseq.txt", + "shm_usage-fillseq-cspp.txt", "rss_usage-fillrandom.txt", "rss_usage-fillseq.txt", "rss_usage-fillrandom-omit.txt", "rss_usage-fillseq-omit.txt", + "rss_usage-fillseq-cspp.txt", + "rss_usage-fillseq-cspp-omit.txt", "statm_series-fillrandom.txt", "statm_series-fillseq.txt", + "statm_series-fillseq-cspp.txt", "time-fillrandom.txt", "time-fillseq.txt", "time-fillrandom-omit.txt", "time-fillseq-omit.txt", + "time-fillseq-cspp.txt", + "time-fillseq-cspp-omit.txt", "bench_settings.txt", *YAML_USED_NAMES, "engine-meta.json", @@ -1360,12 +1480,18 @@ def emit(args: argparse.Namespace) -> None: "db_bench_fillrandom": engines_data.get(eng, {}).get( "db_bench_fillrandom", [] ), + "db_bench_fillseq_cspp": engines_data.get(eng, {}).get( + "db_bench_fillseq_cspp", [] + ), "db_bench_omit_fillrandom": engines_data.get(eng, {}).get( "db_bench_omit_fillrandom", [] ), "db_bench_omit_fillseq": engines_data.get(eng, {}).get( "db_bench_omit_fillseq", [] ), + "db_bench_omit_fillseq_cspp": engines_data.get(eng, {}).get( + "db_bench_omit_fillseq_cspp", [] + ), "shm_usage": engines_data.get(eng, {}).get("shm_usage") or {wl: None for wl in SHM_WORKLOADS}, "rss_usage": rss_by_eng.get(eng) @@ -1457,12 +1583,17 @@ def _render_latest_section( eng_rss_raw = engines.get(e, {}).get("rss_usage") or {} rss_data[e] = {wl: v for wl, v in eng_rss_raw.items()} if e in ROCKSDB_ENGINES: - for src, dst in (("fillrandom", "fillrandom-omit"), ("fillseq", "fillseq-omit")): + for src, dst in ( + ("fillrandom", "fillrandom-omit"), + ("fillseq", "fillseq-omit"), + ("fillseq-cspp", "fillseq-cspp-omit"), + ): if rss_data[e].get(dst) is None and rss_data[e].get(src) is not None: rss_data[e][dst] = rss_data[e][src] if ( rss_data[e].get("fillrandom-omit") is not None or rss_data[e].get("fillseq-omit") is not None + or rss_data[e].get("fillseq-cspp-omit") is not None ): rss_derived_engines.add(e) if pages_root is not None: @@ -1484,6 +1615,26 @@ def _render_latest_section( omit_fs_table = build_lazy_load_compare( {e: engines.get(e, {}).get("db_bench_omit_fillseq") or [] for e in LAZY_ENGINES} ) + fs_cspp_compare = "" + omit_fs_cspp_block = "" + if _has_topling_fillseq_cspp(engines): + fs_cspp_compare = ( + "

Comparison: db_bench fillseq suite (CSPP) (perf)

\n" + '

Same as fillrandom, except ToplingDB fillseq uses CSPP. ' + "RocksDB fillseq benefits from shortcuts: trivial_move on " + "non-overlapping SSTs; refit level skips zstd on L6: faster, " + "larger size. Seqno-zeroing compact still runs.

\n" + f"{build_db_bench_compare(_db_bench_by_engine(engines, topling_key='db_bench_fillseq_cspp'))}" + ) + omit_fs_cspp_block = ( + "

scan-omit-value on data from fillseq (CSPP)

\n" + + build_lazy_load_compare( + { + e: engines.get(e, {}).get("db_bench_omit_fillseq_cspp") or [] + for e in LAZY_ENGINES + } + ) + ) t_eng = engines.get("zipkeyonly") or {} r_eng = engines.get("rocksdb-v8.10") or {} @@ -1533,12 +1684,14 @@ def _render_latest_section(

Comparison: db_bench fillseq suite (perf)

Same as fillrandom, except ToplingDB fillseq uses OffsetSkipList (fillrandom still uses CSPP). RocksDB fillseq benefits from shortcuts: trivial_move on non-overlapping SSTs; refit level skips zstd on L6: faster, larger size. Seqno-zeroing compact still runs.

{db_compare_fs} + {fs_cspp_compare}

Lazy load demo (scan; RocksDB v8.10 baseline)

zipkey* needs an extra omit pass: scan_omit_key/value enables lazy value load (no real value load). RocksDB has no lazy load, so the baseline is readseq×3 already present in the main fill* suite (no extra pass). RocksDB nextwithkey cells are =readseq. master omitted here (v8.10 is the stronger RocksDB baseline). {_color_sign()}.

scan-omit-value on data from fillrandom

{omit_fr_table}

scan-omit-value on data from fillseq

{omit_fs_table} + {omit_fs_cspp_block}

memtablerep_bench: OffsetSkipList and CSPP vs skiplist

Focus: {_hl('OffsetSkipList / CSPP (ToplingDB)', 'faster')} vs skiplist. Baseline = RocksDB v8.10 skiplist. {_color_sign()}.

{memtablerep_compare} diff --git a/.github/scripts/bench_pages_common.py b/.github/scripts/bench_pages_common.py index 0504eec037..6d7d74ca8d 100644 --- a/.github/scripts/bench_pages_common.py +++ b/.github/scripts/bench_pages_common.py @@ -76,6 +76,7 @@ def stage_window_rss_bytes( SUITE_READRANDOM = ( ("fillrandom", "db_bench_fillrandom", "fillrandom-readrandom"), ("fillseq", "db_bench", "fillseq-readrandom"), + ("fillseq-cspp", "db_bench_fillseq_cspp", "fillseq-cspp-readrandom"), ) @@ -104,6 +105,7 @@ def attach_suite_readrandom_rss( SHM_SUITE_LABELS = { "fillrandom": "fillrandom suite", "fillseq": "fillseq suite", + "fillseq-cspp": "fillseq suite (CSPP)", } RSS_WORKLOAD_ORDER = ( "fillrandom", @@ -112,6 +114,9 @@ def attach_suite_readrandom_rss( "fillseq", "fillseq-readrandom", "fillseq-omit", + "fillseq-cspp", + "fillseq-cspp-readrandom", + "fillseq-cspp-omit", ) RSS_WORKLOAD_LABELS = { "fillrandom": "fillrandom suite peak", @@ -120,6 +125,9 @@ def attach_suite_readrandom_rss( "fillseq-readrandom": "fillseq suite readrandom", "fillrandom-omit": "fillrandom scan-omit-value", "fillseq-omit": "fillseq scan-omit-value", + "fillseq-cspp": "fillseq suite (CSPP) peak", + "fillseq-cspp-readrandom": "fillseq suite (CSPP) readrandom", + "fillseq-cspp-omit": "fillseq scan-omit-value (CSPP)", } RSS_WORKLOAD_TIPS = { "fillrandom-readrandom": ( @@ -128,6 +136,9 @@ def attach_suite_readrandom_rss( "fillseq-readrandom": ( "peak RSS during the readrandom stage of the fillseq suite" ), + "fillseq-cspp-readrandom": ( + "peak RSS during the readrandom stage of the fillseq suite (CSPP)" + ), "fillrandom-omit": ( "restart process with reuse db data of fillrandom, " "scan without access value, benefited by lazy load value (ToplingDB feature)" @@ -136,6 +147,10 @@ def attach_suite_readrandom_rss( "restart process with reuse db data of fillseq, " "scan without access value, benefited by lazy load value (ToplingDB feature)" ), + "fillseq-cspp-omit": ( + "restart process with reuse db data of fillseq (CSPP), " + "scan without access value, benefited by lazy load value (ToplingDB feature)" + ), } @@ -445,6 +460,8 @@ def combine_db_bench_logs(engine_raw: Path) -> None: "db_bench-fillrandom-omit.log", "db_bench.log", "db_bench-fillseq-omit.log", + "db_bench-fillseq-cspp.log", + "db_bench-fillseq-cspp-omit.log", ) chunks = [ (engine_raw / name).read_bytes().rstrip(b"\n") @@ -504,6 +521,7 @@ def build_rss_svg_section( for suite, bench_key in [ ("fillrandom", "db_bench_fillrandom"), ("fillseq", "db_bench"), + ("fillseq-cspp", "db_bench_fillseq_cspp"), ]: series_path = eng_dir / f"statm_series-{suite}.txt" if not series_path.is_file(): diff --git a/.github/scripts/test_bench_rss_series.py b/.github/scripts/test_bench_rss_series.py index 88b3275ac2..c8188705ea 100755 --- a/.github/scripts/test_bench_rss_series.py +++ b/.github/scripts/test_bench_rss_series.py @@ -301,6 +301,97 @@ def check_pages_contract(mod, variant: str) -> None: assert "dcompact bench →" in home assert "offloads most CPU and memory cost" in home assert "not RocksDB CompactionService" not in home + assert "Comparison: db_bench fillseq suite (CSPP)" not in home + assert "fillseq suite (CSPP)" not in home + assert "db_bench (fillseq suite, CSPP)" not in result_html + + +def check_fillseq_cspp_pages(mod) -> None: + """ToplingDB CSPP fillseq is a full twin of the OffsetSkipList fillseq suite.""" + with tempfile.TemporaryDirectory() as tmp: + tmp_path = Path(tmp) + log_root = tmp_path / "logs" + emit_out = tmp_path / "emit" + site = tmp_path / "site" + _write_min_logs(log_root) + bench_body = ( + _DB_BENCH_LINE + + "readrandom : 1.0 micros/op 1000 ops/sec 1.0 seconds " + "1000 operations; x\n" + ) + for eng in ("zipkeyonly", "zipkeyvalue"): + eng_dir = log_root / eng + (eng_dir / "db_bench-fillseq-omit.log").write_text( + "$ fillseq-omit\n" + "nextwithkey : 1.0 micros/op 1000 ops/sec 1.0 seconds " + "1000 operations; x\n", + encoding="utf-8", + ) + (eng_dir / "db_bench-fillseq-cspp.log").write_text( + "$ fillseq-cspp\n" + bench_body, encoding="utf-8" + ) + (eng_dir / "db_bench-fillseq-cspp-omit.log").write_text( + "$ fillseq-cspp-omit\n" + "nextwithkey : 1.0 micros/op 1000 ops/sec 1.0 seconds " + "1000 operations; x\n", + encoding="utf-8", + ) + (eng_dir / "statm_series-fillseq-cspp.txt").write_text( + _STATM_SERIES, encoding="utf-8" + ) + (eng_dir / "shm_usage-fillseq-cspp.txt").write_text( + "apparent_bytes=1000\nallocated_bytes=2000\n", encoding="utf-8" + ) + (eng_dir / "rss_usage-fillseq-cspp.txt").write_text( + "max_rss_bytes=4096\n", encoding="utf-8" + ) + (eng_dir / "rss_usage-fillseq-cspp-omit.txt").write_text( + "max_rss_bytes=2048\n", encoding="utf-8" + ) + emit_args = argparse.Namespace( + variant="plain", + run_id="cspp-fillseq", + log_root=str(log_root), + engine_meta_root=None, + actions_run_url="", + out=str(emit_out), + ) + mod.emit(emit_args) + run_dirs = list((emit_out / "runs").iterdir()) + assert len(run_dirs) == 1, run_dirs + result_html = (run_dirs[0] / "index.html").read_text(encoding="utf-8") + assert "db_bench (fillseq suite)" in result_html + assert "db_bench (fillseq suite, CSPP)" in result_html + assert "db_bench omit lazy-load (fillseq DB)" in result_html + assert "db_bench omit lazy-load (fillseq CSPP DB)" in result_html + combined = ( + run_dirs[0] / "raw" / "zipkeyonly" / "db_bench-all.log" + ).read_text(encoding="utf-8") + assert combined.index("$ fillseq-omit\n") < combined.index("$ fillseq-cspp\n") + assert combined.index("$ fillseq-cspp\n") < combined.index( + "$ fillseq-cspp-omit\n" + ) + meta = json.loads((emit_out / "run-meta.json").read_text(encoding="utf-8")) + zko_rss = meta["engines"]["zipkeyonly"]["rss_usage"] + assert zko_rss.get("fillseq-readrandom") is not None + assert zko_rss.get("fillseq-cspp-readrandom") is not None + assert zko_rss.get("fillseq-cspp") == 4096 + assert zko_rss.get("fillseq-cspp-omit") == 2048 + mod.merge( + argparse.Namespace( + merge_into=str(site), + from_dir=str(emit_out), + variant="plain", + ) + ) + home = (site / "index.html").read_text(encoding="utf-8") + assert "Comparison: db_bench fillseq suite (perf)" in home + assert "Comparison: db_bench fillseq suite (CSPP) (perf)" in home + assert "scan-omit-value on data from fillseq" in home + assert "scan-omit-value on data from fillseq (CSPP)" in home + assert "fillseq suite (CSPP) peak" in home + assert "fillseq suite (CSPP) readrandom" in home + assert "fillseq scan-omit-value (CSPP)" in home def check_dcompact_home_nav(mod) -> None: @@ -596,6 +687,11 @@ def main() -> int: ("db_bench-fillrandom-omit.log", "$ fillrandom-omit\nomit output\n"), ("db_bench.log", "$ fillseq\nfillseq output\n"), ("db_bench-fillseq-omit.log", "$ fillseq-omit\nomit output\n"), + ("db_bench-fillseq-cspp.log", "$ fillseq-cspp\ncspp output\n"), + ( + "db_bench-fillseq-cspp-omit.log", + "$ fillseq-cspp-omit\ncspp omit output\n", + ), ) for name, content in source_logs: (eng_raw / name).write_text(content, encoding="utf-8") @@ -692,6 +788,7 @@ def main() -> int: assert "TestOS" in html check_readrandom_highlight(mod) check_suite_readrandom_peak(mod) + check_fillseq_cspp_pages(mod) if name == "bench_dcompact_pages": check_dcompact_home_nav(mod) check_dcompact_rss_row_tips(mod) diff --git a/.github/workflows/db_bench-avx512-run.yml b/.github/workflows/db_bench-avx512-run.yml index a11469afd1..9bc697ca6f 100644 --- a/.github/workflows/db_bench-avx512-run.yml +++ b/.github/workflows/db_bench-avx512-run.yml @@ -206,50 +206,59 @@ jobs: record_shm fillrandom rm -rf "$DB_PATH" - # Pass 2: fillseq — OffsetSkipList; prefix 6 zipkeyonly (keep L6). - prepare_db - yaml_fs="${logdir}/db_bench-fillseq.yaml" - python3 .github/scripts/graft_bench_yaml.py \ - --prefix-level-writers 6 zipkeyonly \ - --target-file-size-base 128M \ - --target-file-size-multiplier 1 \ - --memtable-factory '"${offset_skiplist}"' \ - --out "$yaml_fs" \ - "$yaml" - args=( - -json "$yaml_fs" - -num=100000000 - -key_size=8 - -value_size="${VALUE_SIZE}" - -batch_size=1000 - -benchmarks=fillseq,flush,compact,readseq,readseq,readseq,readrandom - -enable_zero_copy - -progress_reports=false - -compact_target_level=6 - ) - echo '$' "$TOPLING/bin/db_bench" "${args[@]}" >"${logdir}/db_bench.log" - /usr/bin/time -f 'max_rss_kb=%M' -o "${logdir}/time-fillseq.txt" -- \ - "$TOPLING/bin/db_bench" "${args[@]}" >>"${logdir}/db_bench.log" 2>&1 - cat "${logdir}/db_bench.log" - args_omit_fs=( - -json "$yaml_fs" - -num=100000000 - -key_size=8 - -value_size="${VALUE_SIZE}" - -batch_size=1000 - -benchmarks=nextwithkey,nextwithkey,nextwithkey,readseq,readseq,readseq - -scan_omit_key -scan_omit_value - -use_existing_db=1 - -enable_zero_copy - -progress_reports=false - ) - echo '$' "$TOPLING/bin/db_bench" "${args_omit_fs[@]}" >"${logdir}/db_bench-fillseq-omit.log" - /usr/bin/time -f 'max_rss_kb=%M' -o "${logdir}/time-fillseq-omit.txt" -- \ - "$TOPLING/bin/db_bench" "${args_omit_fs[@]}" >>"${logdir}/db_bench-fillseq-omit.log" 2>&1 - cat "${logdir}/db_bench-fillseq-omit.log" - record_rss fillseq - record_rss fillseq-omit - record_shm fillseq + # Pass 2: fillseq — OffsetSkipList then CSPP; prefix 6 zipkeyonly (keep L6). + run_topling_fillseq() { + local factory="$1" + local tag="$2" + local yaml_fs="${logdir}/db_bench-${tag}.yaml" + local main_log="${logdir}/db_bench.log" + [ "$tag" = "fillseq" ] || main_log="${logdir}/db_bench-${tag}.log" + local omit_log="${logdir}/db_bench-${tag}-omit.log" + prepare_db + python3 .github/scripts/graft_bench_yaml.py \ + --prefix-level-writers 6 zipkeyonly \ + --target-file-size-base 128M \ + --target-file-size-multiplier 1 \ + --memtable-factory '"${'"${factory}"'}"' \ + --out "$yaml_fs" \ + "$yaml" + args=( + -json "$yaml_fs" + -num=100000000 + -key_size=8 + -value_size="${VALUE_SIZE}" + -batch_size=1000 + -benchmarks=fillseq,flush,compact,readseq,readseq,readseq,readrandom + -enable_zero_copy + -progress_reports=false + -compact_target_level=6 + ) + echo '$' "$TOPLING/bin/db_bench" "${args[@]}" >"$main_log" + /usr/bin/time -f 'max_rss_kb=%M' -o "${logdir}/time-${tag}.txt" -- \ + "$TOPLING/bin/db_bench" "${args[@]}" >>"$main_log" 2>&1 + cat "$main_log" + args_omit_fs=( + -json "$yaml_fs" + -num=100000000 + -key_size=8 + -value_size="${VALUE_SIZE}" + -batch_size=1000 + -benchmarks=nextwithkey,nextwithkey,nextwithkey,readseq,readseq,readseq + -scan_omit_key -scan_omit_value + -use_existing_db=1 + -enable_zero_copy + -progress_reports=false + ) + echo '$' "$TOPLING/bin/db_bench" "${args_omit_fs[@]}" >"$omit_log" + /usr/bin/time -f 'max_rss_kb=%M' -o "${logdir}/time-${tag}-omit.txt" -- \ + "$TOPLING/bin/db_bench" "${args_omit_fs[@]}" >>"$omit_log" 2>&1 + cat "$omit_log" + record_rss "$tag" + record_rss "${tag}-omit" + record_shm "$tag" + } + run_topling_fillseq offset_skiplist fillseq + run_topling_fillseq cspp fillseq-cspp if [ "$run_memtable" = "1" ]; then mt=( -benchmarks=fillrandom,readrandom diff --git a/.github/workflows/db_bench-run.yml b/.github/workflows/db_bench-run.yml index ccc59c0430..fdc6c1cd45 100644 --- a/.github/workflows/db_bench-run.yml +++ b/.github/workflows/db_bench-run.yml @@ -255,54 +255,64 @@ jobs: record_shm fillrandom rm -rf "$DB_PATH" - # Pass 2: fillseq — OffsetSkipList; prefix 6 zipkeyonly (keep L6). - prepare_db - yaml_fs="${logdir}/db_bench-fillseq.yaml" - python3 .github/scripts/graft_bench_yaml.py \ - --prefix-level-writers 6 zipkeyonly \ - --target-file-size-base 128M \ - --target-file-size-multiplier 1 \ - --memtable-factory '"${offset_skiplist}"' \ - --out "$yaml_fs" \ - "$yaml" - args=( - -json "$yaml_fs" - -num="${NUM}" - -key_size=8 - -value_size="${VALUE_SIZE}" - -batch_size=1000 - -benchmarks=fillseq,flush,compact,readseq,readseq,readseq,readrandom - -enable_zero_copy - -progress_reports=false - -report_bench_start_time - -compact_target_level=6 - ) - echo '$' "$TOPLING/bin/db_bench" "${args[@]}" >"${logdir}/db_bench.log" - .github/scripts/run_sample_statm_fdcache.sh "${logdir}/statm_series-fillseq.txt" "${logdir}/time-fillseq.txt" \ - "$TOPLING/bin/db_bench" "${args[@]}" \ - >>"${logdir}/db_bench.log" 2>&1 - cat "${logdir}/db_bench.log" - save_db_log fillseq - args_omit_fs=( - -json "$yaml_fs" - -num="${NUM}" - -key_size=8 - -value_size="${VALUE_SIZE}" - -batch_size=1000 - -benchmarks=nextwithkey,nextwithkey,nextwithkey,readseq,readseq,readseq - -scan_omit_key -scan_omit_value - -use_existing_db=1 - -enable_zero_copy - -progress_reports=false - ) - echo '$' "$TOPLING/bin/db_bench" "${args_omit_fs[@]}" >"${logdir}/db_bench-fillseq-omit.log" - /usr/bin/time -f 'max_rss_kb=%M' -o "${logdir}/time-fillseq-omit.txt" -- \ - "$TOPLING/bin/db_bench" "${args_omit_fs[@]}" >>"${logdir}/db_bench-fillseq-omit.log" 2>&1 - cat "${logdir}/db_bench-fillseq-omit.log" - save_db_log fillseq-omit - record_rss fillseq - record_rss fillseq-omit - record_shm fillseq + # Pass 2: fillseq — OffsetSkipList then CSPP; prefix 6 zipkeyonly (keep L6). + run_topling_fillseq() { + local factory="$1" + local tag="$2" + local yaml_fs="${logdir}/db_bench-${tag}.yaml" + local main_log="${logdir}/db_bench.log" + [ "$tag" = "fillseq" ] || main_log="${logdir}/db_bench-${tag}.log" + local omit_log="${logdir}/db_bench-${tag}-omit.log" + prepare_db + python3 .github/scripts/graft_bench_yaml.py \ + --prefix-level-writers 6 zipkeyonly \ + --target-file-size-base 128M \ + --target-file-size-multiplier 1 \ + --memtable-factory '"${'"${factory}"'}"' \ + --out "$yaml_fs" \ + "$yaml" + args=( + -json "$yaml_fs" + -num="${NUM}" + -key_size=8 + -value_size="${VALUE_SIZE}" + -batch_size=1000 + -benchmarks=fillseq,flush,compact,readseq,readseq,readseq,readrandom + -enable_zero_copy + -progress_reports=false + -report_bench_start_time + -compact_target_level=6 + ) + echo '$' "$TOPLING/bin/db_bench" "${args[@]}" >"$main_log" + .github/scripts/run_sample_statm_fdcache.sh \ + "${logdir}/statm_series-${tag}.txt" "${logdir}/time-${tag}.txt" \ + "$TOPLING/bin/db_bench" "${args[@]}" \ + >>"$main_log" 2>&1 + cat "$main_log" + save_db_log "$tag" + args_omit_fs=( + -json "$yaml_fs" + -num="${NUM}" + -key_size=8 + -value_size="${VALUE_SIZE}" + -batch_size=1000 + -benchmarks=nextwithkey,nextwithkey,nextwithkey,readseq,readseq,readseq + -scan_omit_key -scan_omit_value + -use_existing_db=1 + -enable_zero_copy + -progress_reports=false + ) + echo '$' "$TOPLING/bin/db_bench" "${args_omit_fs[@]}" >"$omit_log" + /usr/bin/time -f 'max_rss_kb=%M' -o "${logdir}/time-${tag}-omit.txt" -- \ + "$TOPLING/bin/db_bench" "${args_omit_fs[@]}" >>"$omit_log" 2>&1 + cat "$omit_log" + save_db_log "${tag}-omit" + record_rss "$tag" + record_rss "${tag}-omit" + record_shm "$tag" + } + run_topling_fillseq offset_skiplist fillseq + run_topling_fillseq cspp fillseq-cspp if [ "$run_memtable" = "1" ]; then mt=( -benchmarks=fillrandom,readrandom diff --git a/AGENTS.md b/AGENTS.md new file mode 100644 index 0000000000..398bb90949 --- /dev/null +++ b/AGENTS.md @@ -0,0 +1,5 @@ +# Formatting + +- Keep each single statement on one line when the complete line, including indentation, fits within 100 columns. This includes calls, declarations, assignments, returns, and `if` / `while` / `for` headers; do not collapse block bodies. +- If that line would exceed 100 columns, wrap it to 80 columns per line, including indentation. Preserve the surrounding continuation-indent style. +- Format only uncommitted added or modified code. Preserve untouched existing code; do not run whole-file formatting. diff --git a/Makefile b/Makefile index 20649fb0fe..9ab6500523 100644 --- a/Makefile +++ b/Makefile @@ -148,6 +148,13 @@ endif include make_config.mk PLATFORM_CCFLAGS := $(filter-out -fno-builtin-memcmp, ${PLATFORM_CCFLAGS}) PLATFORM_CXXFLAGS := $(filter-out -fno-builtin-memcmp, ${PLATFORM_CXXFLAGS}) +# Nothing in this library is meant to be interposed -- it is the implementation, +# not a layer someone else overrides. Under default interposition semantics a +# call to a same-DSO definition still has to go through the PLT anyway, and +# under -flto it also keeps the inliner from using the whole-program view. The +# flag has to be on the compile line; adding it to LDFLAGS alone does nothing +# (measured). +PLATFORM_CXXFLAGS += -fno-semantic-interposition # defined in make_config.mk ROCKSDB_FULL_VERSION := ${ROCKSDB_MAJOR}.${ROCKSDB_MINOR}.${ROCKSDB_PATCH} @@ -244,6 +251,11 @@ OPTION_lto := lto-0 ifeq ($(USE_LTO), 1) ifeq (${DEBUG_LEVEL},0) CXXFLAGS += -flto + # Lets headers hand the LTO inliner an explicit mandate where it is + # wanted (see table/get_context.h): GCC defines no macro of its own + # for -flto, and `always_inline` on a cross-TU body is a hard error + # without it. + CXXFLAGS += -DTOPLINGDB_HAVE_LTO LDFLAGS += -flto=auto -fuse-linker-plugin OPTION_lto := lto-$(if $(filter 1,${USE_LTO}),1,0) endif @@ -451,6 +463,9 @@ ifndef WITH_TOPLING_ROCKS # default 1 WITH_TOPLING_ROCKS := 1 endif +ifeq ($(filter 0,${DEBUG_LEVEL})$(wildcard sideplugin/topling-rocks/src/table/top_patent_algo.cc),) + override WITH_TOPLING_ROCKS := 0 +endif ifeq (${WITH_TOPLING_ROCKS},1) ifneq (,$(wildcard sideplugin/topling-rocks)) @@ -493,6 +508,9 @@ endif # allow override by env or cmd line WITH_CSPP_MEMTABLE ?= 1 +ifeq ($(filter 0,${DEBUG_LEVEL})$(wildcard sideplugin/cspp-memtable/cspp_memtable.cc),) + override WITH_CSPP_MEMTABLE := 0 +endif ifeq (${WITH_CSPP_MEMTABLE}${WITH_TOPLING_ROCKS},10) $(error "When WITH_CSPP_MEMTABLE is 1, WITH_TOPLING_ROCKS must be 1 also") @@ -1883,6 +1901,9 @@ db_bench_rls: $(OBJ_DIR)/tools/db_bench.o $(BENCH_OBJECTS) $(TESTUTIL) $(LIBRARY $(AM_LINK) endif +crash_recover_bench: $(OBJ_DIR)/tools/crash_recover_bench.o $(LIBRARY) + $(AM_LINK) + trace_analyzer: $(OBJ_DIR)/tools/trace_analyzer.o $(ANALYZE_OBJECTS) $(TOOLS_LIBRARY) $(LIBRARY) $(AM_LINK) @@ -2068,6 +2089,9 @@ db_dynamic_level_test: $(OBJ_DIR)/db/db_dynamic_level_test.o $(TEST_LIBRARY) $(L db_flush_test: $(OBJ_DIR)/db/db_flush_test.o $(TEST_LIBRARY) $(LIBRARY) $(AM_LINK) +db_memtable_convert_test: $(OBJ_DIR)/db/db_memtable_convert_test.o $(TEST_LIBRARY) $(LIBRARY) + $(AM_LINK) + db_inplace_update_test: $(OBJ_DIR)/db/db_inplace_update_test.o $(TEST_LIBRARY) $(LIBRARY) $(AM_LINK) @@ -2131,6 +2155,9 @@ db_universal_compaction_test: $(OBJ_DIR)/db/db_universal_compaction_test.o $(TES db_wal_test: $(OBJ_DIR)/db/db_wal_test.o $(TEST_LIBRARY) $(LIBRARY) $(AM_LINK) +db_cspp_crash_safe_test: $(OBJ_DIR)/db/db_cspp_crash_safe_test.o $(TEST_LIBRARY) $(LIBRARY) + $(AM_LINK) + db_io_failure_test: $(OBJ_DIR)/db/db_io_failure_test.o $(TEST_LIBRARY) $(LIBRARY) $(AM_LINK) diff --git a/README-zh_cn.md b/README-zh_cn.md index fc17d6b5a4..699f6a1305 100644 --- a/README-zh_cn.md +++ b/README-zh_cn.md @@ -37,6 +37,11 @@ ToplingDB 兼容 RocksDB API 的同时,增加了很多非常重要的功能与 1. 内置 Prometheus 指标的支持,这是在[内嵌 Http](https://github.com/topling/rockside/wiki/WebView) 中实现的 1. 修复了很多 RocksDB 的 bug,我们已将其中易于合并到 RocksDB 的很多修复与改进给上游 RocksDB 发了 [Pull Request](https://github.com/facebook/rocksdb/pulls?q=is%3Apr+author%3Arockeet) +## 进程崩溃后的恢复 +恢复机制、配置方式及适用边界见 [MemTable Crash-Safe Recovery](https://github.com/topling/rockside/wiki/Crash-Safe-Recovery)。 +底层数据结构为何同时支持读侧无等待与进程崩溃后的读取,见 [读侧无等待与 Crash-Safe 的同构性](https://github.com/topling/rockside/wiki/Wait-Free-Reads-and-Crash-Safe)。 +异常退出后 `DB::Open` 的耗时见 [crash_recover_bench.md](tools/crash_recover_bench.md)。 + ## ToplingDB 云原生数据库服务 1. [MyTopling](https://github.com/topling/mytopling)(MySQL on ToplingDB), [阿里云上的 MyTopling](https://market.aliyun.com/products?k=mytopling) 1. [Todis](https://github.com/topling/todis)(Redis on ToplingDB) diff --git a/README.md b/README.md index 639040abbb..ea287b94cf 100644 --- a/README.md +++ b/README.md @@ -39,6 +39,11 @@ ToplingDB has much more key features than RocksDB: 1. Builtin Prometheus metrics support, this is based on [Embedded Http Server](https://github.com/topling/sideplugin-wiki-en/wiki/WebView) 1. Many bugfixes for RocksDB, a small part of such fixes was [Pull Requested](https://github.com/facebook/rocksdb/pulls?q=is%3Apr+author%3Arockeet) to [upstream RocksDB](https://github.com/facebook/rocksdb) +## Crash-safe recovery +See the [crash-safe recovery guide](https://github.com/topling/sideplugin-wiki-en/wiki/Crash-Safe-Recovery) for the recovery mechanism, configuration, and limitations. +For the underlying data-structure principles, see [The Isomorphism Between Wait-Free Reads and Crash Safety](https://github.com/topling/sideplugin-wiki-en/wiki/Wait-Free-Reads-and-Crash-Safe). +`DB::Open` after an abnormal exit is timed in [crash_recover_bench.md](tools/crash_recover_bench.md). + ## ToplingDB cloud native DB services 1. [MyTopling](https://github.com/topling/mytopling)(MySQL on ToplingDB), [MyTopling on aliyun](https://market.aliyun.com/products?k=mytopling) 1. [Todis](https://github.com/topling/todis)(Redis on ToplingDB) diff --git a/db/column_family.cc b/db/column_family.cc index 2ea7cb8579..f69a82d778 100644 --- a/db/column_family.cc +++ b/db/column_family.cc @@ -40,6 +40,7 @@ #include "util/autovector.h" #include "util/cast_util.h" #include "util/compression.h" +#include namespace ROCKSDB_NAMESPACE { @@ -468,6 +469,15 @@ ColumnFamilyOptions SanitizeOptions(const ImmutableDBOptions& db_options, } #endif + if (result.min_write_buffer_number_to_merge > 1 && + result.memtable_factory->SupportConvertToSST()) { + ROCKS_LOG_WARN(db_options.logger, + "ConvertToSST converts each memtable separately; " + "min_write_buffer_number_to_merge > 1 is incompatible " + "and is sanitized to 1"); + result.min_write_buffer_number_to_merge = 1; + } + return result; } @@ -1164,7 +1174,12 @@ uint64_t ColumnFamilyData::GetLiveSstFilesSize() const { void ColumnFamilyData::PrepareNewMemtableInBackground( const MutableCFOptions& mutable_cf_options) { - #if !defined(ROCKSDB_UNIT_TEST) + bool use_cache = true; + TEST_SYNC_POINT_CALLBACK( + "ColumnFamilyData::PrepareNewMemtableInBackground:UseCache", &use_cache); + if (!use_cache) { + return; + } { std::lock_guard lk(precreated_memtable_mutex_); if (precreated_memtable_list_.full()) { @@ -1172,11 +1187,13 @@ void ColumnFamilyData::PrepareNewMemtableInBackground( return; } } - auto beg = ioptions_.clock->NowNanos(); + auto beg = terark::qtime::now(); + uint64_t number = ioptions_.memtable_factory->SupportCrashSafe() + ? dummy_versions_->version_set()->NewFileNumber() : 0; auto tab = new MemTable(internal_comparator_, ioptions_, mutable_cf_options, - write_buffer_manager_, 0/*earliest_seq*/, id_); - auto end = ioptions_.clock->NowNanos(); - RecordInHistogram(ioptions_.stats, MEMTAB_CONSTRUCT_NANOS, end - beg); + write_buffer_manager_, 0/*earliest_seq*/, id_, number); + auto end = terark::qtime::now(); + RecordInHistogram(ioptions_.stats, MEMTAB_CONSTRUCT_NANOS, (end - beg).ns()); { std::lock_guard lk(precreated_memtable_mutex_); if (LIKELY(!precreated_memtable_list_.full())) { @@ -1191,38 +1208,100 @@ void ColumnFamilyData::PrepareNewMemtableInBackground( "precreated_memtable_list_ is full, discard the newly created memtab"); delete tab; } - #endif } MemTable* ColumnFamilyData::ConstructNewMemtable( const MutableCFOptions& mutable_cf_options, SequenceNumber earliest_seq) { MemTable* tab = nullptr; - #if !defined(ROCKSDB_UNIT_TEST) - { + bool use_cache = true; + TEST_SYNC_POINT_CALLBACK("ColumnFamilyData::ConstructNewMemtable:UseCache", + &use_cache); + if (use_cache) { std::lock_guard lk(precreated_memtable_mutex_); if (!precreated_memtable_list_.empty()) { tab = precreated_memtable_list_.front().release(); precreated_memtable_list_.pop_front(); } } - #endif if (tab) { tab->SetCreationSeq(earliest_seq); tab->SetEarliestSequenceNumber(earliest_seq); } else { - #if !defined(ROCKSDB_UNIT_TEST) - auto beg = ioptions_.clock->NowNanos(); - #endif + auto beg = terark::qtime::now(); + // dummy_versions_ remains alive for the lifetime of this CF, unlike current_. + uint64_t number = ioptions_.memtable_factory->SupportCrashSafe() + ? dummy_versions_->version_set()->NewFileNumber() : 0; tab = new MemTable(internal_comparator_, ioptions_, mutable_cf_options, - write_buffer_manager_, earliest_seq, id_); - #if !defined(ROCKSDB_UNIT_TEST) - auto end = ioptions_.clock->NowNanos(); - RecordInHistogram(ioptions_.stats, MEMTAB_CONSTRUCT_NANOS, end - beg); - #endif + write_buffer_manager_, earliest_seq, id_, number); + auto end = terark::qtime::now(); + RecordInHistogram(ioptions_.stats, MEMTAB_CONSTRUCT_NANOS, (end - beg).ns()); } return tab; } +MemTable* ColumnFamilyData::PeekPrecreatedMemtable() { + std::lock_guard lk(precreated_memtable_mutex_); + return precreated_memtable_list_.empty() + ? nullptr : precreated_memtable_list_.front().get(); +} + +void ColumnFamilyData::ApplyMemTableFileEdit(const VersionEdit& edit) { + ROCKSDB_ASSERT_EQ(edit.GetColumnFamily(), id_); + if (edit.IsColumnFamilyDrop()) { + memtable_files_.clear(); + return; + } + for (uint64_t number : edit.GetMemTableFileDeletions()) { + memtable_files_.erase(number); + } + for (uint64_t number : edit.GetMemTableFileAdditions()) { + memtable_files_.insert(number); + } +} + +void ColumnFamilyData::AddMemTableFileEdits(VersionEdit* edit) { + if (!ioptions_.memtable_crash_safe_recover) return; + std::lock_guard lk(precreated_memtable_mutex_); + auto add = [&](MemTable* mem) { + if (!mem->IsFileRegistered()) { + edit->AddMemTableFile(mem->GetFileNumber()); + edit->SetMemTableFileTracking(); + } + }; + add(mem_); + for (size_t i = 0; i < precreated_memtable_list_.size(); ++i) { + add(precreated_memtable_list_[i].get()); + } +} + +void ColumnFamilyData::PublishRegisteredMemTables() { + if (!ioptions_.memtable_crash_safe_recover) return; + TEST_SYNC_POINT("FlushJob::MemTableCache:BeforePublish"); + { + std::lock_guard lk(precreated_memtable_mutex_); + auto publish = [&](MemTable* mem) { + if (memtable_files_.count(mem->GetFileNumber())) { + mem->MarkFileRegistered(); + } + }; + publish(mem_); + for (size_t i = 0; i < precreated_memtable_list_.size(); ++i) { + publish(precreated_memtable_list_[i].get()); + } + } + TEST_SYNC_POINT("FlushJob::MemTableCache:AfterPublish"); +} + +void ColumnFamilyData::AddMemTableFileNumbers(std::vector* live) { + if (!ioptions_.memtable_factory->SupportCrashSafe()) return; + if (mem_ != nullptr) live->push_back(mem_->GetFileNumber()); + imm_.AddMemTableFileNumbers(live); + std::lock_guard lk(precreated_memtable_mutex_); + for (size_t i = 0; i < precreated_memtable_list_.size(); ++i) { + live->push_back(precreated_memtable_list_[i]->GetFileNumber()); + } +} + void ColumnFamilyData::CreateNewMemtable( const MutableCFOptions& mutable_cf_options, SequenceNumber earliest_seq) { if (mem_ != nullptr) { @@ -1473,6 +1552,12 @@ void ColumnFamilyData::ResetThreadLocalSuperVersions() { Status ColumnFamilyData::ValidateOptions( const DBOptions& db_options, const ColumnFamilyOptions& cf_options) { + if (db_options.memtable_crash_safe_recover && + !cf_options.memtable_factory->SupportCrashSafe()) { + return Status::InvalidArgument( + "memtable_crash_safe_recover requires a FileMmap memtable factory", + cf_options.memtable_factory->Name()); + } Status s; s = CheckCompressionSupported(cf_options); if (s.ok() && db_options.allow_concurrent_memtable_write) { diff --git a/db/column_family.h b/db/column_family.h index 367b94160e..47b5229aaa 100644 --- a/db/column_family.h +++ b/db/column_family.h @@ -373,6 +373,13 @@ class ColumnFamilyData { uint64_t OldestLogToKeep(); void PrepareNewMemtableInBackground(const MutableCFOptions&); + MemTable* PeekPrecreatedMemtable(); + // DB mutex must be held for registration and live-file collection. + const std::set& GetMemTableFiles() const { return memtable_files_; } + void ApplyMemTableFileEdit(const VersionEdit& edit); + void AddMemTableFileEdits(VersionEdit* edit); + void AddMemTableFileNumbers(std::vector* live); + void PublishRegisteredMemTables(); // See Memtable constructor for explanation of earliest_seq param. MemTable* ConstructNewMemtable(const MutableCFOptions& mutable_cf_options, @@ -612,14 +619,13 @@ class ColumnFamilyData { WriteBufferManager* write_buffer_manager_; - #if !defined(ROCKSDB_UNIT_TEST) // precreated_memtable_list_.size() is normally 1 terark::fixed_circular_queue, 4> precreated_memtable_list_; std::mutex precreated_memtable_mutex_; - #endif MemTable* mem_; MemTableList imm_; + std::set memtable_files_; // Protected by DB mutex. SuperVersion* super_version_; // An ordinal representing the current SuperVersion. Updated by diff --git a/db/compaction/compaction_iterator.cc b/db/compaction/compaction_iterator.cc index aeaa7d3f39..66c80d4344 100644 --- a/db/compaction/compaction_iterator.cc +++ b/db/compaction/compaction_iterator.cc @@ -1269,7 +1269,6 @@ void CompactionIterator::DecideOutputLevel() { } } -ROCKSDB_FLATTEN void CompactionIterator::PrepareOutput() { if (Valid()) { if (LIKELY(!is_range_del_)) { diff --git a/db/compaction/compaction_job.cc b/db/compaction/compaction_job.cc index 9d0cc0690e..25368c8523 100644 --- a/db/compaction/compaction_job.cc +++ b/db/compaction/compaction_job.cc @@ -932,7 +932,46 @@ if (stats_) { uint64_t expected = compaction_stats_.stats.num_input_records - num_input_range_del; uint64_t actual = compaction_job_stats_->num_input_records; - if (expected != actual) { + auto can_verify_record_count = [&] { + auto* c = compact_->compaction; + auto* cfd = c->column_family_data(); + auto* tc = cfd->table_cache(); + const auto* cf_options = c->mutable_cf_options(); + const ReadOptions ro(Env::IOActivity::kCompaction); + for (const auto& input : *c->inputs()) { + for (const auto* file : input.files) { + bool supported; + if (auto* reader = file->fd.table_reader) { + TEST_SYNC_POINT("CompactionJob::VerifyRecordCount:PinnedReader"); + supported = reader->IsNumEntriesExact(); + } else { + TEST_SYNC_POINT("CompactionJob::VerifyRecordCount:FindTable"); + TableCache::TypedHandle* handle = nullptr; + Status s = tc->FindTable( + ro, file_options_, cfd->internal_comparator(), *file, + &handle, cf_options->block_protection_bytes_per_key, + cf_options->prefix_extractor); + supported = true; + if (s.ok()) { + supported = tc->GetTableReaderFromHandle(handle)->IsNumEntriesExact(); + tc->ReleaseHandle(handle); + } + TEST_SYNC_POINT_CALLBACK("CompactionJob::VerifyRecordCount:FindTableStatus", &s); + if (!s.ok()) { + return true; // Keep the original mismatch error. + } + } + if (!supported) { + TEST_SYNC_POINT("CompactionJob::VerifyRecordCount:Unsupported"); + return false; + } + } + } + TEST_SYNC_POINT("CompactionJob::VerifyRecordCount:Supported"); + return true; + }; + if (expected != actual && can_verify_record_count()) { + TEST_SYNC_POINT("CompactionJob::VerifyRecordCount:Mismatch"); std::string msg = "Total number of input records: " + std::to_string(expected) + ", but processed " + std::to_string(actual) + " records."; diff --git a/db/db_compaction_test.cc b/db/db_compaction_test.cc index 4975b1cef3..d23232a865 100644 --- a/db/db_compaction_test.cc +++ b/db/db_compaction_test.cc @@ -10131,9 +10131,16 @@ TEST_F(DBCompactionTest, VerifyRecordCount) { *(bool*)stop_ptr = true; } }); + int supported = 0, mismatches = 0; + SyncPoint::GetInstance()->SetCallBack( + "CompactionJob::VerifyRecordCount:Supported", [&](void*) { supported++; }); + SyncPoint::GetInstance()->SetCallBack( + "CompactionJob::VerifyRecordCount:Mismatch", [&](void*) { mismatches++; }); SyncPoint::GetInstance()->EnableProcessing(); Status s = db_->CompactRange(CompactRangeOptions(), nullptr, nullptr); + ASSERT_GT(supported, 0); + ASSERT_GT(mismatches, 0); ASSERT_TRUE(s.IsCorruption()); const char* expect = "Compaction number of input keys does not match number of keys " diff --git a/db/db_cspp_crash_safe_test.cc b/db/db_cspp_crash_safe_test.cc new file mode 100644 index 0000000000..fd5fede282 --- /dev/null +++ b/db/db_cspp_crash_safe_test.cc @@ -0,0 +1,5870 @@ +// Copyright (c) 2026-present, Topling Inc. +// Crash-safe leftover recover: Convert + WAL tail, sync-point injection. + +#include +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include + +#include +#include +#include +#include + +#include "db/column_family.h" +#include "db/db_impl/db_impl.h" +#include "db/db_test_util.h" +#include "db/log_reader.h" +#include "db/log_writer.h" +#include "db/memtable.h" +#include "db/pre_release_callback.h" +#include "db/version_set.h" +#include "file/filename.h" +#include "file/file_util.h" +#include "file/sequence_file_reader.h" +#include "file/writable_file_writer.h" +#include "port/port.h" +#include "port/stack_trace.h" +#include "rocksdb/io_status.h" +#include "rocksdb/convenience.h" +#include "rocksdb/statistics.h" +#include "rocksdb/sst_file_reader.h" +#include "rocksdb/utilities/checkpoint.h" +#include "rocksdb/utilities/transaction_db.h" +#include "rocksdb/wal_filter.h" +#include "table/format.h" +#include "table/get_context.h" +#include "table/sst_file_dumper.h" +#include "table/table_builder.h" +#include "table/table_reader.h" +#include "table/top_table_reader.h" +#include "utilities/merge_operators.h" +#include "utilities/fault_injection_fs.h" +#include "test_util/sync_point.h" +#include "test_util/testutil.h" + +namespace ROCKSDB_NAMESPACE { + +extern MemTableRepFactory* NewCSPPMemTabForPlain(const std::string&); +std::shared_ptr EasyNewMemTableRep(Slice cls, Slice js); + +namespace { + +struct PublishedSeqOnDisk { + uint64_t magic = 0; + uint32_t version = 0; + uint32_t header_size = 0; + uint32_t wal_offset_kind = 0; + uint32_t kind_since_wal = 0; + uint64_t generation = 0; + uint64_t pubseq = 0; + uint64_t wal_number = 0; + uint64_t wal_offset = 0; + uint64_t padding = 0; +}; + +void SetupCspp(Options* options, bool file_mmap) { + const char* js = file_mmap + ? R"({"mem_cap":16777216,"convert_to_sst":"kFileMmap"})" + : R"({"mem_cap":16777216,"convert_to_sst":"kDontConvert"})"; + options->memtable_factory.reset(NewCSPPMemTabForPlain(js)); + const SidePluginRepo repo; + options->table_factory = PluginFactorySP::AcquirePlugin( + "CSPPMemTabTable", json::parse(js), repo); +} + +void SetupOsl(Options* options, bool file_mmap) { + const char* js = file_mmap + ? R"({"mem_cap":16777216,"convert_to_sst":"kFileMmap"})" + : R"({"mem_cap":16777216,"convert_to_sst":"kDontConvert"})"; + options->memtable_factory = EasyNewMemTableRep("OffsetSkipList", js); + const SidePluginRepo repo; + options->table_factory = PluginFactorySP::AcquirePlugin( + "OffsetSkipListTable", json::parse(js), repo); +} + +Options BaseCrashSafeOptions(const std::string& dbname, bool recover, + bool log_index) { + Options options; + options.create_if_missing = true; + options.error_if_exists = false; + options.memtable_crash_safe_recover = recover; + options.memtable_as_log_index = log_index; + options.avoid_flush_during_shutdown = false; + options.avoid_flush_during_recovery = true; + options.disable_auto_compactions = true; + options.write_buffer_size = 64 << 20; + options.max_write_buffer_number = 8; + options.level0_file_num_compaction_trigger = 1 << 20; + options.env = Env::Default(); + options.wal_dir = dbname; + SetupCspp(&options, true); + return options; +} + +std::vector ListLeftovers(const Options& options, + const std::string& dir) { + std::vector leftovers; + Env* env = options.env ? options.env : Env::Default(); + std::string current; + Status s = env->FileExists(CurrentFileName(dir)); + if (s.IsNotFound()) return leftovers; + EXPECT_OK(s); + if (!s.ok()) return leftovers; + s = ReadFileToString(env, CurrentFileName(dir), ¤t); + if (!s.ok()) { + ADD_FAILURE() << s.ToString(); + return leftovers; + } + EXPECT_FALSE(current.empty()); + if (current.empty()) return leftovers; + if (current.back() == '\n') current.pop_back(); + const std::string manifest = dir + "/" + current; + std::unique_ptr file; + s = env->GetFileSystem()->NewSequentialFile(manifest, FileOptions(), &file, + nullptr); + EXPECT_OK(s); + if (!s.ok()) return leftovers; + struct Reporter : log::Reader::Reporter { + void Corruption(size_t, const Status& status) override { + ADD_FAILURE() << status.ToString(); + } + } reporter; + auto input = std::make_unique(std::move(file), manifest); + log::Reader reader(nullptr, std::move(input), &reporter, true, 0); + std::map> registered; + auto apply = [&](const VersionEdit& edit) { + const uint32_t cf = edit.GetColumnFamily(); + if (edit.IsColumnFamilyDrop()) { + registered.erase(cf); + return; + } + for (uint64_t number : edit.GetMemTableFileDeletions()) { + registered[cf].erase(number); + } + for (uint64_t number : edit.GetMemTableFileAdditions()) { + registered[cf].insert(number); + } + }; + AtomicGroupReadBuffer group; + Slice record; + std::string scratch; + while (reader.ReadRecord(&record, &scratch)) { + VersionEdit edit; + s = edit.DecodeFrom(record); + EXPECT_OK(s); + if (!s.ok()) break; + s = group.AddEdit(&edit); + EXPECT_OK(s); + if (!s.ok()) break; + if (!edit.IsInAtomicGroup()) { + apply(edit); + } else if (group.IsFull()) { + for (const auto& member : group.replay_buffer()) apply(member); + group.Clear(); + } + } + const std::string path = !options.cf_paths.empty() + ? options.cf_paths[0].path + : !options.db_paths.empty() + ? options.db_paths[0].path + : dir; + for (const auto& cf : registered) { + for (uint64_t number : cf.second) { + leftovers.push_back(MakeTableFileName(path, number)); + } + } + return leftovers; +} + +bool ReadPublishedSeqFile(const std::string& dbname, PublishedSeqOnDisk* out) { + const std::string path = CrashSafePubSeqFileName(dbname); + int fd = ::open(path.c_str(), O_RDONLY); + if (fd < 0) { + return false; + } + const ssize_t n = ::pread(fd, out, sizeof(*out), 0); + ::close(fd); + return n == static_cast(sizeof(*out)); +} + +bool SetPublishedSeqGeneration(const std::string& dbname, uint64_t g) { + const std::string path = CrashSafePubSeqFileName(dbname); + int fd = ::open(path.c_str(), O_WRONLY); + if (fd < 0) { + return false; + } + const ssize_t n = + ::pwrite(fd, &g, sizeof(g), offsetof(PublishedSeqOnDisk, generation)); + ::close(fd); + return n == static_cast(sizeof(g)); +} + +bool SetPublishedSeqWalOffsetKind(const std::string& dbname, uint32_t kind) { + const std::string path = CrashSafePubSeqFileName(dbname); + int fd = ::open(path.c_str(), O_WRONLY); + if (fd < 0) { + return false; + } + const ssize_t n = + ::pwrite(fd, &kind, sizeof(kind), + offsetof(PublishedSeqOnDisk, wal_offset_kind)); + ::close(fd); + return n == static_cast(sizeof(kind)); +} + +bool ZeroPublishedSeqFile(const std::string& dbname) { + const std::string path = CrashSafePubSeqFileName(dbname); + const std::string zeros(4096, '\0'); + int fd = ::open(path.c_str(), O_WRONLY); + if (fd < 0) { + return false; + } + const ssize_t n = ::pwrite(fd, zeros.data(), zeros.size(), 0); + ::close(fd); + return n == static_cast(zeros.size()); +} + +int CountL0(DB* db, const std::string& cf_name = "") { + std::vector files; + db->GetLiveFilesMetaData(&files); + int n = 0; + for (const auto& f : files) { + if (f.level == 0 && (cf_name.empty() || f.column_family_name == cf_name)) { + n++; + } + } + return n; +} + +#if !defined(OS_WIN) +// Only exec/_exit run between fork and exec; DB and SyncPoint state are +// initialized in the new process, including Env's background worker threads. +int RunCrashChild(const std::string& dbname, const char* scenario, + const std::string& arg = "") { + char exe[] = "/proc/self/exe"; + char flag[] = "--crash-child"; + char* argv[] = {exe, flag, const_cast(scenario), + const_cast(dbname.c_str()), + const_cast(arg.c_str()), nullptr}; + const pid_t pid = ::fork(); + if (pid == 0) { + ::execv(exe, argv); + ::_exit(127); + } + if (pid < 0) { + return -1; + } + int st = 0; + pid_t waited; + do { + waited = ::waitpid(pid, &st, 0); + } while (waited < 0 && errno == EINTR); + return waited == pid && WIFEXITED(st) ? WEXITSTATUS(st) : -1; +} + +const char* crash_child_db = nullptr; +const char* crash_child_arg = nullptr; + +class CrashChild : public ::testing::Test { + protected: + const std::string dbname_ = crash_child_db ? crash_child_db : ""; + const std::string arg_ = crash_child_arg ? crash_child_arg : ""; +}; + +const int kKindPrepChildCrashed = 42; + +TEST_F(CrashChild, DISABLED_KindPrep) { + Options log_index = BaseCrashSafeOptions(dbname_, true, true); + log_index.avoid_flush_during_shutdown = true; + SyncPoint::GetInstance()->SetCallBack( + arg_, [](void*) { ::_exit(kKindPrepChildCrashed); }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* db = nullptr; + Status s = DB::Open(log_index, dbname_, &db); + ::_exit(s.ok() ? 1 : 2); +} +#endif + +} // namespace + +class DBCsppCrashSafeTest : public DBTestBase { + public: + DBCsppCrashSafeTest() + : DBTestBase("db_cspp_crash_safe_test", /*env_do_fsync=*/false) {} +}; + +#if !defined(OS_WIN) +TEST_F(CrashChild, DISABLED_RegisteredMemTables) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (arg_ == "osl") SetupOsl(&options, true); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "first", "1")); + ASSERT_OK(static_cast(child_db)->TEST_SwitchMemtable()); + ASSERT_OK(child_db->Put(WriteOptions(), "second", "2")); + ASSERT_OK(static_cast(child_db)->TEST_SwitchMemtable()); + // The registered empty active file must also exist for prefix recovery. + ::_exit(42); +} + +TEST_F(DBCsppCrashSafeTest, MissingRegisteredMemTableUsesFullWal) { + Close(); + for (bool osl : {false, true}) { + for (int missing : {0, 1, 2, 3}) { + SCOPED_TRACE(osl); + SCOPED_TRACE(missing); + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "RegisteredMemTables", + osl ? "osl" : "cspp"), 42); + const auto registered = ListLeftovers(options, dbname_); + ASSERT_EQ(registered.size(), 3U); + if (missing == 3) { + for (const auto& path : registered) ASSERT_OK(env_->DeleteFile(path)); + } else { + ASSERT_OK(env_->DeleteFile(registered[missing])); + } + std::atomic converted{0}; + SyncPoint::GetInstance()->SetCallBack( + "MemTableRep::ConvertToSST:After", + [&](void*) { ++converted; }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(options)); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + // Recovery may convert an intact prefix before discovering the missing + // registered file, but must discard that prefix and replay the full WAL. + ASSERT_EQ(converted.load(), missing == 3 ? 0 : missing); + ASSERT_EQ(Get("first"), "1"); + ASSERT_EQ(Get("second"), "2"); + ASSERT_EQ(NumTableFilesAtLevel(0), 0); + ASSERT_OK(Flush()); + ASSERT_EQ(NumTableFilesAtLevel(0), 1); + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("first"), "1"); + ASSERT_EQ(Get("second"), "2"); + ASSERT_EQ(NumTableFilesAtLevel(0), 1); + Close(); + } + } +} + +TEST_F(DBCsppCrashSafeTest, FailedRegistrationCannotAcceptWrites) { + Close(); + for (bool osl : {false, true}) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("before", "safe")); + ASSERT_OK(dbfull()->TEST_SwitchMemtable()); + const auto caller = std::this_thread::get_id(); + std::atomic failed{0}; + std::atomic registering_cache{false}; + SyncPoint::GetInstance()->SetCallBack( + "FlushJob::BeforeManifest", [&](void*) { registering_cache.store(true); }); + SyncPoint::GetInstance()->SetCallBack( + "VersionSet::ProcessManifestWrites:AfterSyncManifest", [&](void* p) { + if (!registering_cache.load()) return; + EXPECT_NE(std::this_thread::get_id(), caller); + ++failed; + *static_cast(p) = IOStatus::IOError("register injection"); + }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_NOK(Flush()); + ASSERT_GT(failed.load(), 0); + auto* pending = dbfull()->GetVersionSet()->GetColumnFamilySet() + ->GetDefault()->PeekPrecreatedMemtable(); + ASSERT_NE(pending, nullptr); + ASSERT_FALSE(pending->IsFileRegistered()); + ASSERT_NOK(dbfull()->TEST_SwitchMemtable()); + ASSERT_NOK(Put("unregistered", "must-not-commit")); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_EQ(Get("before"), "safe"); + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("before"), "safe"); + ASSERT_EQ(Get("unregistered"), "NOT_FOUND"); + Close(); + } +} + +TEST_F(DBCsppCrashSafeTest, FailedInitialRegistrationClearsDbPointer) { + Close(); + for (bool osl : {false, true}) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + Destroy(options); + std::atomic failed{0}; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::RegisterMemTableFile:AfterLogAndApply", [&](void* p) { + ++failed; + *static_cast(p) = Status::IOError("initial register injection"); + }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* opened = nullptr; + const Status status = DB::Open(options, dbname_, &opened); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_NOK(status); + ASSERT_GT(failed.load(), 0); + ASSERT_EQ(opened, nullptr); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("after-failure", "safe")); + Close(); + } +} + +TEST_F(CrashChild, DISABLED_RegistrationCommitWindow) { + ASSERT_EQ(arg_.size(), 3U); + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (arg_[0] == '1') SetupOsl(&options, true); + const char* point = arg_[2] == '0' + ? (arg_[1] == '0' ? "DBImpl::RegisterMemTableFile:AfterLogAndApply" + : "DBImpl::RegisterMemTableFile:BeforeInstall") + : (arg_[1] == '0' ? "FlushJob::MemTableCache:BeforePublish" + : "FlushJob::MemTableCache:AfterPublish"); + const auto arm = [&] { + SyncPoint::GetInstance()->SetCallBack(point, [](void*) { ::_exit(42); }); + SyncPoint::GetInstance()->EnableProcessing(); + }; + if (arg_[2] == '0') arm(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "before-switch", "preserved")); + if (arg_[2] == '1') arm(); + ASSERT_OK(child_db->Flush(FlushOptions())); + ::_exit(1); +} + +TEST_F(DBCsppCrashSafeTest, RegistrationCommitCrashKeepsManifestInventory) { + Close(); + for (bool osl : {false, true}) { + for (bool marked : {false, true}) { + for (bool switching : {false, true}) { + SCOPED_TRACE(osl); + SCOPED_TRACE(marked); + SCOPED_TRACE(switching); + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + Destroy(options); + const std::string arg = std::to_string(osl) + std::to_string(marked) + + std::to_string(switching); + ASSERT_EQ(RunCrashChild(dbname_, "RegistrationCommitWindow", arg), 42); + const auto registered = ListLeftovers(options, dbname_); + ASSERT_EQ(registered.size(), 2U); + for (const auto& path : registered) ASSERT_OK(env_->FileExists(path)); + ASSERT_OK(TryReopen(options)); + ASSERT_TRUE(dbfull()->GetVersionSet()->HasMemTableFileTracking()); + ASSERT_EQ(Get("before-switch"), switching ? "preserved" : "NOT_FOUND"); + ASSERT_OK(Put("after-crash", "committed")); + Close(); + for (const auto& path : ListLeftovers(options, dbname_)) + ASSERT_OK(env_->FileExists(path)); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("before-switch"), switching ? "preserved" : "NOT_FOUND"); + ASSERT_EQ(Get("after-crash"), "committed"); + Close(); + } + } + } +} + +TEST_F(DBCsppCrashSafeTest, FailedNewColumnFamilyRegistrationRemainsReopenable) { + Close(); + for (bool osl : {false, true}) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("existing", "preserved")); + std::atomic failed{0}; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::RegisterMemTableFile:AfterLogAndApply", [&](void* p) { + ++failed; + *static_cast(p) = Status::IOError("new CF register injection"); + }); + SyncPoint::GetInstance()->EnableProcessing(); + ColumnFamilyHandle* handle = nullptr; + const Status created = db_->CreateColumnFamily(options, "failed", &handle); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_NOK(created); + ASSERT_GT(failed.load(), 0); + ASSERT_EQ(handle, nullptr); + ASSERT_EQ(Get("existing"), "preserved"); + // Closing and reopening also exercises manifest snapshots over this CF. + Close(); + ASSERT_OK(TryReopenWithColumnFamilies({kDefaultColumnFamilyName, "failed"}, + options)); + ASSERT_EQ(Get(0, "existing"), "preserved"); + ASSERT_EQ(Get(1, "existing"), "NOT_FOUND"); + std::unique_ptr iterator(db_->NewIterator(ReadOptions(), handles_[1])); + iterator->SeekToFirst(); + ASSERT_FALSE(iterator->Valid()); + ASSERT_OK(iterator->status()); + iterator.reset(); + Close(); + } +} + +TEST_F(DBCsppCrashSafeTest, FileMmapRejectsReadOnlyAndSecondary) { + Close(); + for (bool osl : {false, true}) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "WalFilterFallsBackToWal", + osl ? "10" : "00"), 0); + std::vector before; + ASSERT_OK(env_->GetChildren(dbname_, &before)); + std::sort(before.begin(), before.end()); + for (bool secondary : {false, true}) { + SCOPED_TRACE(osl); + SCOPED_TRACE(secondary); + DB* rejected = nullptr; + const Status s = secondary + ? DB::OpenAsSecondary(options, dbname_, dbname_ + "_secondary", + &rejected) + : DB::OpenForReadOnly(options, dbname_, &rejected); + ASSERT_TRUE(s.IsInvalidArgument()) << s.ToString(); + ASSERT_EQ(rejected, nullptr); + std::vector after; + ASSERT_OK(env_->GetChildren(dbname_, &after)); + std::sort(after.begin(), after.end()); + ASSERT_EQ(after, before); + } + } +} + +TEST_F(DBCsppCrashSafeTest, ReadWriteWalRecoveryFailureDoesNotAbort) { + class CorruptSecondRecord final : public WalFilter { + public: + int calls = 0; + const char* Name() const override { return "CorruptSecondRecord"; } + WalProcessingOption LogRecordFound(unsigned long long, const std::string&, + const WriteBatch&, WriteBatch*, + bool*) override { + return ++calls == 1 ? WalProcessingOption::kContinueProcessing + : WalProcessingOption::kCorruptedRecord; + } + }; + Close(); + for (bool osl : {false, true}) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "WalFilterFallsBackToWal", + osl ? "10" : "00"), 0); + auto list_sst = [&] { + std::vector children; + EXPECT_OK(env_->GetChildren(dbname_, &children)); + std::vector files; + for (const auto& child : children) { + if (child.size() >= 4 && child.compare(child.size() - 4, 4, ".sst") == 0) + files.push_back(child); + } + std::sort(files.begin(), files.end()); + return files; + }; + const auto before = list_sst(); + ASSERT_FALSE(before.empty()); + std::map original_files; + for (const auto& name : before) { + ASSERT_OK(ReadFileToString(env_, dbname_ + "/" + name, &original_files[name])); + } + CorruptSecondRecord filter; + options.memtable_crash_safe_recover = false; + options.wal_filter = &filter; + options.wal_recovery_mode = WALRecoveryMode::kAbsoluteConsistency; + DB* failed = nullptr; + const Status status = DB::Open(options, dbname_, &failed); + ASSERT_TRUE(status.IsCorruption()) << status.ToString(); + ASSERT_EQ(failed, nullptr); + ASSERT_EQ(filter.calls, 2); + for (const auto& file : original_files) { + std::string after; + ASSERT_OK(ReadFileToString(env_, dbname_ + "/" + file.first, &after)); + ASSERT_EQ(after, file.second); + } + options.wal_filter = nullptr; + options.memtable_crash_safe_recover = true; + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("a"), std::string(128, 'a')); + ASSERT_EQ(Get("b"), std::string(128, 'b')); + Close(); + } +} + +TEST_F(DBCsppCrashSafeTest, ManifestRolloverPreservesMemTableRegistry) { + Close(); + for (bool osl : {false, true}) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.max_manifest_file_size = 1; + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("first", "1")); + std::string before; + ASSERT_OK(ReadFileToString(env_, CurrentFileName(dbname_), &before)); + ASSERT_OK(Flush()); + ASSERT_OK(Put("second", "2")); + std::string after; + ASSERT_OK(ReadFileToString(env_, CurrentFileName(dbname_), &after)); + ASSERT_NE(before, after); + const auto registered = dbfull()->GetVersionSet()->GetColumnFamilySet() + ->GetDefault()->GetMemTableFiles(); + ASSERT_EQ(registered.size(), 2U); + const auto disk = ListLeftovers(options, dbname_); + ASSERT_EQ(disk.size(), 2U); + for (uint64_t number : registered) { + const auto path = MakeTableFileName(dbname_, number); + ASSERT_EQ(std::count(disk.begin(), disk.end(), path), 1); + ASSERT_OK(env_->FileExists(path)); + } + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("first"), "1"); + ASSERT_EQ(Get("second"), "2"); + Close(); + } +} + +TEST_F(CrashChild, DISABLED_LegacyManifestWal) { + Options options = BaseCrashSafeOptions(dbname_, false, false); + options.memtable_factory = std::make_shared(); + options.table_factory = Options().table_factory; + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + WriteOptions write; + write.sync = true; + ASSERT_OK(child_db->Put(write, "legacy-first", "one")); + ASSERT_OK(child_db->Put(write, "legacy-second", "two")); + ::_exit(42); +} + +TEST_F(DBCsppCrashSafeTest, LegacyManifestWithoutTrackingUsesFullWal) { + Close(); + for (bool osl : {false, true}) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "LegacyManifestWal"), 42); + ASSERT_TRUE(ListLeftovers(options, dbname_).empty()); + std::vector children; + ASSERT_OK(env_->GetChildren(dbname_, &children)); + uint64_t wal_number = 0; + for (const auto& child : children) { + uint64_t number; + FileType type; + if (ParseFileName(child, &number, &type) && type == kWalFile) + wal_number = std::max(wal_number, number); + } + ASSERT_NE(wal_number, 0U); + uint64_t wal_size = 0; + ASSERT_OK(env_->GetFileSize(LogFileName(dbname_, wal_number), &wal_size)); + ASSERT_GT(wal_size, 0U); + // Valid classic sidecar deliberately points past both keys. An inventory + // without tracking cannot justify skipping this WAL prefix. + PublishedSeqOnDisk record; + record.magic = 0x5145534255505343ULL; + record.version = 1; + record.header_size = sizeof(record); + record.wal_offset_kind = 1; + record.kind_since_wal = static_cast(wal_number); + record.generation = 2; + record.pubseq = 2; + record.wal_number = wal_number; + record.wal_offset = wal_size; + std::string sidecar(4096, '\0'); + std::memcpy(&sidecar[0], &record, sizeof(record)); + ASSERT_OK(WriteStringToFile(env_, sidecar, CrashSafePubSeqFileName(dbname_))); + std::atomic converted{0}; + std::atomic reads{0}; + SyncPoint::GetInstance()->SetCallBack( + "MemTableRep::ConvertToSST:After", [&](void*) { ++converted; }); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::RecoverLogFiles:BeforeReadWal", [&](void*) { ++reads; }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(options)); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_EQ(converted.load(), 0); + ASSERT_GT(reads.load(), 0); + ASSERT_EQ(Get("legacy-first"), "one"); + ASSERT_EQ(Get("legacy-second"), "two"); + ASSERT_TRUE(dbfull()->GetVersionSet()->HasMemTableFileTracking()); + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("legacy-first"), "one"); + ASSERT_EQ(Get("legacy-second"), "two"); + Close(); + } +} + +TEST_F(CrashChild, DISABLED_FlushManifestWindow) { + ASSERT_EQ(arg_.size(), 2U); + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (arg_[0] == '1') SetupOsl(&options, true); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "converted", "durable")); + SyncPoint::GetInstance()->SetCallBack( + arg_[1] == '0' ? "FlushJob::BeforeManifest" + : "FlushJob::AfterManifest", + [](void*) { ::_exit(42); }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(child_db->Flush(FlushOptions())); + ::_exit(1); +} + +TEST_F(DBCsppCrashSafeTest, ConversionCrashAcrossManifestCommit) { + Close(); + for (bool osl : {false, true}) { + for (bool committed : {false, true}) { + SCOPED_TRACE(osl); + SCOPED_TRACE(committed); + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + Destroy(options); + const std::string arg = std::string(osl ? "1" : "0") + + (committed ? "1" : "0"); + ASSERT_EQ(RunCrashChild(dbname_, "FlushManifestWindow", arg), 42); + const auto registered = ListLeftovers(options, dbname_); + ASSERT_FALSE(registered.empty()); + for (const auto& path : registered) ASSERT_OK(env_->FileExists(path)); + std::atomic converts{0}; + SyncPoint::GetInstance()->SetCallBack( + "MemTableRep::ConvertToSST:After", + [&](void*) { ++converts; }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(options)); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_EQ(converts.load(), committed ? 0 : 1); + ASSERT_EQ(Get("converted"), "durable"); + ASSERT_EQ(CountL0(db_), 1); + std::vector files; + db_->GetLiveFilesMetaData(&files); + ASSERT_EQ(files.size(), 1U); + if (!committed) { + const std::string path = MakeTableFileName(dbname_, files[0].file_number); + ASSERT_EQ(std::count(registered.begin(), registered.end(), path), 1); + ASSERT_OK(env_->FileExists(path)); + } + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("converted"), "durable"); + ASSERT_EQ(CountL0(db_), 1); + Close(); + } + } +} + +TEST_F(DBCsppCrashSafeTest, GarbageCollectionKeepsActiveAndCachedMemTables) { + Close(); + for (bool osl : {false, true}) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + Destroy(options); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(options)); + std::atomic published{0}; + SyncPoint::GetInstance()->SetCallBack( + "FlushJob::MemTableCache:BeforePublish", [&](void*) { + std::vector expected; + for (auto* cfd : *dbfull()->GetVersionSet()->GetColumnFamilySet()) { + for (uint64_t number : cfd->GetMemTableFiles()) { + expected.push_back(MakeTableFileName(dbname_, number)); + } + } + ASSERT_EQ(ListLeftovers(options, dbname_), expected); + for (const auto& path : expected) ASSERT_OK(env_->FileExists(path)); + }); + SyncPoint::GetInstance()->SetCallBack( + "FlushJob::MemTableCache:AfterPublish", [&](void*) { ++published; }); + ASSERT_OK(Put("first", "1")); + ASSERT_OK(Flush()); + ASSERT_GT(published.load(), 0); + ASSERT_OK(Put("active", "2")); + const auto registered = dbfull()->GetVersionSet()->GetColumnFamilySet() + ->GetDefault()->GetMemTableFiles(); + ASSERT_GE(registered.size(), 2U); + ASSERT_OK(db_->DisableFileDeletions()); + ASSERT_OK(db_->EnableFileDeletions(true)); + for (uint64_t number : registered) { + ASSERT_OK(env_->FileExists(MakeTableFileName(dbname_, number))); + } + ASSERT_EQ(Get("first"), "1"); + ASSERT_EQ(Get("active"), "2"); + Close(); + const auto kept = ListLeftovers(options, dbname_); + ASSERT_FALSE(kept.empty()); + for (const auto& path : kept) ASSERT_OK(env_->FileExists(path)); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("first"), "1"); + ASSERT_EQ(Get("active"), "2"); + Close(); + } +} + +TEST_F(DBCsppCrashSafeTest, EmptyFlushRetiresSourceFileWithoutFullScan) { + Close(); + for (bool osl : {false, true}) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.delete_obsolete_files_period_micros = UINT64_MAX; + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_OK(TryReopen(options)); + auto* cfd = dbfull()->GetVersionSet()->GetColumnFamilySet()->GetDefault(); + const std::string source = + MakeTableFileName(dbname_, cfd->mem()->GetFileNumber()); + ASSERT_OK(env_->FileExists(source)); + ASSERT_OK(dbfull()->TEST_SwitchMemtable()); + ASSERT_OK(Flush()); + ASSERT_OK(dbfull()->TEST_WaitForBackgroundWork()); + ASSERT_OK(dbfull()->TEST_WaitForPurge()); + std::vector files; + db_->GetLiveFilesMetaData(&files); + ASSERT_TRUE(files.empty()); + ASSERT_TRUE(env_->FileExists(source).IsNotFound()); + const auto active = ListLeftovers(options, dbname_); + ASSERT_EQ(active.size(), 2U); + for (const auto& path : active) ASSERT_OK(env_->FileExists(path)); + const auto converted = MakeTableFileName(dbname_, cfd->mem()->GetFileNumber()); + ASSERT_OK(Put("converted", "preserved")); + ASSERT_OK(Flush()); + ASSERT_OK(dbfull()->TEST_WaitForBackgroundWork()); + ASSERT_OK(dbfull()->TEST_WaitForPurge()); + db_->GetLiveFilesMetaData(&files); + ASSERT_EQ(files.size(), 1U); + ASSERT_EQ(MakeTableFileName(dbname_, files[0].file_number), converted); + ASSERT_OK(env_->FileExists(converted)); + for (const auto& path : ListLeftovers(options, dbname_)) + ASSERT_OK(env_->FileExists(path)); + ASSERT_EQ(Get("converted"), "preserved"); + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("converted"), "preserved"); + Close(); + } +} + +TEST_F(DBCsppCrashSafeTest, DroppedCfRetiresSourceFilesWithoutFullScan) { + Close(); + for (bool osl : {false, true}) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.delete_obsolete_files_period_micros = UINT64_MAX; + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_OK(TryReopen(options)); + CreateAndReopenWithCF({"retire"}, options); + ASSERT_OK(Put(0, "keep", "one")); + ASSERT_OK(Put(1, "drop", "two")); + const auto registered = dbfull()->GetVersionSet()->GetColumnFamilySet() + ->GetColumnFamily(1)->GetMemTableFiles(); + ASSERT_EQ(registered.size(), 2U); + for (uint64_t number : registered) + ASSERT_OK(env_->FileExists(MakeTableFileName(dbname_, number))); + ASSERT_OK(db_->DropColumnFamily(handles_[1])); + ASSERT_OK(db_->DestroyColumnFamilyHandle(handles_[1])); + handles_.pop_back(); + // An ordinary flush provides normal obsolete-file GC, without a scan. + ASSERT_OK(Flush(0)); + ASSERT_OK(dbfull()->TEST_WaitForBackgroundWork()); + ASSERT_OK(dbfull()->TEST_WaitForPurge()); + for (uint64_t number : registered) + ASSERT_TRUE(env_->FileExists(MakeTableFileName(dbname_, number)).IsNotFound()); + ASSERT_EQ(dbfull()->GetVersionSet()->GetColumnFamilySet() + ->GetColumnFamily(1), nullptr); + for (const auto& path : ListLeftovers(options, dbname_)) + ASSERT_OK(env_->FileExists(path)); + std::vector files; + db_->GetLiveFilesMetaData(&files); + ASSERT_EQ(files.size(), 1U); + ASSERT_OK(env_->FileExists(MakeTableFileName(dbname_, files[0].file_number))); + ASSERT_EQ(Get(0, "keep"), "one"); + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("keep"), "one"); + Close(); + } +} + +TEST_F(DBCsppCrashSafeTest, CheckpointDoesNotHardLinkWritableMemTable) { + Close(); + for (bool osl : {false, true}) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("checkpoint", "original")); + const auto registered = dbfull()->GetVersionSet()->GetColumnFamilySet() + ->GetDefault()->GetMemTableFiles(); + ASSERT_EQ(registered.size(), 2U); + const uint64_t number = dbfull()->GetVersionSet()->GetColumnFamilySet() + ->GetDefault()->mem()->GetFileNumber(); + ASSERT_EQ(registered.count(number), 1U); + const std::string checkpoint_dir = dbname_ + ".checkpoint"; + Options copy_options = options; + copy_options.wal_dir = checkpoint_dir; + ASSERT_OK(DestroyDB(checkpoint_dir, copy_options)); + Checkpoint* raw = nullptr; + ASSERT_OK(Checkpoint::Create(db_, &raw)); + std::unique_ptr checkpoint(raw); + ASSERT_OK(checkpoint->CreateCheckpoint(checkpoint_dir, UINT64_MAX)); + // This memtable is still mutable, and hence must not be a live SST. + const auto& after = dbfull()->GetVersionSet()->GetColumnFamilySet() + ->GetDefault()->GetMemTableFiles(); + if (after.count(number)) { + ASSERT_TRUE(env_->FileExists(MakeTableFileName(checkpoint_dir, number)) + .IsNotFound()); + } + ASSERT_OK(Put("checkpoint", "source-changed")); + DB* copy_raw = nullptr; + ASSERT_OK(DB::Open(copy_options, checkpoint_dir, ©_raw)); + std::unique_ptr copy(copy_raw); + std::string value; + ASSERT_OK(copy->Get(ReadOptions(), "checkpoint", &value)); + ASSERT_EQ(value, "original"); + copy.reset(); + checkpoint.reset(); + ASSERT_OK(DestroyDB(checkpoint_dir, copy_options)); + Close(); + } +} + + + +TEST_F(DBCsppCrashSafeTest, OpenAndCreateColumnFamilyRegisterNextMemTable) { + Close(); + for (bool osl : {false, true}) { + for (bool atomic : {false, true}) { + SCOPED_TRACE(osl); + SCOPED_TRACE(atomic); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.atomic_flush = atomic; + options.avoid_flush_during_shutdown = true; + if (osl) SetupOsl(&options, true); + Destroy(options); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(options)); + auto* default_cfd = dbfull()->GetVersionSet()->GetColumnFamilySet() + ->GetColumnFamily(0); + dbfull()->TEST_LockMutex(); + auto* default_head = default_cfd->PeekPrecreatedMemtable(); + const uint64_t default_cache = default_head ? default_head->GetFileNumber() : 0; + dbfull()->TEST_UnlockMutex(); + ASSERT_NE(default_cache, 0U); + ASSERT_EQ(default_cfd->GetMemTableFiles() + .count(default_cache), 1U); + + ColumnFamilyHandle* handle = nullptr; + ASSERT_OK(db_->CreateColumnFamily(options, "bootstrap", &handle)); + auto* created_cfd = dbfull()->GetVersionSet()->GetColumnFamilySet() + ->GetColumnFamily(handle->GetID()); + dbfull()->TEST_LockMutex(); + auto* created_head = created_cfd->PeekPrecreatedMemtable(); + const uint64_t created_cache = created_head ? created_head->GetFileNumber() : 0; + dbfull()->TEST_UnlockMutex(); + ASSERT_NE(created_cache, 0U); + ASSERT_EQ(created_cfd->GetMemTableFiles() + .count(created_cache), 1U); + std::atomic front_registrations{0}; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::RegisterMemTableFile:BeforeLogAndApply", + [&](void*) { ++front_registrations; }); + ASSERT_OK(Put("default", "1")); + ASSERT_OK(Flush()); + ASSERT_EQ(default_cfd->mem()->GetFileNumber(), default_cache); + ASSERT_TRUE(default_cfd->mem()->IsFileRegistered()); + ASSERT_OK(db_->Put(WriteOptions(), handle, "created", "2")); + ASSERT_OK(db_->Flush(FlushOptions(), handle)); + ASSERT_EQ(created_cfd->mem()->GetFileNumber(), created_cache); + ASSERT_TRUE(created_cfd->mem()->IsFileRegistered()); + ASSERT_EQ(front_registrations.load(), 0); + ASSERT_OK(db_->DestroyColumnFamilyHandle(handle)); + Close(); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + } + } +} + +TEST_F(DBCsppCrashSafeTest, SwitchWaitsForPendingCacheRegistration) { + Close(); + for (bool osl : {false, true}) { + for (bool atomic : {false, true}) { + for (int mode : {0, 1, 2, 3}) { // Success, failure, shutdown, already committed. + const bool fail = mode == 1; + const bool shutdown = mode == 2; + SCOPED_TRACE(osl); + SCOPED_TRACE(atomic); + SCOPED_TRACE(mode); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.atomic_flush = atomic; + options.max_bgerror_resume_count = 0; + options.avoid_flush_during_shutdown = true; + if (osl) SetupOsl(&options, true); + Destroy(options); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(options)); + auto* cfd = dbfull()->GetVersionSet()->GetColumnFamilySet() + ->GetColumnFamily(0); + ASSERT_OK(Put("first", "1")); + + std::mutex mu; + std::condition_variable cv; + bool paused = false, release = false, waiting = false; + bool flush_done = false, switch_done = false, pause_timeout = false; + std::atomic frontend_registrations{0}; + const auto caller = std::this_thread::get_id(); + std::atomic arm{false}, inject{false}; + SyncPoint::GetInstance()->SetCallBack( + "FlushJob::BeforeManifest", [&](void*) { arm.store(true); }); + SyncPoint::GetInstance()->SetCallBack( + "VersionSet::LogAndApply:WriteManifestStart", [&](void*) { + if (!arm.exchange(false)) return; + std::unique_lock lk(mu); + paused = true; + inject.store(fail); + cv.notify_all(); + // Do not strand the flush thread if an assertion misses the hook. + pause_timeout = !cv.wait_for(lk, std::chrono::seconds(30), + [&] { return release; }); + }); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::SwitchMemtable:BeforeInstallMemTable", [&](void*) { + if (mode != 3) return; + std::unique_lock lk(mu); + if (!paused) return; + waiting = true; + cv.notify_all(); + if (!cv.wait_for(lk, std::chrono::seconds(30), + [&] { return release && flush_done; })) + std::abort(); + }); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::SwitchMemtable:MemTableCacheMiss", [&](void* p) { + EXPECT_NE(static_cast(p)->GetFileNumber(), 0U); + if (mode == 3) return; + std::lock_guard lk(mu); + waiting = true; + cv.notify_all(); + }); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::RegisterMemTableFile:BeforeLogAndApply", [&](void*) { + EXPECT_NE(std::this_thread::get_id(), caller); + ++frontend_registrations; + }); + SyncPoint::GetInstance()->SetCallBack( + "VersionSet::ProcessManifestWrites:AfterSyncManifest", [&](void* p) { + if (inject.exchange(false)) { + *static_cast(p) = IOStatus::IOError("cache wait injection"); + } + }); + Status flush_status, switch_status; + std::thread flush_thread([&] { + flush_status = Flush(); + std::lock_guard lk(mu); + flush_done = true; + cv.notify_all(); + }); + bool reached_pause; + { + std::unique_lock lk(mu); + reached_pause = cv.wait_for(lk, std::chrono::seconds(10), + [&] { return paused || flush_done; }) && paused; + } + std::vector cache_files; + std::vector before_switch, after_switch; + uint64_t active_file = 0; + if (reached_pause) { + dbfull()->TEST_LockMutex(); + if (auto* head = cfd->PeekPrecreatedMemtable()) { + cache_files.push_back(head->GetFileNumber()); + } + active_file = cfd->mem()->GetFileNumber(); + dbfull()->TEST_UnlockMutex(); + EXPECT_OK(env_->GetChildren(dbname_, &before_switch)); + EXPECT_OK(Put("active", "2")); + } + std::thread switch_thread([&] { + switch_status = dbfull()->TEST_SwitchMemtable(); + std::lock_guard lk(mu); + switch_done = true; + cv.notify_all(); + }); + bool reached_wait; + { + std::unique_lock lk(mu); + reached_wait = cv.wait_for(lk, std::chrono::seconds(10), + [&] { return waiting || switch_done; }) && waiting; + } + if (reached_pause && reached_wait) { + // The popped head remains protected by the switch's pending output. + std::vector still_cached; + dbfull()->TEST_LockMutex(); + if (auto* head = cfd->PeekPrecreatedMemtable()) { + still_cached.push_back(head->GetFileNumber()); + } + dbfull()->TEST_UnlockMutex(); + EXPECT_TRUE(still_cached.empty()); + EXPECT_OK(db_->DisableFileDeletions()); + EXPECT_OK(db_->EnableFileDeletions(true)); + for (uint64_t number : cache_files) { + EXPECT_OK(env_->FileExists(MakeTableFileName(dbname_, number))); + } + EXPECT_OK(env_->GetChildren(dbname_, &after_switch)); + auto only_ssts = [](const std::vector& children) { + std::set files; + for (const auto& child : children) { + if (child.size() >= 4 && child.compare(child.size() - 4, 4, ".sst") == 0) + files.insert(child); + } + return files; + }; + EXPECT_EQ(only_ssts(before_switch), only_ssts(after_switch)); + std::lock_guard lk(mu); + EXPECT_FALSE(switch_done); + } + if (shutdown && reached_wait) { + CancelAllBackgroundWork(db_, false); + } + { + std::unique_lock lk(mu); + release = true; + cv.notify_all(); + if (!cv.wait_for(lk, std::chrono::seconds(30), + [&] { return flush_done && switch_done; })) { + // A missed wakeup must fail this test instead of hanging in join. + std::abort(); + } + } + flush_thread.join(); + switch_thread.join(); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_TRUE(reached_pause); + ASSERT_TRUE(reached_wait); + ASSERT_FALSE(pause_timeout); + ASSERT_EQ(cache_files.size(), 1U); + ASSERT_EQ(frontend_registrations.load(), mode == 3 ? 0 : 1); + if (shutdown) { + ASSERT_TRUE(switch_status.IsShutdownInProgress()); + ASSERT_TRUE(flush_status.ok() || flush_status.IsShutdownInProgress()); + ASSERT_EQ(cfd->mem()->GetFileNumber(), active_file); + } else if (fail) { + ASSERT_TRUE(flush_status.IsIOError()); + ASSERT_TRUE(switch_status.IsIOError()); + ASSERT_TRUE(dbfull()->TEST_GetBGError().IsIOError()); + ASSERT_EQ(cfd->mem()->GetFileNumber(), active_file); + ASSERT_OK(db_->DisableFileDeletions()); + ASSERT_OK(db_->EnableFileDeletions(true)); + ASSERT_OK(env_->FileExists(MakeTableFileName(dbname_, active_file))); + } else { + ASSERT_OK(flush_status); + ASSERT_OK(switch_status); + ASSERT_EQ(cfd->mem()->GetFileNumber(), cache_files.front()); + ASSERT_TRUE(cfd->mem()->IsFileRegistered()); + ASSERT_OK(Put("switched", "3")); + } + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("first"), "1"); + ASSERT_EQ(Get("active"), "2"); + if (!fail && !shutdown) { + ASSERT_EQ(Get("switched"), "3"); + } + Close(); + } + } + } +} + +TEST_F(DBCsppCrashSafeTest, CacheMissRegistersWhileFlushIsPaused) { + Close(); + for (bool osl : {false, true}) { + for (bool atomic : {false, true}) { + for (bool fail : {false, true}) { + SCOPED_TRACE(osl); + SCOPED_TRACE(atomic); + SCOPED_TRACE(fail); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.atomic_flush = atomic; + options.paranoid_checks = !fail; + options.avoid_flush_during_shutdown = true; + if (osl) SetupOsl(&options, true); + Destroy(options); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(options)); + auto* cfd = dbfull()->GetVersionSet()->GetColumnFamilySet() + ->GetColumnFamily(0); + ASSERT_OK(Put("first", "1")); + if (atomic) { + CreateColumnFamilies({"aux"}, options); + ASSERT_EQ(handles_.size(), 1U); + } + auto* aux = atomic ? handles_.back() : nullptr; + if (atomic) { + ASSERT_OK(db_->Put(WriteOptions(), aux, "aux-first", "3")); + } + std::mutex mu; + std::condition_variable cv; + bool paused = false, release = false, done = false, waiting = false; + bool pause_timeout = false; + std::atomic first_convert{true}; + std::atomic conversions{0}; + const auto caller = std::this_thread::get_id(); + std::thread::id switch_id; + std::atomic registrations{0}; + std::atomic foreground_registrations{0}; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::SwitchMemtable:MemTableCacheMiss", [&](void*) { + std::lock_guard lk(mu); + waiting = true; + cv.notify_all(); + }); + SyncPoint::GetInstance()->SetCallBack( + "MemTableRep::ConvertToSST:Before", [&](void*) { + if (!first_convert.exchange(false)) return; + std::unique_lock lk(mu); + paused = true; + cv.notify_all(); + pause_timeout = !cv.wait_for(lk, std::chrono::seconds(30), + [&] { return release; }); + }); + SyncPoint::GetInstance()->SetCallBack( + "FlushJob::ConvertToSST:Status", [&](void* p) { + EXPECT_TRUE(static_cast(p)->ok()); + // Atomic flush runs aux first, then default; fail only the last. + if (++conversions == (atomic ? 2 : 1) && fail) + *static_cast(p) = Status::IOError("ignored table flush injection"); + }); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::RegisterMemTableFile:BeforeLogAndApply", [&](void*) { + std::lock_guard lk(mu); + EXPECT_NE(std::this_thread::get_id(), caller); + if (std::this_thread::get_id() == switch_id) + ++foreground_registrations; + else + ++registrations; + }); + FlushOptions flush_options; + flush_options.wait = false; + const Status flush_status = atomic + ? db_->Flush(flush_options, {db_->DefaultColumnFamily(), aux}) + : db_->Flush(flush_options); + bool reached_pause; + { + std::unique_lock lk(mu); + reached_pause = cv.wait_for(lk, std::chrono::seconds(10), + [&] { return paused; }); + } + dbfull()->TEST_LockMutex(); + const bool cache_empty = cfd->PeekPrecreatedMemtable() == nullptr; + dbfull()->TEST_UnlockMutex(); + EXPECT_TRUE(cache_empty); + EXPECT_OK(Put("second", "2")); + Status switch_status; + std::thread switch_thread([&] { + { + std::lock_guard lk(mu); + switch_id = std::this_thread::get_id(); + } + switch_status = dbfull()->TEST_SwitchMemtable(); + std::lock_guard lk(mu); + done = true; + cv.notify_all(); + }); + bool reached_wait; + bool completed_before_release; + { + std::unique_lock lk(mu); + reached_wait = cv.wait_for( + lk, std::chrono::seconds(10), [&] { return waiting || done; }) && + waiting; + } + { + std::unique_lock lk(mu); + completed_before_release = cv.wait_for( + lk, std::chrono::seconds(10), [&] { return done; }); + release = true; + cv.notify_all(); + if (!cv.wait_for(lk, std::chrono::seconds(30), [&] { return done; })) + std::abort(); + } + switch_thread.join(); + ASSERT_OK(dbfull()->TEST_WaitForBackgroundWork()); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_TRUE(reached_pause); + ASSERT_TRUE(reached_wait); + ASSERT_TRUE(completed_before_release); + ASSERT_FALSE(pause_timeout); + ASSERT_OK(flush_status); + ASSERT_OK(switch_status); + ASSERT_OK(dbfull()->TEST_GetBGError()); + ASSERT_TRUE(cfd->mem()->IsFileRegistered()); + ASSERT_EQ(foreground_registrations.load(), 1); + ASSERT_EQ(conversions.load(), atomic ? 2 : 1); + ASSERT_EQ(registrations.load(), 0); + if (atomic) { + auto* aux_cfd = dbfull()->GetVersionSet()->GetColumnFamilySet() + ->GetColumnFamily(aux->GetID()); + dbfull()->TEST_LockMutex(); + auto* aux_head = aux_cfd->PeekPrecreatedMemtable(); + const uint64_t aux_cache_number = aux_head ? aux_head->GetFileNumber() : 0; + dbfull()->TEST_UnlockMutex(); + ASSERT_NE(aux_cache_number, 0U); + ASSERT_EQ(dbfull()->GetVersionSet()->GetColumnFamilySet() + ->GetColumnFamily(aux->GetID())->GetMemTableFiles() + .count(aux_cache_number), fail ? 0U : 1U); + if (fail) { + ASSERT_OK(dbfull()->TEST_SwitchMemtable(aux_cfd)); + ASSERT_EQ(aux_cfd->mem()->GetFileNumber(), aux_cache_number); + ASSERT_TRUE(aux_cfd->mem()->IsFileRegistered()); + ASSERT_EQ(aux_cfd->GetMemTableFiles().count(aux_cache_number), 1U); + } + } + Close(); + if (atomic) { + ASSERT_OK(TryReopenWithColumnFamilies({"default", "aux"}, options)); + ASSERT_EQ(Get(0, "first"), "1"); + ASSERT_EQ(Get(0, "second"), "2"); + ASSERT_EQ(Get(1, "aux-first"), "3"); + } else { + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("first"), "1"); + ASSERT_EQ(Get("second"), "2"); + } + Close(); + } + } + } +} + +TEST_F(DBCsppCrashSafeTest, FrontendCacheMissRegistrationFailure) { + Close(); + for (bool osl : {false, true}) { + for (bool after_sync : {false, true}) { + SCOPED_TRACE(osl); + SCOPED_TRACE(after_sync); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.max_bgerror_resume_count = 0; + options.avoid_flush_during_shutdown = true; + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("first", "1")); + ASSERT_OK(dbfull()->TEST_SwitchMemtable()); + ASSERT_OK(Put("active", "2")); + auto* cfd = dbfull()->GetVersionSet()->GetColumnFamilySet()->GetDefault(); + const uint64_t active = cfd->mem()->GetFileNumber(); + ASSERT_EQ(cfd->PeekPrecreatedMemtable(), nullptr); + const auto caller = std::this_thread::get_id(); + int injected = 0; + uint64_t candidate = 0; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::SwitchMemtable:MemTableCacheMiss", [&](void* p) { + candidate = static_cast(p)->GetFileNumber(); + EXPECT_NE(candidate, active); + EXPECT_EQ(cfd->PeekPrecreatedMemtable(), nullptr); + }); + SyncPoint::GetInstance()->SetCallBack( + after_sync ? "VersionSet::ProcessManifestWrites:AfterSyncManifest" + : "DBImpl::RegisterMemTableFile:AfterLogAndApply", + [&](void* p) { + EXPECT_EQ(std::this_thread::get_id(), caller); + ++injected; + if (after_sync) { + *static_cast(p) = IOStatus::IOError("frontend register injection"); + } else { + *static_cast(p) = Status::IOError("frontend register injection"); + } + }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_TRUE(dbfull()->TEST_SwitchMemtable().IsIOError()); + ASSERT_EQ(injected, 1); + ASSERT_TRUE(dbfull()->TEST_GetBGError().IsIOError()); + ASSERT_EQ(cfd->mem()->GetFileNumber(), active); + ASSERT_NE(candidate, 0U); + ASSERT_EQ(cfd->GetMemTableFiles().count(candidate), after_sync ? 0U : 1U); + ASSERT_EQ(cfd->PeekPrecreatedMemtable(), nullptr); + ASSERT_OK(env_->FileExists(MakeTableFileName(dbname_, candidate))); + ASSERT_NOK(Put("rejected", "3")); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_OK(db_->DisableFileDeletions()); + ASSERT_OK(db_->EnableFileDeletions(true)); + ASSERT_OK(env_->FileExists(MakeTableFileName(dbname_, active))); + ASSERT_EQ(Get("first"), "1"); + ASSERT_EQ(Get("active"), "2"); + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("first"), "1"); + ASSERT_EQ(Get("active"), "2"); + ASSERT_EQ(Get("rejected"), "NOT_FOUND"); + Close(); + } + } +} + +TEST_F(DBCsppCrashSafeTest, FailedCacheRegistrationSurvivesGcAndReopen) { + Close(); + for (bool osl : {false, true}) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.max_bgerror_resume_count = 0; + if (osl) SetupOsl(&options, true); + Destroy(options); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("retry", "preserved")); + std::vector pending; + std::atomic pending_commit{false}; + SyncPoint::GetInstance()->SetCallBack("FlushJob::BeforeManifest", [&](void*) { + std::vector children; + ASSERT_OK(env_->GetChildren(dbname_, &children)); + for (const auto& child : children) { + if (child.size() >= 4 && child.compare(child.size() - 4, 4, ".sst") == 0) + pending.push_back(dbname_ + "/" + child); + } + pending_commit.store(true); + }); + std::atomic injected{false}; + std::atomic published{0}; + SyncPoint::GetInstance()->SetCallBack( + "VersionSet::ProcessManifestWrites:AfterSyncManifest", [&](void* p) { + if (pending_commit.load() && !injected.exchange(true)) + *static_cast(p) = IOStatus::IOError("cache register injection"); + }); + SyncPoint::GetInstance()->SetCallBack( + "FlushJob::MemTableCache:AfterPublish", [&](void*) { ++published; }); + ASSERT_NOK(Flush()); + ASSERT_TRUE(injected.load()); + ASSERT_EQ(published.load(), 0); + ASSERT_GE(pending.size(), 3U); // converting input, active, pending cache + ASSERT_OK(db_->DisableFileDeletions()); + ASSERT_OK(db_->EnableFileDeletions(true)); + for (const auto& path : pending) ASSERT_OK(env_->FileExists(path)); + SyncPoint::GetInstance()->ClearCallBack("FlushJob::BeforeManifest"); + // Plain MANIFEST IOError is fatal under the existing error policy. + // Resume preserves that error; reopening is the supported recovery path. + const Status resumed = db_->Resume(); + ASSERT_TRUE(resumed.IsIOError()); + ASSERT_EQ(Get("retry"), "preserved"); + Close(); + SyncPoint::GetInstance()->ClearCallBack( + "VersionSet::ProcessManifestWrites:AfterSyncManifest"); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("retry"), "preserved"); + published.store(0); + ASSERT_OK(Put("after-reopen", "committed")); + ASSERT_OK(Flush()); + ASSERT_GT(published.load(), 0); + ASSERT_EQ(Get("retry"), "preserved"); + Close(); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("retry"), "preserved"); + ASSERT_EQ(Get("after-reopen"), "committed"); + Close(); + } +} +#endif + +TEST_F(DBCsppCrashSafeTest, CrashSafeRequiresFileMmapFactories) { + Close(); + for (bool osl : {false, true}) { + for (const char* mode : {"kDontConvert", "kDumpMem", "SkipList"}) { + SCOPED_TRACE(osl); + SCOPED_TRACE(mode); + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + Options unsupported = options; + unsupported.memtable_factory = std::string(mode) == "SkipList" + ? std::shared_ptr(new SkipListFactory) + : EasyNewMemTableRep(osl ? "OffsetSkipList" : "CSPPMemTab", + json({{"mem_cap", 16777216}, {"convert_to_sst", mode}}).dump()); + Destroy(options); + DB* rejected = nullptr; + ASSERT_TRUE(DB::Open(unsupported, dbname_, &rejected).IsInvalidArgument()); + ASSERT_EQ(rejected, nullptr); + ASSERT_OK(TryReopen(options)); + ColumnFamilyHandle* handle = nullptr; + ASSERT_TRUE(db_->CreateColumnFamily(unsupported, "mixed", &handle) + .IsInvalidArgument()); + ASSERT_EQ(handle, nullptr); + Close(); + options.memtable_crash_safe_recover = false; + options.avoid_flush_during_shutdown = true; + unsupported.memtable_crash_safe_recover = false; + ASSERT_OK(TryReopen(options)); + ASSERT_OK(db_->CreateColumnFamily(unsupported, "mixed", &handle)); + handles_.push_back(handle); + ASSERT_OK(Put("file", "value")); + ASSERT_OK(db_->Put(WriteOptions(), handle, "mixed", "value")); + if (unsupported.memtable_factory->SupportConvertToSST()) { + ASSERT_OK(db_->Flush(FlushOptions(), handle)); + } + for (auto* cfd : *dbfull()->GetVersionSet()->GetColumnFamilySet()) { + ASSERT_TRUE(cfd->GetMemTableFiles().empty()); + } + Close(); + ASSERT_OK(TryReopenWithColumnFamilies( + {"default", "mixed"}, std::vector{options, unsupported})); + ASSERT_EQ(Get(0, "file"), "value"); + ASSERT_EQ(Get(1, "mixed"), "value"); + Close(); + } + } +} + +TEST_F(DBCsppCrashSafeTest, RecoverOffKeepsUnregisteredMemTableFiles) { + Close(); + for (bool osl : {false, true}) { + SCOPED_TRACE(osl); + Options options = BaseCrashSafeOptions(dbname_, false, false); + options.experimental_mempurge_threshold = 2.0; + if (osl) SetupOsl(&options, true); + Destroy(options); + int registrations = 0, waits = 0, inspected = 0; + std::atomic converts{0}, purges{0}; + SyncPoint::GetInstance()->SetCallBack( + "MemTableRep::ConvertToSST:After", [&](void*) { ++converts; }); + for (const char* point : {"DBImpl::FlushJob:MemPurgeSuccessful", + "DBImpl::FlushJob:MemPurgeUnsuccessful"}) { + SyncPoint::GetInstance()->SetCallBack(point, [&](void*) { ++purges; }); + } + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::RegisterMemTableFile:BeforeLogAndApply", + [&](void*) { ++registrations; }); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::SwitchMemtable:MemTableCacheMiss", + [&](void*) { ++waits; }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(options)); + // Open reuses this hook for ordinary recovery MANIFEST writes. + registrations = 0; + ASSERT_OK(Put("sst", "1")); + ASSERT_OK(Flush()); + ASSERT_GT(converts.load(), 0); + ASSERT_EQ(purges.load(), 0); + auto* cfd = dbfull()->GetVersionSet()->GetColumnFamilySet()->GetDefault(); + ASSERT_NE(cfd->PeekPrecreatedMemtable(), nullptr); + const uint64_t cached = cfd->PeekPrecreatedMemtable()->GetFileNumber(); + ASSERT_OK(Put("imm", "2")); + const uint64_t immutable = cfd->mem()->GetFileNumber(); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::SwitchMemtable:BeforeInstallMemTable", [&](void*) { + ASSERT_OK(db_->DisableFileDeletions()); + ASSERT_OK(db_->EnableFileDeletions(true)); + ASSERT_OK(env_->FileExists(MakeTableFileName(dbname_, cached))); + ++inspected; + }); + ASSERT_OK(dbfull()->TEST_SwitchMemtable()); + SyncPoint::GetInstance()->ClearCallBack( + "DBImpl::SwitchMemtable:BeforeInstallMemTable"); + ASSERT_EQ(inspected, 1); + ASSERT_EQ(cfd->mem()->GetFileNumber(), cached); + ASSERT_OK(Put("active", "3")); + // Refill through the ordinary cache producer to check all three live roles. + cfd->PrepareNewMemtableInBackground(*cfd->GetLatestMutableCFOptions()); + auto* precreated = cfd->PeekPrecreatedMemtable(); + ASSERT_NE(precreated, nullptr); + const uint64_t next = precreated->GetFileNumber(); + for (auto* cfd : *dbfull()->GetVersionSet()->GetColumnFamilySet()) { + ASSERT_TRUE(cfd->GetMemTableFiles().empty()); + } + ASSERT_OK(db_->DisableFileDeletions()); + ASSERT_OK(db_->EnableFileDeletions(true)); + for (uint64_t number : {immutable, cached, next}) { + ASSERT_OK(env_->FileExists(MakeTableFileName(dbname_, number))); + } + ASSERT_EQ(registrations, 0); + ASSERT_EQ(waits, 0); + ASSERT_EQ(Get("imm"), "2"); + ASSERT_EQ(Get("active"), "3"); + Close(); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("sst"), "1"); + ASSERT_EQ(Get("imm"), "2"); + ASSERT_EQ(Get("active"), "3"); + Close(); + } +} + +TEST_F(DBCsppCrashSafeTest, ImmutableFactoryConvertMode) { + Close(); + const SidePluginRepo repo; + const json query = {{"html", false}}; + auto check = [&](const auto& factory, const auto* manip, const char* mode) { + auto state = [&] { + return json::parse(manip->ToString(*factory, query, repo)); + }; + auto update = [&](const json& body) { + try { + manip->HandleUpdate(factory.get(), query, body, repo); + return Status::OK(); + } catch (const Status& s) { + return s; + } + }; + ASSERT_EQ(state()["convert_to_sst"], mode); + ASSERT_OK(update({{"token_use_idle", false}})); + ASSERT_EQ(state()["token_use_idle"], false); + ASSERT_OK(update({{"convert_to_sst", mode}})); + const json before = state(); + for (const char* other : {"kDontConvert", "kDumpMem", "kFileMmap", "invalid"}) { + if (std::string(other) == mode) continue; + SCOPED_TRACE(other); + Status s = update({{"convert_to_sst", other}, + {"token_use_idle", true}, {"populate_read", false}}); + ASSERT_TRUE(s.IsInvalidArgument()); + ASSERT_EQ(state(), before); + } + }; + for (const char* mode : {"kDontConvert", "kDumpMem", "kFileMmap"}) { + SCOPED_TRACE(mode); + const json params = {{"mem_cap", 16777216}, {"convert_to_sst", mode}}; + for (const char* cls : {"CSPPMemTab", "OffsetSkipList"}) { + SCOPED_TRACE(cls); + auto factory = PluginFactorySP::AcquirePlugin( + cls, params, repo); + ASSERT_EQ(factory->SupportCrashSafe(), std::string(mode) == "kFileMmap"); + auto* manip = PluginManip::AcquirePlugin(cls, {}, repo); + check(factory, manip, mode); + InternalKeyComparator icmp(BytewiseComparator()); + MemTable::KeyComparator cmp(icmp); + Arena arena; + MutableCFOptions moptions(Options{}); + const std::string path = MakeTableFileName(dbname_, 900000); + if (std::string(mode) == "kFileMmap") { + ASSERT_DEATH(factory->CreateMemTableRep( + "", moptions, cmp, &arena, nullptr, nullptr, 0), "memtable_file_path"); + } + std::unique_ptr rep(factory->CreateMemTableRep( + path, moptions, cmp, &arena, nullptr, nullptr, 0)); + ASSERT_EQ(enum_stdstr(rep->GetConvertKind()), mode); + ASSERT_EQ(rep->SupportConvertToSST(), std::string(mode) != "kDontConvert"); + ASSERT_EQ(rep->SupportCrashSafe(), std::string(mode) == "kFileMmap"); + rep.reset(); + if (std::string(mode) == "kFileMmap") { + ASSERT_OK(env_->DeleteFile(path)); + } + } + for (const char* cls : {"CSPPMemTabTable", "OffsetSkipListTable"}) { + SCOPED_TRACE(cls); + auto factory = PluginFactorySP::AcquirePlugin(cls, params, repo); + auto* manip = PluginManip::AcquirePlugin(cls, {}, repo); + check(factory, manip, mode); + } + } +} + +TEST_F(DBCsppCrashSafeTest, RecoveredTableUsesFileSequenceBound) { + Close(); + for (bool osl : {false, true}) { + SCOPED_TRACE(osl ? "OSL" : "CSPP"); + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) { + SetupOsl(&options, true); + } + Destroy(options); + ASSERT_OK(env_->CreateDirIfMissing(dbname_)); + options.cf_paths = {{dbname_, 0}}; + InternalKeyComparator icmp(options.comparator); + ImmutableOptions ioptions(options); + MutableCFOptions moptions(options); + WriteBufferManager wb(options.db_write_buffer_size); + std::unique_ptr mem(new MemTable( + icmp, ioptions, moptions, &wb, kMaxSequenceNumber, 0, 1)); + ASSERT_OK(mem->Add(1, kTypeValue, "key", "old", nullptr)); + ASSERT_OK(mem->Add(3, kTypeValue, "key", "new", nullptr)); + ASSERT_OK(mem->Add(4, kTypeValue, "ghost", "unpublished", nullptr)); + mem->MarkImmutable(); + const std::string leftover = MakeTableFileName(dbname_, 2); + CopyFile(MakeTableFileName(dbname_, 1), leftover); + mem.reset(); + + IntTblPropCollectorFactories collectors; + TableBuilderOptions tbo(ioptions, moptions, icmp, &collectors, + options.compression, options.compression_opts, 0, + kDefaultColumnFamilyName, 0); + FileMetaData meta; + meta.fd = FileDescriptor(2, 0, 0); + meta.fd.smallest_seqno = 0; + // The published bound need not be the sequence of any physical entry. + meta.fd.largest_seqno = 2; + ASSERT_OK(options.memtable_factory->RecoverCrashSafeMemTableToSST( + leftover, &meta, tbo)); + ASSERT_GT(meta.fd.GetFileSize(), 0U); + ASSERT_EQ(meta.fd.smallest_seqno, 0U); + ASSERT_EQ(meta.fd.largest_seqno, 2U); + + const std::string fname = TableFileName(options.cf_paths, 2, 0); + for (SequenceNumber limit : {meta.fd.largest_seqno, SequenceNumber(4), + SequenceNumber(0), kMaxSequenceNumber}) { + SCOPED_TRACE(limit); + std::unique_ptr file; + ASSERT_OK(env_->GetFileSystem()->NewRandomAccessFile( + fname, FileOptions(), &file, nullptr)); + std::unique_ptr reader( + new RandomAccessFileReader(std::move(file), fname)); + EnvOptions env_options; + TableReaderOptions tro(ioptions, options.prefix_extractor, env_options, + icmp, 0); + // The file header owns visibility, independently of the caller's bound. + FileDescriptor fd = meta.fd; + fd.largest_seqno = limit; + tro.largest_seqno = fd.largest_seqno; + if (osl) { + const auto block = ReadMetaBlockE( + reader.get(), fd.GetFileSize(), 0x62546d654d4c534fULL, ioptions, + "OffsetSkipList"); + ASSERT_EQ(block.data.size(), 48U); + // Packed metadata has a 40-byte prefix followed by uint64_t pubseq. + uint64_t pubseq; + memcpy(&pubseq, block.data.data() + 40, sizeof(pubseq)); + ASSERT_EQ(pubseq, 2U); + } + std::unique_ptr table; + ASSERT_OK(options.table_factory->NewTableReader( + ReadOptions(), tro, std::move(reader), fd.GetFileSize(), &table, + true)); + const auto& compression = table->GetTableProperties()->compression_options; + ASSERT_EQ(compression.substr(0, compression.find(';', 1)), ";pubseq:2"); + const auto* view = dynamic_cast(table.get()); + ASSERT_NE(view, nullptr); + ASSERT_EQ(json::parse(view->ToWebViewString({{"html", false}}))["pubseq"], 2); + for (const char* key : {"key", "ghost"}) { + PinnableSlice value; + GetContext get_context( + options.comparator, nullptr, nullptr, nullptr, GetContext::kNotFound, + key, &value, nullptr, nullptr, nullptr, true, nullptr, nullptr); + InternalKey ikey(key, kMaxSequenceNumber, kTypeValue); + ASSERT_OK(table->Get(ReadOptions(), ikey.Encode(), &get_context, + nullptr)); + const bool visible = key[0] == 'k'; + ASSERT_EQ(get_context.State(), + visible ? GetContext::kFound : GetContext::kNotFound); + if (visible) { + ASSERT_EQ(value.ToString(), "old"); + } + } + } + SstFileReader standalone(options); + ASSERT_OK(standalone.Open(fname)); + std::unique_ptr standalone_it(standalone.NewIterator(ReadOptions())); + standalone_it->SeekToFirst(); + ASSERT_TRUE(standalone_it->Valid()); + ASSERT_EQ(standalone_it->key().ToString(), "key"); + ASSERT_EQ(standalone_it->value().ToString(), "old"); + standalone_it->Next(); + ASSERT_FALSE(standalone_it->Valid()); + ASSERT_OK(standalone_it->status()); + SstFileDumper dumper(options, fname, Temperature::kUnknown, 0, + true, false, false, EnvOptions(), true); + ASSERT_OK(dumper.getStatus()); + ASSERT_OK(dumper.ReadSequential(false, 0, false, "", false, "")); + ASSERT_EQ(dumper.GetReadNumber(), 1U); + } +} + +TEST_F(CrashChild, DISABLED_RecoveredHiddenRecordsCompact) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (arg_ == "OSL") { + SetupOsl(&options, true); + } + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "keep", "old")); + auto* cfd = static_cast(child_db)->GetVersionSet()->GetColumnFamilySet()->GetDefault(); + // These physical entries are beyond the published sequence and absent in WAL. + ASSERT_OK(cfd->mem()->Add(2, kTypeValue, "keep", "ghost", nullptr)); + ASSERT_OK(cfd->mem()->Add(3, kTypeValue, "ghost", "unpublished", nullptr)); + ::_exit(42); +} + +TEST_F(DBCsppCrashSafeTest, RecoveredHiddenRecordsCompact) { + Close(); + for (const auto& config : + {std::make_pair(false, -1), {false, 12}, {true, -1}, {true, 12}, + {false, 0}}) { + const bool osl = config.first; + const bool fail_lookup = config.second == 0; + SCOPED_TRACE(osl); + SCOPED_TRACE(config.second); + Options options = BaseCrashSafeOptions(dbname_, true, false); + // Capacity is max_open_files - 10; capacity / 4 must be zero to avoid pinning. + options.max_open_files = fail_lookup ? 12 : config.second; + if (osl) { + SetupOsl(&options, true); + } + ASSERT_TRUE(options.compaction_verify_record_count); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "RecoveredHiddenRecordsCompact", osl ? "OSL" : "CSPP"), 42); + // Compaction writes ordinary tables while dispatch reads the recovered format. + SidePluginRepo repo; + repo.Put("default", Options().table_factory); + repo.Put("converted", options.table_factory); + options.table_factory = PluginFactorySP::AcquirePlugin( + "Dispatch", {{"default", "$default"}}, repo); + DispatcherTableBackPatch(options.table_factory.get(), repo); + ASSERT_OK(TryReopen(options)); + TablePropertiesCollection properties; + ASSERT_OK(db_->GetPropertiesOfAllTables(&properties)); + uint64_t physical_entries = 0; + for (const auto& property : properties) { + physical_entries += property.second->num_entries; + } + ASSERT_EQ(physical_entries, 3U); + auto check_visible = [&] { + std::unique_ptr it(db_->NewIterator(ReadOptions())); + it->SeekToFirst(); + ASSERT_TRUE(it->Valid()); + ASSERT_EQ(it->key().ToString(), "keep"); + ASSERT_EQ(it->value().ToString(), "old"); + it->Next(); + ASSERT_FALSE(it->Valid()); + ASSERT_OK(it->status()); + ASSERT_EQ(Get("ghost"), "NOT_FOUND"); + }; + check_visible(); + // Overlap prevents a trivial move, forcing the record-count verification path. + ASSERT_OK(Put("keep", "old")); + ASSERT_OK(Flush()); + ASSERT_EQ(CountL0(db_), 2); + auto* cfd = dbfull()->GetVersionSet()->GetColumnFamilySet()->GetDefault(); + for (const auto* file : cfd->current()->storage_info()->LevelFiles(0)) { + ASSERT_EQ(file->fd.table_reader != nullptr, config.second == -1); + if (config.second != -1) { + TableCache::Evict(dbfull()->TEST_table_cache(), file->fd.GetNumber()); + } + } + std::atomic unsupported{0}, find_table{0}, pinned_reader{0}, mismatch{0}; + auto* sync = SyncPoint::GetInstance(); + sync->SetCallBack("CompactionJob::VerifyRecordCount:Unsupported", + [&](void*) { unsupported++; }); + sync->SetCallBack("CompactionJob::VerifyRecordCount:FindTable", + [&](void*) { find_table++; }); + sync->SetCallBack("CompactionJob::VerifyRecordCount:PinnedReader", + [&](void*) { pinned_reader++; }); + sync->SetCallBack("CompactionJob::VerifyRecordCount:Mismatch", [&](void*) { mismatch++; }); + if (fail_lookup) { + sync->SetCallBack("CompactionJob::VerifyRecordCount:FindTableStatus", + [](void* arg) { + *static_cast(arg) = Status::IOError("injected reader lookup failure"); + }); + } + sync->EnableProcessing(); + Status compact = db_->CompactRange(CompactRangeOptions(), nullptr, nullptr); + sync->DisableProcessing(); + sync->ClearCallBack("CompactionJob::VerifyRecordCount:Unsupported"); + sync->ClearCallBack("CompactionJob::VerifyRecordCount:FindTable"); + sync->ClearCallBack("CompactionJob::VerifyRecordCount:PinnedReader"); + sync->ClearCallBack("CompactionJob::VerifyRecordCount:FindTableStatus"); + sync->ClearCallBack("CompactionJob::VerifyRecordCount:Mismatch"); + if (fail_lookup) { + ASSERT_TRUE(compact.IsCorruption()); + ASSERT_NE(std::strstr(compact.getState(), "Compaction number of input keys"), nullptr); + ASSERT_EQ(unsupported.load(), 0); + ASSERT_GT(mismatch.load(), 0); + ASSERT_GT(find_table.load(), 0); + Close(); + continue; + } + ASSERT_OK(compact); + ASSERT_GT(unsupported.load(), 0); + ASSERT_EQ(mismatch.load(), 0); + if (config.second == -1) { + ASSERT_EQ(find_table.load(), 0); + ASSERT_GT(pinned_reader.load(), 0); + } else { + ASSERT_GT(find_table.load(), 0); + } + ASSERT_EQ(CountL0(db_), 0); + check_visible(); + Close(); + } +} + +TEST_F(DBCsppCrashSafeTest, RecoveredTableIteratorUsesFileSequenceBound) { + Close(); + for (bool osl : {false, true}) { + for (bool reverse : {false, true}) { + SCOPED_TRACE(osl ? "OSL" : "CSPP"); + SCOPED_TRACE(reverse); + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + if (reverse) options.comparator = ReverseBytewiseComparator(); + Destroy(options); + ASSERT_OK(env_->CreateDirIfMissing(dbname_)); + options.cf_paths = {{dbname_, 0}}; + InternalKeyComparator icmp(options.comparator); + ImmutableOptions ioptions(options); + MutableCFOptions moptions(options); + WriteBufferManager wb(options.db_write_buffer_size); + auto mem = std::make_unique( + icmp, ioptions, moptions, &wb, kMaxSequenceNumber, 0, 1); + std::vector physical; + for (const auto& entry : {std::make_pair("b", 1), {"d", 2}, {"b", 3}, + {"d", 5}, {"c", 6}, {"b", 7}, {"a", 8}, + {"z", 9}}) { + ASSERT_OK(mem->Add(entry.second, kTypeValue, entry.first, + std::to_string(entry.second), nullptr)); + physical.push_back( + InternalKey(entry.first, entry.second, kTypeValue).Encode().ToString()); + } + auto less = [&](const std::string& a, const std::string& b) { + return icmp.Compare(a, b) < 0; + }; + std::sort(physical.begin(), physical.end(), less); + mem->MarkImmutable(); + const std::string leftover = MakeTableFileName(dbname_, 2); + CopyFile(MakeTableFileName(dbname_, 1), leftover); + mem.reset(); + IntTblPropCollectorFactories collectors; + TableBuilderOptions tbo(ioptions, moptions, icmp, &collectors, + options.compression, options.compression_opts, 0, + kDefaultColumnFamilyName, 0); + FileMetaData meta; + meta.fd = FileDescriptor(2, 0, 0); + meta.fd.smallest_seqno = 0; + meta.fd.largest_seqno = 4; + ASSERT_OK(options.memtable_factory->RecoverCrashSafeMemTableToSST( + leftover, &meta, tbo)); + const std::string fname = TableFileName(options.cf_paths, 2, 0); + for (SequenceNumber limit : {SequenceNumber(0), SequenceNumber(4), + SequenceNumber(9), kMaxSequenceNumber}) { + SCOPED_TRACE(limit); + std::unique_ptr file; + ASSERT_OK(env_->GetFileSystem()->NewRandomAccessFile( + fname, FileOptions(), &file, nullptr)); + auto reader = std::make_unique(std::move(file), fname); + EnvOptions env_options; + TableReaderOptions tro(ioptions, options.prefix_extractor, env_options, + icmp, 0); + tro.largest_seqno = limit; + if (osl) { + const auto block = ReadMetaBlockE( + reader.get(), meta.fd.GetFileSize(), 0x62546d654d4c534fULL, + ioptions, "OffsetSkipList"); + ASSERT_EQ(block.data.size(), 48U); + uint64_t pubseq; + memcpy(&pubseq, block.data.data() + 40, sizeof(pubseq)); + ASSERT_EQ(pubseq, 4U); + } + std::unique_ptr table; + ASSERT_OK(options.table_factory->NewTableReader( + ReadOptions(), tro, std::move(reader), meta.fd.GetFileSize(), + &table, true)); + const auto& compression = table->GetTableProperties()->compression_options; + ASSERT_EQ(compression.substr(0, compression.find(';', 1)), ";pubseq:4"); + const auto* view = dynamic_cast(table.get()); + ASSERT_NE(view, nullptr); + ASSERT_EQ(json::parse(view->ToWebViewString({{"html", false}}))["pubseq"], 4); + std::vector expected; + for (const auto& key : physical) { + if (GetInternalKeySeqno(key) <= 4) expected.push_back(key); + } + for (bool use_arena : {false, true}) { + SCOPED_TRACE(use_arena); + Arena arena; + auto destroy = [&](InternalIterator* p) { + if (use_arena) p->~InternalIterator(); + else delete p; + }; + std::unique_ptr it( + table->NewIterator(ReadOptions(), nullptr, + use_arena ? &arena : nullptr, false, + TableReaderCaller::kUserIterator), destroy); + auto check = [&](size_t pos) { + ASSERT_EQ(it->Valid(), pos < expected.size()); + if (it->Valid()) { + ASSERT_EQ(it->key().ToString(), expected[pos]); + ASSERT_EQ(it->value().ToString(), + std::to_string(GetInternalKeySeqno(expected[pos]))); + } + }; + // Exercise every forward entry point, including the fast result API. + for (int advance = 0; advance < 3; ++advance) { + it->SeekToFirst(); + for (size_t i = 0; i < expected.size(); ++i) { + check(i); + ASSERT_TRUE(it->Valid()); + if (advance == 0) { + it->Next(); + } else if (advance == 1) { + ASSERT_EQ(it->NextAndCheckValid(), i + 1 < expected.size()); + } else { + IterateResult result; + ASSERT_EQ(it->NextAndGetResult(&result), i + 1 < expected.size()); + ASSERT_EQ(result.is_valid, i + 1 < expected.size()); + if (result.is_valid) { + ASSERT_EQ(result.key().ToString(), expected[i + 1]); + } + } + } + check(expected.size()); + } + for (bool fast : {false, true}) { + it->SeekToLast(); + for (size_t i = expected.size(); i > 0; --i) { + check(i - 1); + ASSERT_TRUE(it->Valid()); + if (fast) { + ASSERT_EQ(it->PrevAndCheckValid(), i > 1); + } else { + it->Prev(); + } + } + check(expected.size()); + } + auto* mem_it = static_cast(it.get()); + std::vector targets = physical; + for (const char* key : {"", "a", "b", "bb", "c", "d", "zz"}) { + for (SequenceNumber seq : {SequenceNumber(0), SequenceNumber(4), + kMaxSequenceNumber}) { + targets.push_back(InternalKey(key, seq, kTypeValue).Encode().ToString()); + } + } + for (const auto& target : targets) { + const size_t next = std::lower_bound( + expected.begin(), expected.end(), target, less) - expected.begin(); + const size_t upper = std::upper_bound( + expected.begin(), expected.end(), target, less) - expected.begin(); + const size_t prev = upper ? upper - 1 : expected.size(); + it->Seek(target); + check(next); + it->SeekForPrev(target); + check(prev); + std::string encoded; + const char* memkey = EncodeKey(&encoded, target); + mem_it->Seek(target, memkey); + check(next); + mem_it->SeekForPrev(target, memkey); + check(prev); + } + if (osl) { + for (int i = 0; i < 20; ++i) { + mem_it->RandomSeek(); + if (it->Valid()) { + ASSERT_LE(GetInternalKeySeqno(it->key()), 4U); + } + } + } + } + } + } + } +} + +TEST_F(DBCsppCrashSafeTest, ConvertedTableUsesUnboundedHeaderSequence) { + Close(); + for (bool osl : {false, true}) { + for (bool file_mmap : {false, true}) { + SCOPED_TRACE(osl ? "OSL" : "CSPP"); + SCOPED_TRACE(file_mmap); + Options options = BaseCrashSafeOptions(dbname_, true, false); + const char* js = file_mmap + ? R"({"mem_cap":16777216,"convert_to_sst":"kFileMmap"})" + : R"({"mem_cap":16777216,"convert_to_sst":"kDumpMem"})"; + if (osl) { + options.memtable_factory = EasyNewMemTableRep("OffsetSkipList", js); + } else { + options.memtable_factory.reset(NewCSPPMemTabForPlain(js)); + } + const SidePluginRepo repo; + options.table_factory = PluginFactorySP::AcquirePlugin( + osl ? "OffsetSkipListTable" : "CSPPMemTabTable", json::parse(js), repo); + Destroy(options); + ASSERT_OK(env_->CreateDirIfMissing(dbname_)); + options.cf_paths = {{dbname_, 0}}; + InternalKeyComparator icmp(options.comparator); + ImmutableOptions ioptions(options); + MutableCFOptions moptions(options); + WriteBufferManager wb(options.db_write_buffer_size); + auto mem = std::make_unique( + icmp, ioptions, moptions, &wb, kMaxSequenceNumber, 0, 1); + ASSERT_OK(mem->Add(1, kTypeValue, "a", "1", nullptr)); + ASSERT_OK(mem->Add(2, kTypeValue, "b", "2", nullptr)); + ASSERT_OK(mem->Add(3, kTypeValue, "b", "3", nullptr)); + mem->MarkImmutable(); + IntTblPropCollectorFactories collectors; + TableBuilderOptions tbo(ioptions, moptions, icmp, &collectors, + options.compression, options.compression_opts, 0, + kDefaultColumnFamilyName, 0); + FileMetaData meta; + meta.fd = FileDescriptor(1, 0, 0); + meta.fd.smallest_seqno = 1; + meta.fd.largest_seqno = 3; + ASSERT_OK(mem->ConvertToSST(&meta, tbo)); + ASSERT_GT(meta.fd.GetFileSize(), 0U); + mem.reset(); + + const std::string fname = TableFileName(options.cf_paths, 1, 0); + if (!osl && !file_mmap) { + const int fd = ::open(fname.c_str(), O_RDONLY); + ASSERT_GE(fd, 0); + terark::DFA_MmapHeader header{}; + const ssize_t read = ::pread(fd, &header, sizeof(header), 0); + ::close(fd); + ASSERT_EQ(read, static_cast(sizeof(header))); + uint32_t prefix[3]; + memcpy(prefix, header.reserved, sizeof(prefix)); + ASSERT_EQ(prefix[0], 0x50505343U); // CSPP crash-safe magic. + ASSERT_EQ(prefix[1], 0U); // No WAL references in this conversion. + ASSERT_EQ(prefix[2], 0U); + // pubseq follows the 16-byte prefix and sixteen 24-byte WAL slots. + uint64_t pubseq; + memcpy(&pubseq, header.reserved + 400, sizeof(pubseq)); + ASSERT_EQ(pubseq, 0U); // Unbounded ordinary conversion. + ASSERT_GE(header.crc32cLevel, 1U); + ASSERT_EQ(header.header_crc32, + terark::Crc32c_update(0, &header, sizeof(header) - 4)); + } + const std::type_info* unfiltered_type = nullptr; + for (SequenceNumber limit : {kMaxSequenceNumber, SequenceNumber(0), + SequenceNumber(2), meta.fd.largest_seqno}) { + std::unique_ptr file; + ASSERT_OK(env_->GetFileSystem()->NewRandomAccessFile( + fname, FileOptions(), &file, nullptr)); + auto reader = std::make_unique(std::move(file), fname); + EnvOptions env_options; + TableReaderOptions tro(ioptions, options.prefix_extractor, env_options, + icmp, 0); + tro.largest_seqno = limit; + if (osl) { + const auto block = ReadMetaBlockE( + reader.get(), meta.fd.GetFileSize(), 0x62546d654d4c534fULL, + ioptions, "OffsetSkipList"); + ASSERT_EQ(block.data.size(), 48U); + uint64_t pubseq; + memcpy(&pubseq, block.data.data() + 40, sizeof(pubseq)); + ASSERT_EQ(pubseq, 0U); + } + std::unique_ptr table; + ASSERT_OK(options.table_factory->NewTableReader( + ReadOptions(), tro, std::move(reader), meta.fd.GetFileSize(), + &table, true)); + const auto& compression = table->GetTableProperties()->compression_options; + ASSERT_EQ(compression.find("pubseq:"), std::string::npos); + ASSERT_EQ(compression.find("VisFilter:"), std::string::npos); + const auto* view = dynamic_cast(table.get()); + ASSERT_NE(view, nullptr); + ASSERT_EQ(json::parse(view->ToWebViewString({{"html", false}}))["pubseq"], + "kMaxSequenceNumber"); + std::unique_ptr it(table->NewIterator( + ReadOptions(), nullptr, nullptr, false, + TableReaderCaller::kUserIterator)); + if (limit == kMaxSequenceNumber) { + unfiltered_type = &typeid(*it); + } else { + ASSERT_NE(unfiltered_type, nullptr); + // A caller's finite maximum must not select VisibleIter. + ASSERT_EQ(typeid(*it), *unfiltered_type); + } + size_t count = 0; + for (it->SeekToFirst(); it->Valid(); it->Next()) ++count; + ASSERT_EQ(count, 3U); + } + } + } +} + +TEST_F(DBCsppCrashSafeTest, CsppSelfMmapUnmapsWholeFile) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + ASSERT_OK(env_->CreateDirIfMissing(dbname_)); + options.cf_paths = {{dbname_, 0}}; + InternalKeyComparator icmp(options.comparator); + ImmutableOptions ioptions(options); + MutableCFOptions moptions(options); + WriteBufferManager wb(options.db_write_buffer_size); + std::unique_ptr mem(new MemTable( + icmp, ioptions, moptions, &wb, kMaxSequenceNumber, 0, 1)); + ASSERT_OK(mem->Add(1, kTypeValue, "key", "value", nullptr)); + const std::string path = MakeTableFileName(dbname_, 2); + CopyFile(MakeTableFileName(dbname_, 1), path); + mem.reset(); + const int fd = ::open(path.c_str(), O_RDWR); + ASSERT_GE(fd, 0); + terark::DFA_MmapHeader hdr{}; + ASSERT_EQ(::pread(fd, &hdr, sizeof(hdr), 0), + static_cast(sizeof(hdr))); + ASSERT_EQ(::ftruncate(fd, hdr.file_size + 2 * ::sysconf(_SC_PAGESIZE)), 0); + const size_t physical_size = hdr.file_size + 2 * ::sysconf(_SC_PAGESIZE); + const auto original = hdr; + std::vector buffer((physical_size + 7) / 8); + ASSERT_EQ(::pread(fd, buffer.data(), physical_size, 0), + static_cast(physical_size)); + auto check_unmapped = [&] { + std::ifstream maps("/proc/self/maps"); + ASSERT_TRUE(maps.good()); + for (std::string line; std::getline(maps, line);) { + EXPECT_EQ(line.find(path), std::string::npos) << "leaked mapping: " << line; + } + }; + for (int entry = 0; entry < 4; ++entry) { + // self(path), self(fd), load(path), load(fd). + for (int damage = 0; damage < 7; ++damage) { + SCOPED_TRACE(entry); + SCOPED_TRACE(damage); + if (damage == 6 && entry < 2) continue; // load-only format check. + hdr = original; + size_t length = physical_size; + if (damage == 1) hdr.num_blocks = 0; // finish_load_mmap failure. + if (damage == 2) length = 0; + if (damage == 3) length = sizeof(hdr) - 1; + if (damage == 4) hdr.file_size = sizeof(hdr) - 1; + if (damage == 5) hdr.file_size = physical_size + 1; + if (damage == 6) hdr.magic[0] = '!'; + ASSERT_EQ(::ftruncate(fd, physical_size), 0); + ASSERT_EQ(::pwrite(fd, buffer.data(), physical_size, 0), + static_cast(physical_size)); + ASSERT_EQ(::pwrite(fd, &hdr, sizeof(hdr), 0), + static_cast(sizeof(hdr))); + ASSERT_EQ(::ftruncate(fd, length), 0); + auto open = [&] { + if (entry < 2) { + terark::MainPatricia trie(0, 16 << 20, + terark::Patricia::NoWriteReadOnly); + if (entry == 0) trie.self_mmap(path); + else trie.self_mmap(fd, false); + EXPECT_EQ(trie.get_mmap().size(), original.file_size); + } else { + std::unique_ptr trie(entry == 2 + ? terark::BaseDFA::load_mmap(path, false) + : terark::BaseDFA::load_mmap(fd)); + EXPECT_EQ(trie->get_mmap().size(), original.file_size); + } + }; + if (damage == 0) { + ASSERT_NO_THROW(open()); + } else { + try { + open(); + FAIL() << "expected invalid_argument"; + } catch (const std::invalid_argument& ex) { + if ((entry == 0 || entry == 2) && damage >= 2 && damage <= 5) { + EXPECT_NE(std::string(ex.what()).find(path), std::string::npos); + } + } + } + ASSERT_NE(::fcntl(fd, F_GETFD), -1); // Caller retains its descriptor. + check_unmapped(); + } + } + for (bool load : {false, true}) { + for (int damage = 0; damage < 5; ++damage) { + SCOPED_TRACE(load); + SCOPED_TRACE(damage); + auto* header = reinterpret_cast(buffer.data()); + *header = original; + const void* data = buffer.data(); + size_t length = physical_size; + if (damage == 1) { data = nullptr; length = 0; } + if (damage == 2) length = sizeof(original) - 1; + if (damage == 3) header->file_size = sizeof(original) - 1; + if (damage == 4) header->file_size = physical_size + 1; + buffer.back() = 0x87654321; + auto borrow = [&] { + if (load) { + std::unique_ptr trie( + terark::BaseDFA::load_mmap_user_mem(data, length)); + ASSERT_NE(trie, nullptr); + EXPECT_EQ(trie->get_mmap().size(), original.file_size); + } else { + terark::MainPatricia trie(0, 16 << 20, + terark::Patricia::NoWriteReadOnly); + trie.self_mmap_user_mem(data, length); + EXPECT_EQ(trie.get_mmap().size(), original.file_size); + } + }; + if (damage == 0) { + ASSERT_NO_THROW(borrow()); + } else { + ASSERT_THROW(borrow(), std::invalid_argument); + } + // Borrowers neither free the buffer nor alter its logical or extra tail. + ASSERT_EQ(header->file_size, damage == 3 ? sizeof(original) - 1 + : damage == 4 ? physical_size + 1 + : original.file_size); + ASSERT_EQ(buffer.back(), 0x87654321U); + buffer.back() = 0x12345678; + ASSERT_EQ(buffer.back(), 0x12345678U); + } + } + ::close(fd); +} + +TEST_F(DBCsppCrashSafeTest, OslRecoveryUnmapsWholeFile) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + SetupOsl(&options, true); + Destroy(options); + ASSERT_OK(env_->CreateDirIfMissing(dbname_)); + options.cf_paths = {{dbname_, 0}}; + InternalKeyComparator icmp(options.comparator); + ImmutableOptions ioptions(options); + MutableCFOptions moptions(options); + WriteBufferManager wb(options.db_write_buffer_size); + std::unique_ptr mem(new MemTable( + icmp, ioptions, moptions, &wb, kMaxSequenceNumber, 0, 1)); + ASSERT_OK(mem->Add(1, kTypeValue, "key", "value", nullptr)); + const std::string leftover = MakeTableFileName(dbname_, 2); + CopyFile(MakeTableFileName(dbname_, 1), leftover); + mem.reset(); + IntTblPropCollectorFactories collectors; + TableBuilderOptions tbo(ioptions, moptions, icmp, &collectors, + options.compression, options.compression_opts, 0, + kDefaultColumnFamilyName, 0); + FileMetaData meta; + meta.fd = FileDescriptor(2, 0, 0); + meta.fd.smallest_seqno = 0; + meta.fd.largest_seqno = 1; + ASSERT_OK(options.memtable_factory->RecoverCrashSafeMemTableToSST( + leftover, &meta, tbo)); + std::ifstream maps("/proc/self/maps"); + ASSERT_TRUE(maps.good()); + const auto fname = TableFileName(options.cf_paths, 2, 0); + for (std::string line; std::getline(maps, line);) { + EXPECT_EQ(line.find(fname), std::string::npos) << "leaked mapping: " << line; + } +} + +#if !defined(OS_WIN) +TEST_F(CrashChild, DISABLED_SecondCrashSeed) { + ASSERT_EQ(arg_.size(), 3U); + Options options = BaseCrashSafeOptions(dbname_, true, arg_[1] == '1'); + if (arg_[0] == '1') SetupOsl(&options, true); + DB* db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &db)); + ASSERT_OK(db->Put(WriteOptions(), "a", "1")); + ASSERT_OK(static_cast(db)->TEST_SwitchMemtable()); + ASSERT_OK(db->Put(WriteOptions(), "b", "2")); + ASSERT_OK(static_cast(db)->TEST_SwitchMemtable()); + ASSERT_OK(db->Put(WriteOptions(), "c", "3")); + ::_exit(42); +} + +TEST_F(CrashChild, DISABLED_SecondCrashRecover) { + ASSERT_EQ(arg_.size(), 3U); + Options options = BaseCrashSafeOptions(dbname_, true, arg_[1] == '1'); + if (arg_[0] == '1') SetupOsl(&options, true); + SyncPoint::GetInstance()->SetCallBack( + "CrashSafeRecover::ConvertToSST:InjectStatus", [](void* arg) { + *static_cast(arg) = Status::IOError("review: failed conversion"); + }); + SyncPoint::GetInstance()->SetCallBack( + arg_[2] == '1' ? "DBImpl::Open:Opened" + : "DBImpl::RecoverLogFiles:BeforeReadWal", + [](void*) { ::_exit(43); }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* db = nullptr; + DB::Open(options, dbname_, &db).PermitUncheckedError(); + ::_exit(17); +} +#endif + +TEST_F(DBCsppCrashSafeTest, SecondCrashAfterConvertFailure) { + Close(); + for (bool osl : {false, true}) { + for (bool log_index : {false, true}) { + for (bool after_open : {false, true}) { + SCOPED_TRACE(osl ? "OSL" : "CSPP"); + SCOPED_TRACE(log_index); + SCOPED_TRACE(after_open); + Options options = BaseCrashSafeOptions(dbname_, true, log_index); + if (osl) SetupOsl(&options, true); + Destroy(options); + const std::string child_options = + std::to_string(osl) + std::to_string(log_index) + + std::to_string(after_open); + ASSERT_EQ(RunCrashChild(dbname_, "SecondCrashSeed", child_options), 42); + ASSERT_GE(ListLeftovers(options, dbname_).size(), 3U); + ASSERT_EQ(RunCrashChild(dbname_, "SecondCrashRecover", child_options), 43); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + // Recovery does not invalidate a valid published cursor: registered + // source files survive conversion failure and can be retried. + ASSERT_EQ(rec.generation & 1, 0U); + ASSERT_OK(TryReopen(options)); + EXPECT_EQ(Get("a"), "1"); + EXPECT_EQ(Get("b"), "2"); + EXPECT_EQ(Get("c"), "3"); + ASSERT_OK(Put("after", "recovery")); + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.generation & 1, 0U); + Reopen(options); + EXPECT_EQ(Get("a"), "1"); + EXPECT_EQ(Get("b"), "2"); + EXPECT_EQ(Get("c"), "3"); + EXPECT_EQ(Get("after"), "recovery"); + Close(); + } + } + } +} + +#if !defined(OS_WIN) +TEST_F(CrashChild, DISABLED_PatriciaSingleWriterMmapConstructor) { + const auto mode = static_cast( + std::stoi(arg_)); + const std::string path = dbname_ + "/review-patricia-" + arg_; + alignas(terark::MainPatricia) unsigned char storage[sizeof(terark::MainPatricia)]; + std::memset(storage, 0xa5, sizeof(storage)); + auto* trie = new (storage) terark::MainPatricia( + 4, 16 << 20, mode, terark::fstring(path)); + trie->~MainPatricia(); + ::_exit(0); +} +#endif + +TEST_F(DBCsppCrashSafeTest, PatriciaSingleWriterMmapConstructor) { + Close(); + ASSERT_OK(env_->CreateDirIfMissing(dbname_)); + for (auto mode : {terark::Patricia::SingleThreadStrict, + terark::Patricia::SingleThreadShared, + terark::Patricia::OneWriteMultiRead}) { + SCOPED_TRACE(int(mode)); + ASSERT_EQ(RunCrashChild(dbname_, "PatriciaSingleWriterMmapConstructor", + std::to_string(int(mode))), 0); + } +} + +TEST_F(DBCsppCrashSafeTest, RecoverOffDoesNotCreatePublishedSeq) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, false, false); + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "v")); + Close(); + ASSERT_TRUE(env_->FileExists(CrashSafePubSeqFileName(dbname_)).IsNotFound()); +} + +TEST_F(DBCsppCrashSafeTest, RecoverOnCreatesPublishedSeqAndFlushWalBuffer) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.manual_wal_flush = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_FALSE(dbfull()->GetDBOptions().manual_wal_flush); + ASSERT_OK(Put("k", "v")); + ASSERT_TRUE(dbfull()->WALBufferIsEmpty()); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.generation % 2, 0U); + ASSERT_GT(rec.pubseq, 0U); + ASSERT_GT(rec.wal_offset, 0U); + Close(); + ASSERT_OK(env_->FileExists(CrashSafePubSeqFileName(dbname_))); +} + +TEST_F(DBCsppCrashSafeTest, SkipListRejectsRecoverAndUsesWalWhenDisabled) { + Close(); + Options options = CurrentOptions(); + options.memtable_factory = std::make_shared(); + options.memtable_crash_safe_recover = true; + options.create_if_missing = true; + Destroy(options); + ASSERT_TRUE(TryReopen(options).IsInvalidArgument()); + ASSERT_EQ(db_, nullptr); + options.memtable_crash_safe_recover = false; + options.avoid_flush_during_shutdown = true; + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("a", "1")); + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("a"), "1"); +} + +#if !defined(OS_WIN) +static Options LogRefCrashOptions(const std::string& dbname, + const std::string& config) { + Options options = BaseCrashSafeOptions(dbname, true, true); + const bool osl = config[0] == 'O'; + const json params = { + {"mem_cap", 16777216}, {"convert_to_sst", "kFileMmap"}, + {"log_ref_format", config[1] == 'P' ? "kPlainLogRef" : "kShortLogRef"}}; + options.memtable_factory = EasyNewMemTableRep( + osl ? "OffsetSkipList" : "CSPPMemTab", params.dump()); + const SidePluginRepo repo; + options.table_factory = PluginFactorySP::AcquirePlugin( + osl ? "OffsetSkipListTable" : "CSPPMemTabTable", params, repo); + return options; +} + +static Options PersistentStatsOptions(const std::string& dbname, const std::string& config) { + Options options = BaseCrashSafeOptions(dbname, true, config[1] != 'N'); + options.cf_paths = {{dbname, 0}}; + const bool osl = config[0] == 'O'; + const json params = { + {"mem_cap", 16777216}, {"convert_to_sst", "kFileMmap"}, + {"enable_gc", config[2] == 'E'}, + {"log_ref_format", config[1] == 'S' ? "kShortLogRef" : "kPlainLogRef"}}; + options.memtable_factory = EasyNewMemTableRep( + osl ? "OffsetSkipList" : "CSPPMemTab", params.dump()); + const SidePluginRepo repo; + options.table_factory = PluginFactorySP::AcquirePlugin( + osl ? "OffsetSkipListTable" : "CSPPMemTabTable", params, repo); + options.merge_operator = MergeOperators::CreateStringAppendOperator(); + return options; +} + +TEST_F(CrashChild, DISABLED_PersistentMemTableStats) { + Options options = PersistentStatsOptions(dbname_, arg_); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + std::mutex gate; + std::condition_variable ready_cv; + size_t ready = 0; + const bool exited = arg_[2] == 'E'; + const bool window = arg_[2] == 'B' || arg_[2] == 'A'; + const size_t value_size = arg_[2] == 'T' || arg_[2] == 'F' ? 600 * 1024 : 128; + auto* cfd = static_cast(child_db)->GetVersionSet()->GetColumnFamilySet()->GetDefault(); + auto write = [&](int id) { + WriteOptions wo; + wo.memtable_insert_hint_per_batch = id == 1; + const std::string prefix = std::to_string(id); + const bool check_memory = arg_[2] == 'T' || arg_[2] == 'F'; + size_t memory_before = 0; + if (check_memory) { + // Warm this writer's TLS allocation before measuring WAL-cache growth. + ASSERT_OK(child_db->Put(wo, prefix + "dup", "")); + memory_before = cfd->mem()->ApproximateMemoryUsage(); + } + WriteBatch batch; + if (!check_memory) { + ASSERT_OK(batch.Put(prefix + "dup", "")); + } + ASSERT_OK(batch.Put(prefix + "dup", std::string(value_size, 'v'))); + ASSERT_OK(batch.Delete(prefix + "del")); + ASSERT_OK(batch.SingleDelete(prefix + "one")); + ASSERT_OK(batch.Put(prefix + "last", "v")); + ASSERT_OK(child_db->Write(wo, &batch)); + ASSERT_OK(child_db->Merge(wo, prefix + "mer", "m")); + if (check_memory) { + const size_t memory_after = cfd->mem()->ApproximateMemoryUsage(); + ASSERT_GE(memory_after, memory_before + value_size); + ASSERT_LT(memory_after, memory_before + value_size + 64 * 1024); + } + std::unique_lock lock(gate); + ready++; + ready_cv.notify_all(); + if (!exited) { + ready_cv.wait(lock, [] { return false; }); // TLS remains live at _exit. + } + }; + std::thread first(write, 0); + { + std::unique_lock lock(gate); + ready_cv.wait(lock, [&] { return ready == 1; }); + } + if (exited) first.join(); + const std::string active = MakeTableFileName(dbname_, cfd->mem()->GetFileNumber()); + ASSERT_OK(WriteStringToFile(options.env, active, dbname_ + "/stats-active")); + if (window) { + SyncPoint::GetInstance()->SetCallBack( + arg_[2] == 'B' ? "MemTableStats::BeforePublish" + : "MemTableStats::AfterPublish", + [](void*) { ::_exit(42); }); + SyncPoint::GetInstance()->EnableProcessing(); + std::thread next([&] { + ASSERT_OK(child_db->Put(WriteOptions(), "never-counted", "v")); + }); + next.join(); + ::_exit(1); // The first insertion must reach the publication hook. + } + std::thread second(write, 1); + { + std::unique_lock lock(gate); + ready_cv.wait(lock, [&] { return ready == 2; }); + } + if (exited) second.join(); + // Independent oracle: each writer has six entries, two deletions, one merge, + // 25 user-key bytes plus six tags, and value_size + 2 real value bytes. + ASSERT_EQ(cfd->mem()->num_entries(), 12U); + ASSERT_EQ(cfd->mem()->num_deletes(), 4U); + ASSERT_EQ(cfd->mem()->num_merges(), 2U); + ASSERT_EQ(cfd->mem()->raw_key_size(), 146U); + ASSERT_EQ(cfd->mem()->raw_value_size(), 2 * (value_size + 2)); + if (arg_[2] == 'F') { + FlushOptions flush; + flush.wait = true; + ASSERT_OK(child_db->Flush(flush)); + ColumnFamilyMetaData meta; + child_db->GetColumnFamilyMetaData(&meta); + ASSERT_EQ(meta.blob_files.size(), 1U); + ASSERT_EQ(meta.blob_files[0].total_blob_count, 2U); + ASSERT_EQ(meta.blob_files[0].total_blob_bytes, 2 * value_size); + } + ::_exit(42); +} + +TEST_F(DBCsppCrashSafeTest, PersistentWalStatsNormalConversion) { + Close(); + for (const char* config : {"CPF", "OPF"}) { + SCOPED_TRACE(config); + Options options = PersistentStatsOptions(dbname_, config); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "PersistentMemTableStats", config), 42); + } +} + +TEST_F(CrashChild, DISABLED_InterruptedWalStatsRecovery) { + Options options = PersistentStatsOptions(dbname_, arg_); + std::string active; + ASSERT_OK(ReadFileToString(options.env, dbname_ + "/stats-active", &active)); + uint64_t file_number; + FileType file_type; + ASSERT_TRUE(ParseFileName(active.substr(active.find_last_of('/') + 1), &file_number, &file_type)); + InternalKeyComparator icmp(options.comparator); + ImmutableOptions ioptions(options); + MutableCFOptions moptions(options); + IntTblPropCollectorFactories collectors; + TableBuilderOptions tbo(ioptions, moptions, icmp, &collectors, + options.compression, options.compression_opts, 0, + kDefaultColumnFamilyName, 0); + uint64_t next_blob = 800000; + tbo.generate_file_no = [&] { return next_blob++; }; + tbo.add_blob_file = [](BlobFileAddition) {}; + const char* hooks[] = {"MemTableStats::RecoverWals:AfterReset", + "MemTableStats::RecoverWals:AfterCount", + "MemTableStats::RecoverWals:AfterBytes", + "MemTableStats::RecoverWals:AfterNode"}; + SyncPoint::GetInstance()->SetCallBack(hooks[arg_[3] - '0'], [](void*) { ::_exit(42); }); + SyncPoint::GetInstance()->EnableProcessing(); + FileMetaData meta; + meta.fd = FileDescriptor(file_number, 0, 0); + meta.fd.largest_seqno = kMaxSequenceNumber; + ASSERT_OK(options.memtable_factory->RecoverCrashSafeMemTableToSST(active, &meta, tbo)); + ::_exit(1); // Each selected hook must interrupt actual WAL aggregation. +} + +TEST_F(DBCsppCrashSafeTest, PersistentMemTableStatsSurviveWriterLifetime) { + Close(); + for (const char* config : {"CNL", "CNE", "CPL", "CPE", "CSL", "CSE", + "ONL", "ONE", "OPL", "OPE", "OSL", "OSE", + "CNB", "CNA", "ONB", "ONA", "CPT", "OPT"}) { + SCOPED_TRACE(config); + Options options = PersistentStatsOptions(dbname_, config); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "PersistentMemTableStats", config), 42); + std::string active; + ASSERT_OK(ReadFileToString(env_, dbname_ + "/stats-active", &active)); + if (config[1] == 'P' && (config[2] == 'T' || config[2] == 'E')) { + // Two writers: live after threshold reporting, or already exited. + for (char boundary : {'0', '1', '2', '3'}) { + ASSERT_EQ(RunCrashChild(dbname_, "InterruptedWalStatsRecovery", + std::string(config) + boundary), 42); + } + } + const uint64_t batches = config[2] == 'B' || config[2] == 'A' ? 1 : 2; + const uint64_t value_size = config[2] == 'T' ? 600 * 1024 : 128; + uint64_t file_number; + FileType file_type; + ASSERT_TRUE(ParseFileName(active.substr(active.find_last_of('/') + 1), + &file_number, &file_type)); + ASSERT_EQ(file_type, kTableFile); + FileMetaData meta; + meta.fd = FileDescriptor(file_number, 0, 0); + meta.fd.largest_seqno = batches * 6; + InternalKeyComparator icmp(options.comparator); + ImmutableOptions ioptions(options); + MutableCFOptions moptions(options); + IntTblPropCollectorFactories collectors; + TableBuilderOptions tbo(ioptions, moptions, icmp, &collectors, + options.compression, options.compression_opts, 0, + kDefaultColumnFamilyName, 0); + uint64_t next_blob = 900000; + tbo.generate_file_no = [&] { return next_blob++; }; + std::vector blobs; + tbo.add_blob_file = [&](BlobFileAddition blob) { blobs.push_back(blob); }; + // Every conversion rebuilds header totals from durable writer counters. + for (int attempt = 0; attempt < 2; attempt++) { + SCOPED_TRACE(attempt); + blobs.clear(); + int recoveries = 0; + SyncPoint::GetInstance()->SetCallBack( + "MemTableStats::RecoverWals:AfterReset", + [&](void*) { recoveries++; }); + SyncPoint::GetInstance()->EnableProcessing(); + Status s = options.memtable_factory->RecoverCrashSafeMemTableToSST(active, &meta, tbo); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearCallBack("MemTableStats::RecoverWals:AfterReset"); + ASSERT_OK(s); + ASSERT_EQ(recoveries, 1); + ASSERT_EQ(meta.num_entries, batches * 6); + ASSERT_EQ(meta.num_deletions, batches * 2); + ASSERT_EQ(meta.num_merges, batches); + ASSERT_EQ(meta.raw_key_size, batches * 73); + ASSERT_EQ(meta.raw_value_size, batches * (value_size + 2)); + ASSERT_EQ(blobs.size(), config[1] == 'N' ? 0U : 1U); + for (const auto& blob : blobs) { + ASSERT_EQ(blob.GetTotalBlobCount(), batches); + ASSERT_EQ(blob.GetTotalBlobBytes(), batches * value_size); + } + if (config[1] != 'N') { + const int fd = ::open(active.c_str(), O_RDONLY); + ASSERT_GE(fd, 0); + const size_t offset = config[0] == 'O' + ? offsetof(terark::OSL_MmapHeader, reserved) + 2 * sizeof(uint32_t) + : offsetof(terark::DFA_MmapHeader, reserved) + 4 * sizeof(uint32_t); + uint64_t wal[3]; // fileno, cnt, bytes + const ssize_t n = ::pread(fd, wal, sizeof(wal), offset); + const int close_result = ::close(fd); + ASSERT_EQ(n, static_cast(sizeof(wal))); + ASSERT_EQ(close_result, 0); + ASSERT_EQ(wal[1], batches); + ASSERT_EQ(wal[2], batches * value_size); + } + SstFileReader reader(options); + ASSERT_OK(reader.Open(active)); + const auto properties = reader.GetTableProperties(); + ASSERT_EQ(properties->num_entries, batches * 6); + ASSERT_EQ(properties->num_deletions, batches * 2); + ASSERT_EQ(properties->num_merge_operands, batches); + ASSERT_EQ(properties->raw_key_size, batches * 73); + ASSERT_EQ(properties->raw_value_size, batches * (value_size + 2)); + if (config[1] != 'N') { + const auto& compression = properties->compression_options; + unsigned long long cnt, bytes; + ASSERT_EQ(sscanf(compression.c_str() + compression.find(';') + 1, + "%*u:%*u:%llu:%llu", &cnt, &bytes), 2); + ASSERT_EQ(cnt, batches); + ASSERT_EQ(bytes, batches * value_size); + } + } // Destroy the reader before recovering the same file again. + } +} + +TEST_F(DBCsppCrashSafeTest, PersistentMemTableStatsConcurrentAndDuplicateAdds) { + Close(); + for (const char* config : {"CNE", "ONE"}) { + SCOPED_TRACE(config); + Options options = PersistentStatsOptions(dbname_, config); + Destroy(options); + ASSERT_OK(env_->CreateDirIfMissing(dbname_)); + InternalKeyComparator icmp(options.comparator); + ImmutableOptions ioptions(options); + MutableCFOptions moptions(options); + WriteBufferManager wb(options.db_write_buffer_size); + auto mem = std::make_unique(icmp, ioptions, moptions, &wb, kMaxSequenceNumber, 0, 1); + constexpr size_t writers = 4, per_writer = 16; + std::atomic ready{0}; + std::atomic go{false}; + MemTablePostProcessInfo counters[writers]; + std::vector threads; + for (size_t id = 0; id < writers; id++) { + threads.emplace_back([&, id] { + ready++; + while (!go.load()) std::this_thread::yield(); + void* hint = nullptr; + for (size_t i = 0; i < per_writer; i++) { + std::string key(3, 'a'); + key[0] = char('a' + id); + key[1] = char('A' + i); + ASSERT_OK(mem->Add(id * per_writer + i + 1, kTypeValue, key, + std::string(128, 'v'), nullptr, true, + &counters[id], id % 2 ? &hint : nullptr)); + } + mem->FinishHint(hint); + }); + } + while (ready.load() != writers) std::this_thread::yield(); + go.store(true); + for (auto& thread : threads) thread.join(); + for (const auto& counter : counters) mem->BatchPostProcess(counter); + ASSERT_OK(mem->Add(1000, kTypeValue, "dup", "v", nullptr)); + ASSERT_TRUE(mem->Add(1000, kTypeValue, "dup", "v", nullptr).IsTryAgain()); + // Four writers add distinct three-byte keys; the rejected duplicate adds + // neither a record nor bytes to either the wrapper or persistent counters. + constexpr uint64_t entries = writers * per_writer + 1; + constexpr uint64_t key_bytes = entries * (3 + 8); + constexpr uint64_t value_bytes = writers * per_writer * 128 + 1; + ASSERT_EQ(mem->num_entries(), entries); + ASSERT_EQ(mem->raw_key_size(), key_bytes); + ASSERT_EQ(mem->raw_value_size(), value_bytes); + mem->MarkImmutable(); + const std::string leftover = MakeTableFileName(dbname_, 2); + CopyFile(MakeTableFileName(dbname_, 1), leftover); + mem.reset(); + IntTblPropCollectorFactories collectors; + TableBuilderOptions tbo(ioptions, moptions, icmp, &collectors, + options.compression, options.compression_opts, 0, + kDefaultColumnFamilyName, 0); + FileMetaData meta; + meta.fd = FileDescriptor(2, 0, 0); + meta.fd.largest_seqno = 1000; + ASSERT_OK(options.memtable_factory->RecoverCrashSafeMemTableToSST(leftover, &meta, tbo)); + ASSERT_EQ(meta.num_entries, entries); + ASSERT_EQ(meta.num_deletions, 0U); + ASSERT_EQ(meta.num_merges, 0U); + ASSERT_EQ(meta.raw_key_size, key_bytes); + ASSERT_EQ(meta.raw_value_size, value_bytes); + SstFileReader reader(options); + ASSERT_OK(reader.Open(leftover)); + const auto properties = reader.GetTableProperties(); + ASSERT_EQ(properties->num_entries, entries); + ASSERT_EQ(properties->num_deletions, 0U); + ASSERT_EQ(properties->num_merge_operands, 0U); + ASSERT_EQ(properties->raw_key_size, key_bytes); + ASSERT_EQ(properties->raw_value_size, value_bytes); + } +} + +TEST_F(CrashChild, DISABLED_LogRefRecoveryIgnoresCounters) { + Options options = LogRefCrashOptions(dbname_, arg_); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + const size_t count = arg_[2] == 'L' ? 5000 : 3; + for (size_t i = 0; i < count; ++i) { + ASSERT_OK(child_db->Put(WriteOptions(), std::to_string(i % 2), + std::string(128, 'v'))); + } + ASSERT_OK(child_db->Put(WriteOptions(), "inline", "v")); + auto* cfd = static_cast(child_db)->GetVersionSet() + ->GetColumnFamilySet()->GetDefault(); + ASSERT_OK(WriteStringToFile( + options.env, MakeTableFileName(dbname_, cfd->mem()->GetFileNumber()), + dbname_ + "/active-file")); + ::_exit(42); +} + +TEST_F(DBCsppCrashSafeTest, LogRefRecoveryIgnoresCounters) { + for (const char* config : {"CPS", "CPL", "CSS", "CSL", + "OPS", "OPL", "OSS", "OSL"}) { + SCOPED_TRACE(config); + Close(); + Options options = LogRefCrashOptions(dbname_, config); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "LogRefRecoveryIgnoresCounters", config), 42); + const auto leftovers = ListLeftovers(options, dbname_); + ASSERT_EQ(leftovers.size(), 2U); + std::string active; + ASSERT_OK(ReadFileToString(env_, dbname_ + "/active-file", &active)); + ASSERT_EQ(std::count(leftovers.begin(), leftovers.end(), active), 1); + const int fd = ::open(active.c_str(), O_RDWR); + ASSERT_GE(fd, 0); + // Recovery rebuilds header totals from durable writer counters. + const size_t offset = config[0] == 'O' + ? offsetof(terark::OSL_MmapHeader, reserved) + 2 * sizeof(uint32_t) + : offsetof(terark::DFA_MmapHeader, reserved) + 4 * sizeof(uint32_t); + uint64_t wal[3]; // fileno, cnt, bytes + const ssize_t n = ::pread(fd, wal, sizeof(wal), offset); + ASSERT_EQ(n, static_cast(sizeof(wal))); + if (config[2] == 'S') { + // Deliberately clear header counters. Recovery must retain the real WAL + // references and reconstruct totals from durable writer statistics. + wal[1] = 0; + wal[2] = 0; + ASSERT_EQ(::pwrite(fd, wal, sizeof(wal), offset), + static_cast(sizeof(wal))); + } + ::close(fd); + ASSERT_NE(wal[0], 0U); + for (int reopen = 0; reopen < 2; ++reopen) { + ASSERT_OK(TryReopen(options)); + ASSERT_GT(CountL0(db_), 0); + ColumnFamilyMetaData cf_meta; + db_->GetColumnFamilyMetaData(&cf_meta); + ASSERT_EQ(cf_meta.blob_files.size(), 1U); + const uint64_t count = config[2] == 'L' ? 5000 : 3; + ASSERT_EQ(cf_meta.blob_files[0].total_blob_count, count); + ASSERT_EQ(cf_meta.blob_files[0].total_blob_bytes, count * 128); + ASSERT_EQ(Get("0"), std::string(128, 'v')); + ASSERT_EQ(Get("1"), std::string(128, 'v')); + ASSERT_EQ(Get("inline"), "v"); + Close(); + } + } +} + +TEST_F(CrashChild, DISABLED_LogRefRecoveryMultipleWals) { + Options options = LogRefCrashOptions(dbname_, arg_); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + Options auxiliary = options; + ColumnFamilyHandle* handle = nullptr; + ASSERT_OK(child_db->CreateColumnFamily(auxiliary, "rotate", &handle)); + auto* impl = static_cast(child_db); + auto* cfd = impl->GetVersionSet()->GetColumnFamilySet()->GetColumnFamily( + handle->GetID()); + for (int i = 0; i < 3; ++i) { + ASSERT_OK(child_db->Put(WriteOptions(), std::to_string(i), + std::string(128, 'a' + i))); + if (i != 2) { + ASSERT_OK(impl->TEST_SwitchMemtable(cfd)); + } + } + // Only the auxiliary CF switches: all three WAL slots belong to one primary + // memtable, exercising the parallel mapping array rather than three memtables. + ::_exit(42); +} + +TEST_F(DBCsppCrashSafeTest, LogRefRecoveryMultipleWals) { + for (const char* config : {"CPS", "CSS", "OPS", "OSS"}) { + SCOPED_TRACE(config); + Close(); + Options options = LogRefCrashOptions(dbname_, config); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "LogRefRecoveryMultipleWals", config), 42); + const auto leftovers = ListLeftovers(options, dbname_); + // The empty rotating CF is now FileMmap too; its registered sources follow + // the primary CF's earlier file number in the complete inventory. + ASSERT_GE(leftovers.size(), 2U); + ASSERT_OK(WriteStringToFile(env_, leftovers.front(), dbname_ + "/stats-active")); + for (char boundary : {'0', '1', '2', '3'}) { + ASSERT_EQ(RunCrashChild(dbname_, "InterruptedWalStatsRecovery", + std::string(config) + boundary), 42); + } + const int fd = ::open(leftovers.front().c_str(), O_RDONLY); + ASSERT_GE(fd, 0); + uint32_t num_wals = 0; + const size_t num_wals_offset = config[0] == 'O' + ? offsetof(terark::OSL_MmapHeader, reserved) + sizeof(uint32_t) + : offsetof(terark::DFA_MmapHeader, reserved) + 2 * sizeof(uint32_t); + ASSERT_EQ(::pread(fd, &num_wals, sizeof(num_wals), + num_wals_offset), + static_cast(sizeof(num_wals))); + ::close(fd); + ASSERT_EQ(num_wals, 3U); + for (int reopen = 0; reopen < 2; ++reopen) { + ASSERT_OK(TryReopenWithColumnFamilies({"default", "rotate"}, options)); + ASSERT_EQ(CountL0(db_), 1); + ColumnFamilyMetaData meta; + db_->GetColumnFamilyMetaData(&meta); + ASSERT_EQ(meta.blob_files.size(), 3U); + for (const auto& blob : meta.blob_files) { + ASSERT_EQ(blob.total_blob_count, 1U); + ASSERT_EQ(blob.total_blob_bytes, 128U); + } + std::unique_ptr it(db_->NewIterator(ReadOptions())); + it->SeekToFirst(); + for (int i = 0; i < 3; ++i) { + const std::string value(128, 'a' + i); + ASSERT_EQ(Get(0, std::to_string(i)), value); + ASSERT_TRUE(it->Valid()); + ASSERT_EQ(it->key().ToString(), std::to_string(i)); + ASSERT_EQ(it->value().ToString(), value); + it->Next(); + } + ASSERT_FALSE(it->Valid()); + ASSERT_OK(it->status()); + it.reset(); + Close(); + } + } +} + +TEST_F(CrashChild, DISABLED_ChangedFactoryWithOtherCfLeftover) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + Options other = options; + if (arg_ == "OSL") { + SetupOsl(&options, true); + } else { + SetupOsl(&other, true); + } + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ColumnFamilyHandle* handle = nullptr; + ASSERT_OK(child_db->CreateColumnFamily(other, "other", &handle)); + ASSERT_OK(child_db->Put(WriteOptions(), "key", "primary")); + ASSERT_OK(child_db->Put(WriteOptions(), handle, "key", "other")); + ::_exit(1); +} + +TEST_F(DBCsppCrashSafeTest, ChangedFactoryWithOtherCfLeftoverUsesFullWal) { + for (bool was_osl : {false, true}) { + SCOPED_TRACE(was_osl); + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (!was_osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "ChangedFactoryWithOtherCfLeftover", + was_osl ? "OSL" : "CSPP"), 1); + // Both CFs now use the other's factory, so default's old leftover is absent. + ASSERT_OK(TryReopenWithColumnFamilies({"default", "other"}, options)); + ASSERT_EQ(Get(0, "key"), "primary"); + ASSERT_EQ(Get(1, "key"), "other"); + } +} + +TEST_F(CrashChild, DISABLED_OslLeftoverWithWriterLock) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + SetupOsl(&options, true); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + const std::string key = "locked-osl-key"; + ASSERT_OK(child_db->Put(WriteOptions(), key, "value")); + auto* cfd = static_cast(child_db)->GetVersionSet() + ->GetColumnFamilySet()->GetDefault(); + const auto active = MakeTableFileName(dbname_, cfd->mem()->GetFileNumber()); + int fd = ::open(active.c_str(), O_RDWR); + ASSERT_GE(fd, 0); + char data[4096]; + const ssize_t n = ::pread(fd, data, sizeof(data), 0); + ASSERT_GT(n, 0); + const char* p = std::search(data, data + n, key.begin(), key.end()); + ASSERT_NE(p, data + n); + ASSERT_GE(p - data, 12); + ASSERT_EQ(DecodeFixed32(p - 4), key.size()); + // ValueVec precedes the length-prefixed key. Simulate InsertDup holding its + // lock while the published COW array is still available to readers. + ASSERT_EQ(DecodeFixed32(p - 12), 1U); + char locked_num[4]; + EncodeFixed32(locked_num, 0x80000001U); + ASSERT_EQ(::pwrite(fd, locked_num, sizeof(locked_num), p - 12 - data), 4); + ::close(fd); + std::string value; + ASSERT_OK(child_db->Get(ReadOptions(), key, &value)); + ASSERT_EQ(value, "value"); + std::unique_ptr it(child_db->NewIterator(ReadOptions())); + it->SeekToFirst(); + ASSERT_TRUE(it->Valid()); + ASSERT_EQ(it->key(), key); + ASSERT_EQ(it->value(), "value"); + ::_exit(1); +} + +TEST_F(DBCsppCrashSafeTest, OslLeftoverWithWriterLock) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + SetupOsl(&options, true); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "OslLeftoverWithWriterLock"), 1); + for (int i = 0; i != 2; ++i) { + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("locked-osl-key"), "value"); + std::unique_ptr it(db_->NewIterator(ReadOptions())); + it->SeekToFirst(); + ASSERT_TRUE(it->Valid()); + ASSERT_EQ(it->key(), "locked-osl-key"); + ASSERT_EQ(it->value(), "value"); + it.reset(); + Close(); + } +} + +TEST_F(CrashChild, DISABLED_MixedDumpMemUsesFullWal) { + Options options = BaseCrashSafeOptions(dbname_, false, false); + const bool osl = arg_ == "OSL"; + if (osl) SetupOsl(&options, true); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + Options dump = options; + dump.memtable_factory = EasyNewMemTableRep( + osl ? "OffsetSkipList" : "CSPPMemTab", + R"({"mem_cap":16777216,"convert_to_sst":"kDumpMem"})"); + ColumnFamilyHandle* handle = nullptr; + ASSERT_OK(child_db->CreateColumnFamily(dump, "dump", &handle)); + ASSERT_OK(child_db->Put(WriteOptions(), "mmap", "1")); + ASSERT_OK(child_db->Put(WriteOptions(), handle, "dump", "2")); + ::_exit(1); +} + +TEST_F(DBCsppCrashSafeTest, MixedDumpMemUsesFullWal) { + for (bool osl : {false, true}) { + SCOPED_TRACE(osl); + Close(); + Options options = BaseCrashSafeOptions(dbname_, false, false); + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "MixedDumpMemUsesFullWal", + osl ? "OSL" : "CSPP"), 1); + Options dump = options; + dump.memtable_factory = EasyNewMemTableRep( + osl ? "OffsetSkipList" : "CSPPMemTab", + R"({"mem_cap":16777216,"convert_to_sst":"kDumpMem"})"); + ASSERT_OK(TryReopenWithColumnFamilies( + {"default", "dump"}, std::vector{options, dump})); + ASSERT_EQ(Get(0, "mmap"), "1"); + ASSERT_EQ(Get(1, "dump"), "2"); + } +} + +TEST_F(CrashChild, DISABLED_CloseConvertsLeftovers) { + ASSERT_EQ(arg_.size(), 2U); + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (arg_[0] == '1') SetupOsl(&options, true); + options.atomic_flush = arg_[1] == '1'; + options.min_write_buffer_number_to_merge = 3; + DB* db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &db)); + const auto closing_thread = std::this_thread::get_id(); + std::atomic converts{0}; + SyncPoint::GetInstance()->SetCallBack( + "FlushJob::ConvertToSST:Status", [&](void*) { + EXPECT_NE(std::this_thread::get_id(), closing_thread); + converts.fetch_add(1); + }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(db->Put(WriteOptions(), "k1", "v1")); + ASSERT_OK(static_cast(db)->TEST_SwitchMemtable()); + ASSERT_OK(db->Put(WriteOptions(), "k2", "v2")); + ASSERT_OK(db->Close()); + delete db; + ASSERT_GE(converts.load(), 2); + ASSERT_FALSE(HasFailure()); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, CloseConvertsLeftovers) { + Close(); + for (bool osl : {false, true}) { + for (bool atomic : {false, true}) { + SCOPED_TRACE(osl); + SCOPED_TRACE(atomic); + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + options.atomic_flush = atomic; + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "CloseConvertsLeftovers", + std::to_string(osl) + std::to_string(atomic)), 0); + ASSERT_EQ(ListLeftovers(options, dbname_).size(), 3U); + for (const auto& path : ListLeftovers(options, dbname_)) + ASSERT_OK(env_->FileExists(path)); + ASSERT_OK(env_->FileExists(CrashSafePubSeqFileName(dbname_))); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("k1"), "v1"); + ASSERT_EQ(Get("k2"), "v2"); + ASSERT_GE(CountL0(db_), 1); + Close(); + ASSERT_EQ(ListLeftovers(options, dbname_).size(), 2U); + for (const auto& path : ListLeftovers(options, dbname_)) + ASSERT_OK(env_->FileExists(path)); + } + } +} +#endif + +TEST_F(DBCsppCrashSafeTest, AvoidFlushDuringShutdownKeepsRegisteredMemTable) { + Close(); + for (bool osl : {false, true}) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&options, true); + options.avoid_flush_during_shutdown = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "v")); + const auto registered = ListLeftovers(options, dbname_); + ASSERT_EQ(registered.size(), 2U); + auto* cfd = dbfull()->GetVersionSet()->GetColumnFamilySet()->GetDefault(); + for (auto* mem : {cfd->mem(), cfd->PeekPrecreatedMemtable()}) { + ASSERT_NE(mem, nullptr); + ASSERT_TRUE(mem->IsFileRegistered()); + const auto path = MakeTableFileName(dbname_, mem->GetFileNumber()); + ASSERT_EQ(std::count(registered.begin(), registered.end(), path), 1); + } + Close(); + ASSERT_EQ(ListLeftovers(options, dbname_), registered); + for (const auto& path : registered) ASSERT_OK(env_->FileExists(path)); + ASSERT_OK(env_->FileExists(CrashSafePubSeqFileName(dbname_))); + options.avoid_flush_during_shutdown = false; + std::atomic converted{0}; + SyncPoint::GetInstance()->SetCallBack( + "MemTableRep::ConvertToSST:After", [&](void*) { ++converted; }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(options)); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_EQ(converted.load(), 1); + ASSERT_EQ(Get("k"), "v"); + ASSERT_EQ(CountL0(db_), 1); + Close(); + for (const auto& path : ListLeftovers(options, dbname_)) + ASSERT_OK(env_->FileExists(path)); + + // Second cycle: the reopen above converted the leftovers into L0 and + // retired their inventory entries. Closing again with + // avoid_flush_during_shutdown must leave a freshly registered MemTable + // behind, so that this open converts it too. Were the retired inventory or + // the post-recovery MemTable left unregistered, the cycle would silently + // degrade to a full WAL replay: no error, no data loss, just the end of + // crash-safe recovery for this DB. + options.avoid_flush_during_shutdown = true; + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k2", "v2")); + auto* cfd2 = dbfull()->GetVersionSet()->GetColumnFamilySet()->GetDefault(); + ASSERT_NE(cfd2->mem(), nullptr); + ASSERT_TRUE(cfd2->mem()->IsFileRegistered()); + const auto before_second_close = ListLeftovers(options, dbname_); + ASSERT_EQ(before_second_close.size(), registered.size()); + Close(); + ASSERT_EQ(ListLeftovers(options, dbname_), before_second_close); + for (const auto& path : before_second_close) + ASSERT_OK(env_->FileExists(path)); + ASSERT_OK(env_->FileExists(CrashSafePubSeqFileName(dbname_))); + converted.store(0); + SyncPoint::GetInstance()->SetCallBack( + "MemTableRep::ConvertToSST:After", [&](void*) { ++converted; }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(options)); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_EQ(converted.load(), 1); + ASSERT_EQ(Get("k"), "v"); + ASSERT_EQ(Get("k2"), "v2"); + ASSERT_EQ(CountL0(db_), 2); + Close(); + } +} + +TEST_F(DBCsppCrashSafeTest, AvoidFlushCloseReopenDoesNotProbeWal) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.avoid_flush_during_shutdown = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "v")); + Close(); + int probes = 0; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::RecoverLogFiles:ProbeWalFormat", + [&probes](void*) { probes++; }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(options)); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_EQ(probes, 0); + ASSERT_EQ(Get("k"), "v"); +} + +TEST_F(DBCsppCrashSafeTest, FreshSidecarProbesOnlyOlderWal) { + Close(); + Options off = BaseCrashSafeOptions(dbname_, false, false); + off.avoid_flush_during_shutdown = true; + Destroy(off); + ASSERT_OK(TryReopen(off)); + ASSERT_OK(Put("a", "1")); + Close(); + Options on = BaseCrashSafeOptions(dbname_, true, false); + on.avoid_flush_during_shutdown = true; + std::vector probed; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::RecoverLogFiles:ProbeWalFormat", [&probed](void* arg) { + probed.push_back(*static_cast(arg)); + }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(on)); + ASSERT_EQ(probed.size(), 1U); + ASSERT_OK(Put("b", "2")); + Close(); + // A normal close retains registered sources, enabling prefix conversion. + // Force full WAL replay here to exercise mixed old/new WAL format probing: + // only WALs older than the fresh sidecar's kind boundary need inspection. + const auto registered = ListLeftovers(on, dbname_); + ASSERT_FALSE(registered.empty()); + ASSERT_OK(env_->DeleteFile(registered.front())); + probed.clear(); + ASSERT_OK(TryReopen(on)); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_GT(rec.kind_since_wal, 0U); + ASSERT_EQ(probed.size(), 1U); + ASSERT_LT(probed[0], rec.kind_since_wal); + ASSERT_EQ(Get("a"), "1"); + ASSERT_EQ(Get("b"), "2"); +} + +TEST_F(DBCsppCrashSafeTest, ClassicWalToLogIndexWithoutSidecarIsNotSupported) { + Close(); + std::string value(100000, '\0'); + for (size_t i = 0; i < value.size(); ++i) { + value[i] = static_cast('a' + i % 23); + } + for (bool osl : {false, true}) { + SCOPED_TRACE(osl ? "OSL" : "CSPP"); + for (bool paranoid : {false, true}) { + SCOPED_TRACE(paranoid); + Options classic = BaseCrashSafeOptions(dbname_, false, false); + if (osl) SetupOsl(&classic, true); + classic.avoid_flush_during_shutdown = true; + classic.paranoid_checks = paranoid; + Destroy(classic); + ASSERT_OK(TryReopen(classic)); + ASSERT_OK(Put("large", value)); + Close(); + ASSERT_TRUE(env_->FileExists(CrashSafePubSeqFileName(dbname_)).IsNotFound()); + + Options log_index = classic; + log_index.memtable_crash_safe_recover = true; + log_index.memtable_as_log_index = true; + for (int attempt = 0; attempt < 2; ++attempt) { + const Status s = TryReopen(log_index); + ASSERT_TRUE(s.IsNotSupported()) << s.ToString(); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.wal_offset_kind, 2U); + ASSERT_EQ(rec.kind_since_wal, 0U); + ASSERT_EQ(rec.generation & 1, 0U); + } + + // Failed Open must not force KindPrep with the unestablished log-index + // kind. Reopen in the original format without deleting the sidecar. + classic.memtable_crash_safe_recover = true; + ASSERT_OK(TryReopen(classic)); + ASSERT_EQ(Get("large"), value); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.wal_offset_kind, 1U); + ASSERT_GT(rec.kind_since_wal, 0U); + Close(); + + // An established sidecar still permits the existing KindPrep switch. + ASSERT_OK(TryReopen(log_index)); + ASSERT_EQ(Get("large"), value); + Close(); + } + } +} + +TEST_F(DBCsppCrashSafeTest, LogIndexWalWithoutSidecarRecoversAsClassic) { + Close(); + const std::string value(100000, 'v'); + for (bool osl : {false, true}) { + SCOPED_TRACE(osl ? "OSL" : "CSPP"); + Options options = BaseCrashSafeOptions(dbname_, false, true); + if (osl) SetupOsl(&options, true); + options.avoid_flush_during_shutdown = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("large", value)); + Close(); + ASSERT_TRUE(env_->FileExists(CrashSafePubSeqFileName(dbname_)).IsNotFound()); + + options.memtable_crash_safe_recover = true; + options.memtable_as_log_index = false; + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("large"), value); + Close(); + } +} + +TEST_F(DBCsppCrashSafeTest, RecoverOffDeletesStaleSidecar) { + Close(); + for (bool osl : {false, true}) { + SCOPED_TRACE(osl); + Options on = BaseCrashSafeOptions(dbname_, true, false); + if (osl) SetupOsl(&on, true); + on.avoid_flush_during_shutdown = true; + Destroy(on); + ASSERT_OK(TryReopen(on)); + ASSERT_OK(Put("k", "v")); + Close(); + ASSERT_OK(env_->FileExists(CrashSafePubSeqFileName(dbname_))); + ASSERT_FALSE(ListLeftovers(on, dbname_).empty()); + Options off = on; + off.memtable_crash_safe_recover = false; + ASSERT_OK(TryReopen(off)); + ASSERT_TRUE(env_->FileExists(CrashSafePubSeqFileName(dbname_)).IsNotFound()); + for (auto* cfd : *dbfull()->GetVersionSet()->GetColumnFamilySet()) { + ASSERT_TRUE(cfd->GetMemTableFiles().empty()); + } + ASSERT_TRUE(ListLeftovers(off, dbname_).empty()); + ASSERT_EQ(Get("k"), "v"); + Close(); + ASSERT_OK(TryReopen(off)); + for (auto* cfd : *dbfull()->GetVersionSet()->GetColumnFamilySet()) { + ASSERT_TRUE(cfd->GetMemTableFiles().empty()); + } + ASSERT_EQ(Get("k"), "v"); + Close(); + } +} + +TEST_F(DBCsppCrashSafeTest, OddGenerationKindSwitchUsesKindPrep) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.avoid_flush_during_shutdown = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "v")); + Close(); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_TRUE(SetPublishedSeqGeneration(dbname_, rec.generation | 1)); + Options log_index = BaseCrashSafeOptions(dbname_, true, true); + log_index.avoid_flush_during_shutdown = true; + int prep = 0; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::Open::KindPrep:AfterOpenBeforeFlush", + [&prep](void*) { prep++; }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(log_index)); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_EQ(prep, 1); + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.wal_offset_kind, 2U); + ASSERT_EQ(Get("k"), "v"); +} + +TEST_F(DBCsppCrashSafeTest, DirectDBImplOpenKindMismatchFails) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.avoid_flush_during_shutdown = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "v")); + Close(); + Options log_index = BaseCrashSafeOptions(dbname_, true, true); + std::vector cfs = { + ColumnFamilyDescriptor(kDefaultColumnFamilyName, log_index)}; + std::vector handles; + DB* db = nullptr; + const Status s = DBImpl::Open(DBOptions(log_index), dbname_, cfs, &handles, + &db, false /*seq_per_batch*/, + true /*batch_per_txn*/); + ASSERT_TRUE(s.IsInvalidArgument()) << s.ToString(); + ASSERT_EQ(db, nullptr); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.wal_offset_kind, 1U); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("k"), "v"); +} + +TEST_F(DBCsppCrashSafeTest, OddGenerationPublishesEvenAfterWalRecovery) { + for (bool log_index : {false, true}) { + SCOPED_TRACE(log_index); + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, log_index); + options.avoid_flush_during_shutdown = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("before", "recovery")); + Close(); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_TRUE(SetPublishedSeqGeneration(dbname_, rec.generation | 1)); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("before"), "recovery"); + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.generation % 2, 0U); + ASSERT_EQ(rec.pubseq, 0U); +#if !defined(__AVX__) + int odd = 0; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterOddGeneration", [&](void*) { + PublishedSeqOnDisk in_progress; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &in_progress)); + ASSERT_EQ(in_progress.generation % 2, 1U); + ++odd; + }); + SyncPoint::GetInstance()->EnableProcessing(); +#endif + ASSERT_OK(Put("after", "recovery")); +#if !defined(__AVX__) + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_EQ(odd, 1); +#endif + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.generation % 2, 0U); + ASSERT_EQ(rec.pubseq, dbfull()->GetLatestSequenceNumber()); + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("before"), "recovery"); + ASSERT_EQ(Get("after"), "recovery"); + } +} + +TEST_F(DBCsppCrashSafeTest, LogIndexOffRecoverStillConverts) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("only", "recover")); + ASSERT_OK(dbfull()->TEST_SwitchMemtable()); + ASSERT_OK(Put("tail", "wal")); + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("only"), "recover"); + ASSERT_EQ(Get("tail"), "wal"); +} + +TEST_F(DBCsppCrashSafeTest, AtomicFlushCloseConverts) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.atomic_flush = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "v")); + Close(); + ASSERT_EQ(ListLeftovers(options, dbname_).size(), 2U); + for (const auto& path : ListLeftovers(options, dbname_)) + ASSERT_OK(env_->FileExists(path)); + ASSERT_OK(env_->FileExists(CrashSafePubSeqFileName(dbname_))); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("k"), "v"); + ASSERT_GE(CountL0(db_), 1); +} + +#if !defined(OS_WIN) +TEST_F(CrashChild, DISABLED_AtomicFlushAfterCommitExitConverts) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.atomic_flush = true; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", + [](void* /*arg*/) { ::_exit(1); }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "pub", "yes")); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, AtomicFlushAfterCommitExitConverts) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.atomic_flush = true; + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "AtomicFlushAfterCommitExitConverts"), 1); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("pub"), "yes"); + ASSERT_GE(CountL0(db_), 1); +} + +TEST_F(CrashChild, DISABLED_AfterCommitExitConvertsAndReopens) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", + [](void* /*arg*/) { ::_exit(1); }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "pub", "yes")); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, AfterCommitExitConvertsAndReopens) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "AfterCommitExitConvertsAndReopens"), 1); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("pub"), "yes"); + ColumnFamilyMetaData metadata; + db_->GetColumnFamilyMetaData(&metadata); + ASSERT_EQ(metadata.levels[0].files.size(), 1U); + ASSERT_TRUE(metadata.levels[0].files[0].marked_for_compaction); +} + +// Non-AVX builds publish through the odd generation. +#if !defined(__AVX__) +TEST_F(CrashChild, DISABLED_AfterOddGenerationFallsBackToWal) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterOddGeneration", + [](void* /*arg*/) { ::_exit(1); }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "odd", "wal")); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, AfterOddGenerationFallsBackToWal) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "AfterOddGenerationFallsBackToWal"), 1); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("odd"), "wal"); +} +#endif // !__AVX__ + +TEST_F(CrashChild, DISABLED_AfterWriteToWALBeforePublishKeepsWalTail) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::WriteImpl:AfterWriteToWALBeforePublish", + [](void*) { ::_exit(1); }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(child_db->Put(WriteOptions(), "tail", "2")); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, AfterWriteToWALBeforePublishKeepsWalTail) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + { + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("first", "1")); + Close(); + } + ASSERT_EQ(RunCrashChild(dbname_, "AfterWriteToWALBeforePublishKeepsWalTail"), 1); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("first"), "1"); + ASSERT_EQ(Get("tail"), "2"); +} + +TEST_F(CrashChild, DISABLED_ConvertInjectFailureFallsBackToWal) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", + [](void*) { ::_exit(1); }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "inj", "ok")); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, ConvertInjectFailureFallsBackToWal) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "ConvertInjectFailureFallsBackToWal"), 1); + SyncPoint::GetInstance()->EnableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + SyncPoint::GetInstance()->SetCallBack( + "CrashSafeRecover::ConvertToSST:InjectStatus", [](void* arg) { + *static_cast(arg) = Status::IOError("inject convert"); + }); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("inj"), "ok"); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); +} + +TEST_F(CrashChild, DISABLED_SeekInjectFailureStillConverts) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", + [](void*) { ::_exit(1); }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "seek", "ok")); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, SeekInjectFailureStillConverts) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "SeekInjectFailureStillConverts"), 1); + SyncPoint::GetInstance()->EnableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + SyncPoint::GetInstance()->SetCallBack( + "CrashSafeRecover::SeekToFileOffset:InjectStatus", [](void* arg) { + *static_cast(arg) = IOStatus::IOError("inject seek"); + }); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("seek"), "ok"); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); +} + +class SeekFailureFS : public FileSystemWrapper { + public: + SeekFailureFS(const std::shared_ptr& fs, std::string path, + bool fail_read) + : FileSystemWrapper(fs), path_(std::move(path)), fail_read_(fail_read) {} + const char* Name() const override { return "SeekFailureFS"; } + bool armed = false; + int failures = 0; + int wal_opens = 0; + + class File : public FSSequentialFileOwnerWrapper { + public: + File(std::unique_ptr&& file, SeekFailureFS* fs) + : FSSequentialFileOwnerWrapper(std::move(file)), fs_(fs) {} + IOStatus Read(size_t n, const IOOptions& opts, Slice* result, char* scratch, + IODebugContext* dbg) override { + if (fs_->armed && fs_->fail_read_) { + fs_->armed = false; + IOStatus s = FSSequentialFileOwnerWrapper::Read( + std::min(n, 4), opts, result, scratch, dbg); + if (!s.ok()) return s; + fs_->failures++; + return IOStatus::IOError("seek read failed after consuming bytes"); + } + return FSSequentialFileOwnerWrapper::Read(n, opts, result, scratch, dbg); + } + IOStatus Skip(uint64_t n) override { + IOStatus s = FSSequentialFileOwnerWrapper::Skip(n); + if (s.ok() && n != 0 && fs_->armed && !fs_->fail_read_) { + fs_->armed = false; + fs_->failures++; + return IOStatus::IOError("seek skip failed after advancing file"); + } + return s; + } + private: + SeekFailureFS* fs_; + }; + + IOStatus NewSequentialFile(const std::string& fname, const FileOptions& opts, + std::unique_ptr* file, + IODebugContext* dbg) override { + IOStatus s = FileSystemWrapper::NewSequentialFile(fname, opts, file, dbg); + if (s.ok() && fname == path_) { + wal_opens++; + *file = std::make_unique(std::move(*file), this); + } + return s; + } + private: + const std::string path_; + const bool fail_read_; +}; + +TEST_F(CrashChild, DISABLED_SeekIOFailureReopensWal) { + Options options = BaseCrashSafeOptions(dbname_, true, arg_ == "log-index"); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "prefix", + std::string(log::kBlockSize * 2, 'p'))); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::WriteImpl:AfterWriteToWALBeforePublish", + [](void*) { ::_exit(42); }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(child_db->Put(WriteOptions(), "tail", "unpublished")); + ::_exit(1); +} + +TEST_F(DBCsppCrashSafeTest, SeekIOFailureReopensWal) { + Close(); + for (const std::string mode : {"read", "skip", "log-index"}) { + SCOPED_TRACE(mode); + Options options = BaseCrashSafeOptions(dbname_, true, mode == "log-index"); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "SeekIOFailureReopensWal", mode), 42); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + auto fs = std::make_shared( + options.env->GetFileSystem(), LogFileName(options.wal_dir, rec.wal_number), + mode == "read"); + auto fault_env = NewCompositeEnv(fs); + options.env = fault_env.get(); + options.log_readahead_size = 0; + SyncPoint::GetInstance()->SetCallBack( + "CrashSafeRecover::SeekToFileOffset:Before", + [&](void*) { fs->armed = true; }); + SyncPoint::GetInstance()->EnableProcessing(); + const Status s = TryReopen(options); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + if (s.ok()) { + EXPECT_EQ(Get("prefix"), std::string(log::kBlockSize * 2, 'p')); + EXPECT_EQ(Get("tail"), "unpublished"); + } + Close(); + ASSERT_OK(s); + ASSERT_EQ(fs->failures, 1); + ASSERT_GE(fs->wal_opens, 2); + } +} + +TEST_F(DBCsppCrashSafeTest, TornStampKindZeroRestampsAndProbesWal) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.avoid_flush_during_shutdown = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "v")); + Close(); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.magic, 0x5145534255505343ULL); + ASSERT_EQ(rec.wal_offset_kind, 1U); + ASSERT_TRUE(SetPublishedSeqWalOffsetKind(dbname_, 0)); + int probes = 0; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::RecoverLogFiles:ProbeWalFormat", + [&probes](void*) { probes++; }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(options)); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_GT(probes, 0); + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.wal_offset_kind, 1U); + ASSERT_EQ(Get("k"), "v"); +} + +TEST_F(CrashChild, DISABLED_UninitializedPublishedSeqStillOpens) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::MapPublishedSeqFile:AfterMmapBeforeStamp", + [](void*) { ::_exit(1); }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, UninitializedPublishedSeqStillOpens) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "UninitializedPublishedSeqStillOpens"), 1); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("after", "stamp")); + ASSERT_EQ(Get("after"), "stamp"); +} + +TEST_F(CrashChild, DISABLED_ZeroPublishedSeqWithLeftoverFallsBackToWal) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", + [](void*) { ::_exit(1); }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "k", "v")); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, ZeroPublishedSeqWithLeftoverFallsBackToWal) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "ZeroPublishedSeqWithLeftoverFallsBackToWal"), 1); + const auto before = ListLeftovers(options, dbname_); + ASSERT_FALSE(before.empty()); + ASSERT_TRUE(ZeroPublishedSeqFile(dbname_)); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("k"), "v"); + for (const auto& path : before) { + ASSERT_TRUE(env_->FileExists(path).IsNotFound()); + } +} + +TEST_F(CrashChild, DISABLED_LogIndexSeekInjectKeepsUnpublishedWalTail) { + Options options = BaseCrashSafeOptions(dbname_, true, true); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::WriteImpl:AfterWriteToWALBeforePublish", + [](void*) { ::_exit(1); }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(child_db->Put(WriteOptions(), "b", "2")); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, LogIndexSeekInjectKeepsUnpublishedWalTail) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, true); + Destroy(options); + { + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("a", "1")); + Close(); + } + ASSERT_EQ(RunCrashChild(dbname_, "LogIndexSeekInjectKeepsUnpublishedWalTail"), 1); + SyncPoint::GetInstance()->EnableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + SyncPoint::GetInstance()->SetCallBack( + "CrashSafeRecover::SeekToFileOffset:InjectStatus", [](void* arg) { + *static_cast(arg) = IOStatus::IOError("inject seek"); + }); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("a"), "1"); + ASSERT_EQ(Get("b"), "2"); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); +} +#endif // !OS_WIN + +TEST_F(DBCsppCrashSafeTest, TwoWriteQueuesWalTail) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.two_write_queues = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("q", "1")); + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("q"), "1"); +} + +TEST_F(DBCsppCrashSafeTest, WriteCommittedSkipPrepare) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + TransactionDBOptions txn_opts; + txn_opts.write_policy = TxnDBWritePolicy::WRITE_COMMITTED; + TransactionDB* txn_db = nullptr; + ASSERT_OK(TransactionDB::Open(options, txn_opts, dbname_, &txn_db)); + Transaction* txn = txn_db->BeginTransaction(WriteOptions()); + ASSERT_OK(txn->Put("wc", "1")); + ASSERT_OK(txn->Commit()); + delete txn; + delete txn_db; + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("wc"), "1"); +} + +TEST_F(DBCsppCrashSafeTest, OslFileMmapRecoverRoundTrip) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + SetupOsl(&options, true); + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("osl", "v")); + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("osl"), "v"); +} + +TEST_F(DBCsppCrashSafeTest, LogIndexHeaderHasWalRefWithoutRecover) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, false, true); + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", std::string(64, 'v'))); + ASSERT_TRUE(ListLeftovers(options, dbname_).empty()); + const uint64_t number = dbfull()->GetVersionSet()->GetColumnFamilySet() + ->GetDefault()->mem()->GetFileNumber(); + const int fd = ::open(MakeTableFileName(dbname_, number).c_str(), O_RDONLY); + ASSERT_GE(fd, 0); + terark::DFA_MmapHeader hdr{}; + ASSERT_EQ(::pread(fd, &hdr, sizeof(hdr), 0), + static_cast(sizeof(hdr))); + ::close(fd); + const auto* cs = + reinterpret_cast(hdr.reserved); + ASSERT_EQ(cs[0], 0x50505343U); + ASSERT_NE(cs[1], 0U); + ASSERT_GT(cs[2], 0U); +} + +#if !defined(OS_WIN) +TEST_F(CrashChild, DISABLED_DisableWalIsForcedOff) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + std::atomic hits{0}; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::WriteImpl:AfterWriteToWALBeforePublish", [&hits](void*) { + if (hits.fetch_add(1) >= 1) { + ::_exit(1); + } + }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "pad", "p")); + WriteOptions child_wo; + child_wo.disableWAL = true; + ASSERT_OK(child_db->Put(child_wo, "d", "4")); + ::_exit(0); +} +#endif + +TEST_F(DBCsppCrashSafeTest, DisableWalIsForcedOff) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("a", "1")); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + const uint64_t wal_number = rec.wal_number; + const uint64_t wal_offset = rec.wal_offset; + const uint64_t pubseq = rec.pubseq; + ASSERT_GT(wal_number, 0U); + ASSERT_GT(wal_offset, 0U); + WriteOptions wo; + wo.disableWAL = true; + ASSERT_OK(Put("b", "2", wo)); + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.wal_number, wal_number); + ASSERT_GT(rec.wal_offset, wal_offset); + ASSERT_GT(rec.pubseq, pubseq); + ASSERT_EQ(Get("a"), "1"); + ASSERT_EQ(Get("b"), "2"); +#if !defined(OS_WIN) + Close(); + Destroy(options); + { + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("c", "3")); + Close(); + } + ASSERT_EQ(RunCrashChild(dbname_, "DisableWalIsForcedOff"), 1); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("c"), "3"); + ASSERT_EQ(Get("pad"), "p"); + ASSERT_EQ(Get("d"), "4"); +#endif +} + +#if !defined(OS_WIN) +TEST_F(CrashChild, DISABLED_UnorderedWritePublishesPreviousWalCursor) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.unordered_write = true; + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "pad", "p")); + ASSERT_OK(child_db->Put(WriteOptions(), "d", "4")); + ::_exit(1); +} +#endif + +TEST_F(DBCsppCrashSafeTest, UnorderedWritePublishesPreviousWalCursor) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.unordered_write = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("a", "1")); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.pubseq, 0U); + ASSERT_OK(Put("b", "2")); + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_GT(rec.pubseq, 0U); + ASSERT_GT(rec.wal_offset, 0U); + const uint64_t after_b_seq = rec.pubseq; + const uint64_t after_b_off = rec.wal_offset; + ASSERT_OK(Put("c", "3")); + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_GT(rec.pubseq, after_b_seq); + ASSERT_GT(rec.wal_offset, after_b_off); + ASSERT_EQ(Get("a"), "1"); + ASSERT_EQ(Get("b"), "2"); + ASSERT_EQ(Get("c"), "3"); +#if !defined(OS_WIN) + Close(); + Destroy(options); + { + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "0")); + Close(); + } + ASSERT_EQ(RunCrashChild(dbname_, "UnorderedWritePublishesPreviousWalCursor"), 1); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("k"), "0"); + ASSERT_EQ(Get("pad"), "p"); + ASSERT_EQ(Get("d"), "4"); +#endif +} + +TEST_F(DBCsppCrashSafeTest, UnorderedWriteSyncFailureDoesNotBlockFlushOrClose) { + Close(); + for (bool recover : {false, true}) { + for (bool osl : {false, true}) { + SCOPED_TRACE(recover); + SCOPED_TRACE(osl); + Options options = BaseCrashSafeOptions(dbname_, recover, false); + if (osl) SetupOsl(&options, true); + options.unordered_write = true; + Destroy(options); + auto fs = std::make_shared(FileSystem::Default()); + auto fault_env = NewCompositeEnv(fs); + options.env = fault_env.get(); + DB* raw_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &raw_db)); + std::unique_ptr db(raw_db); + ASSERT_OK(db->Put(WriteOptions(), "seed", "value")); + SyncPoint::GetInstance()->SetCallBack("DBImpl::SyncWAL:Begin", [&](void*) { + fs->SetFilesystemActive(false, IOStatus::IOError("WAL sync failure")); + }); + SyncPoint::GetInstance()->EnableProcessing(); + WriteOptions wo; + wo.sync = true; + Status s = db->Put(wo, "failed", "value"); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + fs->SetFilesystemActive(true); + ASSERT_TRUE(s.IsIOError()); + // A WAL sync error stops the DB, but Flush and Close must still return. + ASSERT_TRUE(db->Flush(FlushOptions()).IsIOError()); + ASSERT_TRUE(db->Close().IsIOError()); + } + } +} + +TEST_F(DBCsppCrashSafeTest, UnorderedWritePreReleaseFailureAccountsWholeGroup) { + class FailPreRelease : public PreReleaseCallback { + public: + size_t calls = 0; + Status Callback(SequenceNumber, bool, uint64_t, size_t, size_t total) override { + EXPECT_EQ(total, 2U); + ++calls; + return Status::Busy("pre-release failure"); + } + }; + for (bool recover : {false, true}) { + for (bool osl : {false, true}) { + SCOPED_TRACE(recover); + SCOPED_TRACE(osl); + Close(); + Options options = BaseCrashSafeOptions(dbname_, recover, false); + if (osl) SetupOsl(&options, true); + options.unordered_write = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("seed", "value")); + // Hold the leader until the other writer has joined the same group. + FailPreRelease callback; + SyncPoint::GetInstance()->LoadDependency({ + {"UnorderedWriteFailure:FollowerJoined", "UnorderedWriteFailure:Leader"}}); + SyncPoint::GetInstance()->SetCallBack( + "WriteThread::JoinBatchGroup:Wait", [&callback](void* arg) { + auto* w = static_cast(arg); + w->pre_release_callback = &callback; + if (w->state == WriteThread::STATE_GROUP_LEADER) { + TEST_SYNC_POINT("UnorderedWriteFailure:Leader"); + } else { + TEST_SYNC_POINT("UnorderedWriteFailure:FollowerJoined"); + } + }); + SyncPoint::GetInstance()->EnableProcessing(); + Status status[2]; + std::thread writers[2]; + for (size_t i = 0; i != 2; ++i) { + writers[i] = std::thread([&, i] { + WriteBatch batch; + ASSERT_OK(batch.Put("failed" + std::to_string(i), "value")); + status[i] = db_->Write(WriteOptions(), &batch); + }); + } + for (auto& writer : writers) writer.join(); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + SyncPoint::GetInstance()->LoadDependency({}); + ASSERT_EQ(callback.calls, 1U); + for (const auto& s : status) ASSERT_TRUE(s.IsBusy()); + ASSERT_EQ(Get("failed0"), "NOT_FOUND"); + ASSERT_EQ(Get("failed1"), "NOT_FOUND"); + ASSERT_OK(Put("after", "value")); + ASSERT_OK(Put("tail", "value")); + if (recover) { + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.pubseq, db_->GetLatestSequenceNumber() - 1); + } + ASSERT_OK(Flush()); + ASSERT_OK(db_->Close()); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("seed"), "value"); + ASSERT_EQ(Get("after"), "value"); + ASSERT_EQ(Get("tail"), "value"); + } + } +} + +#if !defined(OS_WIN) +TEST_F(CrashChild, DISABLED_PipelinedWriteUsesStagedWalCursor) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.enable_pipelined_write = true; + options.merge_operator = MergeOperators::CreateStringAppendOperator(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "pad", "p")); + ASSERT_OK(child_db->Merge(WriteOptions(), "m2", "y")); + ASSERT_OK(child_db->Put(WriteOptions(), "d", "4")); + ::_exit(1); +} +#endif + +TEST_F(DBCsppCrashSafeTest, PipelinedWriteUsesStagedWalCursor) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.enable_pipelined_write = true; + options.merge_operator = MergeOperators::CreateStringAppendOperator(); + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("a", "1")); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.pubseq, 0U); + { + std::thread t1([&] { ASSERT_OK(Merge("m", "x")); }); + std::thread t2([&] { ASSERT_OK(Put("b", "2")); }); + t1.join(); + t2.join(); + } + ASSERT_OK(Put("c", "3")); + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_GT(rec.pubseq, 0U); + ASSERT_GT(rec.wal_number, 0U); + ASSERT_GT(rec.wal_offset, 0U); + ASSERT_EQ(Get("a"), "1"); + ASSERT_EQ(Get("b"), "2"); + ASSERT_EQ(Get("c"), "3"); + ASSERT_EQ(Get("m"), "x"); +#if !defined(OS_WIN) + Close(); + Destroy(options); + { + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "0")); + Close(); + } + ASSERT_EQ(RunCrashChild(dbname_, "PipelinedWriteUsesStagedWalCursor"), 1); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("k"), "0"); + ASSERT_EQ(Get("pad"), "p"); + ASSERT_EQ(Get("m2"), "y"); + ASSERT_EQ(Get("d"), "4"); +#endif +} + +#if !defined(OS_WIN) +TEST_F(CrashChild, DISABLED_LeftoverOnDbPathNotCfPaths0) { + const std::string l0 = dbname_ + "/l0data"; + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.db_paths = {{l0, 1ULL << 30}}; + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "pad", "p")); + ASSERT_OK(child_db->Put(WriteOptions(), "x", "1")); + auto* cfd = static_cast(child_db)->GetVersionSet() + ->GetColumnFamilySet()->GetDefault(); + ASSERT_OK(WriteStringToFile( + options.env, TableFileName(cfd->ioptions()->cf_paths, + cfd->mem()->GetFileNumber(), 0), + dbname_ + "/active-file")); + ::_exit(1); +} +#endif + +TEST_F(DBCsppCrashSafeTest, LeftoverOnDbPathNotCfPaths0) { + Close(); + const std::string l0 = dbname_ + "/l0data"; + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.db_paths = {{l0, 1ULL << 30}}; + Destroy(options); + ASSERT_OK(env_->CreateDirIfMissing(dbname_)); + ASSERT_OK(env_->CreateDirIfMissing(l0)); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "0")); + Close(); + ASSERT_EQ(ListLeftovers(options, dbname_).size(), 2U); + for (const auto& path : ListLeftovers(options, dbname_)) + ASSERT_OK(env_->FileExists(path)); +#if !defined(OS_WIN) + ASSERT_EQ(RunCrashChild(dbname_, "LeftoverOnDbPathNotCfPaths0"), 1); + auto leftovers_l0 = ListLeftovers(options, dbname_); + ASSERT_EQ(leftovers_l0.size(), 2U); + std::string active; + ASSERT_OK(ReadFileToString(env_, dbname_ + "/active-file", &active)); + ASSERT_EQ(std::count(leftovers_l0.begin(), leftovers_l0.end(), active), 1); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("k"), "0"); + ASSERT_EQ(Get("pad"), "p"); + ASSERT_EQ(Get("x"), "1"); + ASSERT_OK(env_->FileExists(active)); + std::vector files; + db_->GetLiveFilesMetaData(&files); + const auto matching = std::count_if(files.begin(), files.end(), [&](const auto& file) { + return MakeTableFileName(file.db_path, file.file_number) == active; + }); + ASSERT_EQ(matching, 1); +#endif +} + +TEST_F(DBCsppCrashSafeTest, EmptyWriteUpdatesWalCursor) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "v")); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + const uint64_t pubseq = rec.pubseq; + const uint64_t wal_number = rec.wal_number; + const uint64_t wal_offset = rec.wal_offset; + ASSERT_GT(pubseq, 0U); + WriteBatch empty; + ASSERT_OK(db_->Write(WriteOptions(), &empty)); + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.pubseq, pubseq); + ASSERT_EQ(rec.wal_number, wal_number); + ASSERT_GT(rec.wal_offset, wal_offset); +} + +TEST_F(DBCsppCrashSafeTest, PublishedSeqFieldsAfterPut) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + ASSERT_OK(TryReopen(options)); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.wal_offset_kind, 1U); // kPhysical, stamped on create + ASSERT_EQ(rec.pubseq, 0U); + ASSERT_OK(Put("k", "v")); + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.generation % 2, 0U); + ASSERT_EQ(rec.pubseq, dbfull()->GetLatestSequenceNumber()); + ASSERT_GT(rec.wal_number, 0U); + ASSERT_GT(rec.wal_offset, 0U); + ASSERT_EQ(rec.wal_offset_kind, 1U); // Persist does not rewrite kind +} + +TEST_F(DBCsppCrashSafeTest, ExistingPublishedSeqKindPrepLogIndexAfterClassic) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.avoid_flush_during_shutdown = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "v")); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.wal_offset_kind, 1U); + Close(); + ASSERT_OK(env_->FileExists(CrashSafePubSeqFileName(dbname_))); + Options log_index = BaseCrashSafeOptions(dbname_, true, true); + log_index.avoid_flush_during_shutdown = true; + ASSERT_OK(TryReopen(log_index)); + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.wal_offset_kind, 2U); + ASSERT_EQ(Get("k"), "v"); + ASSERT_GE(CountL0(db_), 1); +} + +TEST_F(DBCsppCrashSafeTest, ExistingPublishedSeqKindPrepClassicAfterLogIndex) { + Close(); + Options log_index = BaseCrashSafeOptions(dbname_, true, true); + log_index.avoid_flush_during_shutdown = true; + Destroy(log_index); + ASSERT_OK(TryReopen(log_index)); + ASSERT_OK(Put("k", "v")); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.wal_offset_kind, 2U); + Close(); + ASSERT_OK(env_->FileExists(CrashSafePubSeqFileName(dbname_))); + Options classic = BaseCrashSafeOptions(dbname_, true, false); + classic.avoid_flush_during_shutdown = true; + ASSERT_OK(TryReopen(classic)); + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.wal_offset_kind, 1U); + ASSERT_EQ(Get("k"), "v"); + ASSERT_GE(CountL0(db_), 1); +} + +TEST_F(DBCsppCrashSafeTest, KindPrepSkippedWhenKindMatches) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.avoid_flush_during_shutdown = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "v")); + Close(); + int prep = 0; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::Open::KindPrep:AfterOpenBeforeFlush", + [&](void*) { prep++; }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(options)); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_EQ(prep, 0); + ASSERT_EQ(Get("k"), "v"); +} + +TEST_F(DBCsppCrashSafeTest, KindPrepFailureClearsDbPointer) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.avoid_flush_during_shutdown = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "v")); + Close(); + options.memtable_as_log_index = true; + options.error_if_exists = true; + DB* db = reinterpret_cast(uintptr_t(1)); + Status s = DB::Open(options, dbname_, &db); + ASSERT_TRUE(s.IsInvalidArgument()) << s.ToString(); + ASSERT_EQ(db, nullptr); +} + +#if !defined(OS_WIN) +TEST_F(CrashChild, DISABLED_EasyMigrateKindPrep) { + ASSERT_EQ(arg_.size(), 3U); + const bool log_index = arg_[0] == '1'; + const bool txn = arg_[1] == '1'; + const bool recover = arg_[2] == '1'; + // EasyMigrate caches its config, so load it only in this fresh exec child. + const std::string config = dbname_ + ".easy-migrate.json"; + ASSERT_EQ(::setenv("TOPLINGDB_EASY_MIGRATE_CONF", config.c_str(), 1), 0); + Options options = BaseCrashSafeOptions(dbname_, recover, log_index); + int prep = 0; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::Open::KindPrep:AfterOpenBeforeFlush", [&](void*) { ++prep; }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* db = nullptr; + if (txn) { + TransactionDB* txn_db = nullptr; + ASSERT_OK(TransactionDB::Open(options, TransactionDBOptions(), dbname_, + &txn_db)); + db = txn_db; + } else { + ASSERT_OK(DB::Open(options, dbname_, &db)); + } + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_EQ(prep, 1); + std::string value; + ASSERT_OK(db->Get(ReadOptions(), "k", &value)); + ASSERT_EQ(value, "v"); + ASSERT_OK(db->Put(WriteOptions(), "after", "switch")); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.wal_offset_kind, log_index ? 2U : 1U); + ASSERT_OK(db->Close()); + delete db; + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, EasyMigrateKindPrep) { + for (bool log_index : {false, true}) { + for (bool txn : {false, true}) { + for (bool recover : {false, true}) { + const std::string mode = {char('0' + log_index), char('0' + txn), + char('0' + recover)}; + SCOPED_TRACE(mode); + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, !log_index); + options.avoid_flush_during_shutdown = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "v")); + // This convert-only table factory cannot BuildTable during 2PC replay. + ASSERT_OK(Flush()); + Close(); + const std::string config = dbname_ + ".easy-migrate.json"; + const json conf = { + {"DBOptions", {{"default", {{"memtable_crash_safe_recover", true}, + {"memtable_as_log_index", log_index}}}}}, + {"http", {{"auto_start_http", false}}}}; + ASSERT_OK(WriteStringToFile(env_, conf.dump(), config)); + ASSERT_EQ(RunCrashChild(dbname_, "EasyMigrateKindPrep", mode), 0); + ASSERT_OK(env_->DeleteFile(config)); + options.memtable_as_log_index = log_index; + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("k"), "v"); + ASSERT_EQ(Get("after"), "switch"); + } + } + } +} + +TEST_F(CrashChild, DISABLED_KindPrepKeepsUnpublishedWalTail) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.avoid_flush_during_shutdown = true; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::WriteImpl:AfterWriteToWALBeforePublish", + [](void*) { ::_exit(1); }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "b", "2")); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, KindPrepKeepsUnpublishedWalTail) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.avoid_flush_during_shutdown = true; + Destroy(options); + { + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("a", "1")); + Close(); + } + ASSERT_EQ(RunCrashChild(dbname_, "KindPrepKeepsUnpublishedWalTail"), 1); + Options log_index = BaseCrashSafeOptions(dbname_, true, true); + log_index.avoid_flush_during_shutdown = true; + ASSERT_OK(TryReopen(log_index)); + ASSERT_EQ(Get("a"), "1"); + ASSERT_EQ(Get("b"), "2"); +} + +TEST_F(DBCsppCrashSafeTest, KindPrepAfterFlushBeforeClose) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.avoid_flush_during_shutdown = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "v")); + Close(); + ASSERT_EQ(RunCrashChild( + dbname_, "KindPrep", "DBImpl::Open::KindPrep:AfterFlushBeforeClose"), + kKindPrepChildCrashed); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.wal_offset_kind, 1U); + Options log_index = BaseCrashSafeOptions(dbname_, true, true); + log_index.avoid_flush_during_shutdown = true; + ASSERT_OK(TryReopen(log_index)); + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.wal_offset_kind, 2U); + ASSERT_EQ(Get("k"), "v"); +} + +TEST_F(DBCsppCrashSafeTest, KindPrepAfterCloseBeforeDeleteSidecar) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.avoid_flush_during_shutdown = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "v")); + Close(); + ASSERT_EQ(RunCrashChild( + dbname_, "KindPrep", "DBImpl::Open::KindPrep:AfterCloseBeforeDeleteSidecar"), + kKindPrepChildCrashed); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.wal_offset_kind, 1U); + Options log_index = BaseCrashSafeOptions(dbname_, true, true); + log_index.avoid_flush_during_shutdown = true; + ASSERT_OK(TryReopen(log_index)); + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.wal_offset_kind, 2U); + ASSERT_EQ(Get("k"), "v"); +} + +TEST_F(CrashChild, DISABLED_TransactionDBKindPrepClassicToLogIndex) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.avoid_flush_during_shutdown = true; + TransactionDBOptions txn_opts; + txn_opts.write_policy = TxnDBWritePolicy::WRITE_COMMITTED; + TransactionDB* txn_db = nullptr; + ASSERT_OK(TransactionDB::Open(options, txn_opts, dbname_, &txn_db)); + Transaction* txn = txn_db->BeginTransaction(WriteOptions()); + ASSERT_OK(txn->Put("t", "1")); + ASSERT_OK(txn->Commit()); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, TransactionDBKindPrepClassicToLogIndex) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.avoid_flush_during_shutdown = true; + Destroy(options); + TransactionDBOptions txn_opts; + txn_opts.write_policy = TxnDBWritePolicy::WRITE_COMMITTED; + TransactionDB* txn_db = nullptr; + ASSERT_EQ(RunCrashChild(dbname_, "TransactionDBKindPrepClassicToLogIndex"), 0); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.wal_offset_kind, 1U); + Options log_index = BaseCrashSafeOptions(dbname_, true, true); + log_index.avoid_flush_during_shutdown = true; + int prep = 0; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::Open::KindPrep:AfterOpenBeforeFlush", + [&prep](void*) { prep++; }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TransactionDB::Open(log_index, txn_opts, dbname_, &txn_db)); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_EQ(prep, 1); + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.wal_offset_kind, 2U); + std::string v; + ASSERT_OK(txn_db->Get(ReadOptions(), "t", &v)); + ASSERT_EQ(v, "1"); + delete txn_db; +} + +TEST_F(CrashChild, DISABLED_TransactionDBKindPrepPreparedNotSupported) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.avoid_flush_during_shutdown = true; + TransactionDBOptions txn_opts; + txn_opts.write_policy = TxnDBWritePolicy::WRITE_COMMITTED; + TransactionDB* txn_db = nullptr; + ASSERT_OK(TransactionDB::Open(options, txn_opts, dbname_, &txn_db)); + TransactionOptions to; + to.skip_prepare = false; + Transaction* txn = txn_db->BeginTransaction(WriteOptions(), to); + ASSERT_OK(txn->SetName("xid1")); + ASSERT_OK(txn->Put("prep", "v")); + ASSERT_OK(txn->Prepare()); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, TransactionDBKindPrepPreparedNotSupported) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.avoid_flush_during_shutdown = true; + Destroy(options); + TransactionDBOptions txn_opts; + txn_opts.write_policy = TxnDBWritePolicy::WRITE_COMMITTED; + TransactionDB* txn_db = nullptr; + ASSERT_EQ(RunCrashChild(dbname_, "TransactionDBKindPrepPreparedNotSupported"), 0); + Options log_index = BaseCrashSafeOptions(dbname_, true, true); + log_index.avoid_flush_during_shutdown = true; + ASSERT_TRUE(TransactionDB::Open(log_index, txn_opts, dbname_, &txn_db) + .IsNotSupported()); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.wal_offset_kind, 1U); + ASSERT_OK(TransactionDB::Open(options, txn_opts, dbname_, &txn_db)); + std::vector prepared; + txn_db->GetAllPreparedTransactions(&prepared); + ASSERT_EQ(prepared.size(), 1U); + ASSERT_OK(prepared[0]->Rollback()); + delete prepared[0]; + delete txn_db; +} + +TEST_F(DBCsppCrashSafeTest, TransactionDBPreparedCloseThenSwitchKind) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + TransactionDBOptions txn_opts; + txn_opts.write_policy = TxnDBWritePolicy::WRITE_COMMITTED; + TransactionDB* txn_db = nullptr; + ASSERT_OK(TransactionDB::Open(options, txn_opts, dbname_, &txn_db)); + TransactionOptions to; + to.skip_prepare = false; + Transaction* txn = txn_db->BeginTransaction(WriteOptions(), to); + ASSERT_OK(txn->SetName("xid1")); + ASSERT_OK(txn->Put("prep", "v")); + ASSERT_OK(txn->Prepare()); + delete txn; + delete txn_db; // default avoid_flush_during_shutdown=false: Close converts + Options log_index = BaseCrashSafeOptions(dbname_, true, true); + ASSERT_TRUE(TransactionDB::Open(log_index, txn_opts, dbname_, &txn_db) + .IsNotSupported()); + ASSERT_OK(TransactionDB::Open(options, txn_opts, dbname_, &txn_db)); + std::vector prepared; + txn_db->GetAllPreparedTransactions(&prepared); + ASSERT_EQ(prepared.size(), 1U); + ASSERT_OK(prepared[0]->Rollback()); + delete prepared[0]; + delete txn_db; +} + +TEST_F(CrashChild, DISABLED_TransactionDBRollbackCloseThenSwitchKind) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + TransactionDBOptions txn_opts; + txn_opts.write_policy = TxnDBWritePolicy::WRITE_COMMITTED; + TransactionDB* txn_db = nullptr; + ASSERT_OK(TransactionDB::Open(options, txn_opts, dbname_, &txn_db)); + TransactionOptions to; + to.skip_prepare = false; + Transaction* txn = txn_db->BeginTransaction(WriteOptions(), to); + ASSERT_OK(txn->SetName("xid1")); + ASSERT_OK(txn->Put("prep", "v")); + ASSERT_OK(txn->Prepare()); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, TransactionDBRollbackCloseThenSwitchKind) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + TransactionDBOptions txn_opts; + txn_opts.write_policy = TxnDBWritePolicy::WRITE_COMMITTED; + TransactionDB* txn_db = nullptr; + ASSERT_OK(TransactionDB::Open(options, txn_opts, dbname_, &txn_db)); + ASSERT_OK(txn_db->Put(WriteOptions(), "k", "v")); + delete txn_db; + ASSERT_EQ(RunCrashChild(dbname_, "TransactionDBRollbackCloseThenSwitchKind"), 0); + ASSERT_OK(TransactionDB::Open(options, txn_opts, dbname_, &txn_db)); + std::vector prepared; + txn_db->GetAllPreparedTransactions(&prepared); + ASSERT_EQ(prepared.size(), 1U); + ASSERT_OK(prepared[0]->Rollback()); + delete prepared[0]; + delete txn_db; // default avoid_flush_during_shutdown=false: Close converts + Options log_index = BaseCrashSafeOptions(dbname_, true, true); + ASSERT_OK(TransactionDB::Open(log_index, txn_opts, dbname_, &txn_db)); + std::string v; + ASSERT_OK(txn_db->Get(ReadOptions(), "k", &v)); + ASSERT_EQ(v, "v"); + ASSERT_TRUE(txn_db->Get(ReadOptions(), "prep", &v).IsNotFound()); + delete txn_db; +} + +TEST_F(DBCsppCrashSafeTest, KindPrepConvertFailFallsBackToWal) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.avoid_flush_during_shutdown = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "v")); + Close(); + Options log_index = BaseCrashSafeOptions(dbname_, true, true); + log_index.avoid_flush_during_shutdown = true; + ASSERT_FALSE(log_index.check_wal_format); + ASSERT_OK(TryReopen(log_index)); + ASSERT_EQ(Get("k"), "v"); + Close(); + SyncPoint::GetInstance()->SetCallBack( + "CrashSafeRecover::ConvertToSST:InjectStatus", [](void* arg) { + *static_cast(arg) = Status::IOError("inject convert"); + }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(log_index)); + ASSERT_EQ(Get("k"), "v"); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); +} +#endif + +TEST_F(DBCsppCrashSafeTest, DontConvertUsesWal) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, false, false); + SetupCspp(&options, false); + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "v")); + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("k"), "v"); +} + +TEST_F(DBCsppCrashSafeTest, BestEffortsRecoveryFallsBackToWal) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.best_efforts_recovery = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "v")); + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("k"), "v"); +} + +TEST_F(DBCsppCrashSafeTest, LogIndexOnRecoverOffUsesWal) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, false, true); + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "v")); + Close(); + ASSERT_TRUE(env_->FileExists(CrashSafePubSeqFileName(dbname_)).IsNotFound()); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("k"), "v"); +} + +TEST_F(DBCsppCrashSafeTest, ManifestRegistryIgnoresUnregisteredFiles) { + for (bool osl : {false, true}) { + SCOPED_TRACE(osl ? "OSL" : "CSPP"); + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + if (osl) { + SetupOsl(&options, true); + } + Destroy(options); + ASSERT_OK(env_->CreateDirIfMissing(dbname_)); + const std::string high = MakeTableFileName(dbname_, 100); + const std::string low = MakeTableFileName(dbname_, 10); + ASSERT_OK(WriteStringToFile(env_, "", high)); + ASSERT_OK(WriteStringToFile(env_, "", low)); + ASSERT_TRUE(ListLeftovers(options, dbname_).empty()); + // Remove both so collision checks cannot hide a stale counter. + ASSERT_OK(env_->DeleteFile(high)); + ASSERT_OK(env_->DeleteFile(low)); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(ListLeftovers(options, dbname_).size(), 2U); + for (const auto& path : ListLeftovers(options, dbname_)) + ASSERT_OK(env_->FileExists(path)); + Close(); + const auto before = ListLeftovers(options, dbname_); + // The manifest remains authoritative when unrelated files appear. + ASSERT_OK(WriteStringToFile(env_, "", low)); + ASSERT_EQ(ListLeftovers(options, dbname_), before); + ASSERT_OK(env_->DeleteFile(low)); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(ListLeftovers(options, dbname_).size(), 2U); + for (const auto& path : ListLeftovers(options, dbname_)) + ASSERT_OK(env_->FileExists(path)); + Close(); + Destroy(options); + } +} + +TEST_F(DBCsppCrashSafeTest, CreateMemTableRepDoesNotOverwriteLeftover) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("keep", "me")); + auto first = ListLeftovers(options, dbname_); + ASSERT_EQ(first.size(), 2U); + auto* cfd = dbfull()->GetVersionSet()->GetColumnFamilySet()->GetDefault(); + const auto active = MakeTableFileName(dbname_, cfd->mem()->GetFileNumber()); + ASSERT_EQ(std::count(first.begin(), first.end(), active), 1); + std::string before; + ASSERT_OK(ReadFileToString(env_, active, &before)); + ASSERT_OK(dbfull()->TEST_SwitchMemtable()); + ASSERT_OK(Put("other", "x")); + auto after_list = ListLeftovers(options, dbname_); + ASSERT_EQ(after_list, first); + ASSERT_NE(MakeTableFileName(dbname_, cfd->mem()->GetFileNumber()), active); + std::string after; + ASSERT_OK(ReadFileToString(env_, active, &after)); + ASSERT_EQ(before, after); +} + +TEST_F(DBCsppCrashSafeTest, Allow2pcAloneStillConverts) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.allow_2pc = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("k", "v")); + ASSERT_OK(dbfull()->TEST_SwitchMemtable()); + Close(); + ASSERT_EQ(ListLeftovers(options, dbname_).size(), 2U); + for (const auto& path : ListLeftovers(options, dbname_)) + ASSERT_OK(env_->FileExists(path)); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("k"), "v"); +} + +#if !defined(OS_WIN) +TEST_F(CrashChild, DISABLED_WalFilterFallsBackToWal) { + Options options = BaseCrashSafeOptions(dbname_, true, arg_[1] == '1'); + if (arg_[0] == '1') SetupOsl(&options, true); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "a", std::string(128, 'a'))); + ASSERT_OK(child_db->Put(WriteOptions(), "b", std::string(128, 'b'))); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, WalFilterFallsBackToWal) { + class IgnoreRecords final : public WalFilter { + public: + int calls = 0; + const char* Name() const override { return "IgnoreRecords"; } + WalProcessingOption LogRecordFound(unsigned long long, const std::string&, + const WriteBatch&, WriteBatch*, + bool*) override { + ++calls; + return WalProcessingOption::kIgnoreCurrentRecord; + } + }; + for (bool osl : {false, true}) { + for (bool log_index : {false, true}) { + SCOPED_TRACE(osl); + SCOPED_TRACE(log_index); + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, log_index); + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "WalFilterFallsBackToWal", + std::string(osl ? "1" : "0") + + (log_index ? "1" : "0")), 0); + ASSERT_FALSE(ListLeftovers(options, dbname_).empty()); + IgnoreRecords filter; + options.wal_filter = &filter; + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(filter.calls, 2); + ASSERT_EQ(Get("a"), "NOT_FOUND"); + ASSERT_EQ(Get("b"), "NOT_FOUND"); + ASSERT_EQ(CountL0(db_), 0); + Close(); + } + } +} + +TEST_F(CrashChild, DISABLED_LeftoverNoMagicFallsBackToWal) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", + [](void*) { ::_exit(1); }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "bad", "hdr")); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, LeftoverNoMagicFallsBackToWal) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + for (const std::string damage : {"empty", "short-header", "missing-magic"}) { + SCOPED_TRACE(damage); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "LeftoverNoMagicFallsBackToWal"), 1); + auto leftovers = ListLeftovers(options, dbname_); + ASSERT_FALSE(leftovers.empty()); + const int fd = ::open(leftovers[0].c_str(), O_RDWR); + ASSERT_GE(fd, 0); + terark::DFA_MmapHeader hdr{}; + if (damage == "missing-magic") { + ASSERT_EQ(::pread(fd, &hdr, sizeof(hdr), 0), + static_cast(sizeof(hdr))); + std::memset(hdr.reserved, 0, sizeof(hdr.reserved)); + ASSERT_EQ(::pwrite(fd, &hdr, sizeof(hdr), 0), + static_cast(sizeof(hdr))); + } else { + ASSERT_EQ(::ftruncate(fd, damage == "empty" ? 0 : sizeof(hdr) - 1), 0); + } + ::close(fd); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("bad"), "hdr"); + Close(); + } +} + +TEST_F(CrashChild, DISABLED_DualLeftoverSecondConvertFails) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + std::atomic pubs{0}; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", [&pubs](void*) { + if (pubs.fetch_add(1) >= 1) { + ::_exit(1); + } + }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "a", "1")); + ASSERT_OK(static_cast(child_db)->TEST_SwitchMemtable()); + ASSERT_OK(child_db->Put(WriteOptions(), "b", "2")); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, DualLeftoverSecondConvertFails) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "DualLeftoverSecondConvertFails"), 1); + SyncPoint::GetInstance()->EnableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + std::atomic converts{0}; + SyncPoint::GetInstance()->SetCallBack( + "CrashSafeRecover::ConvertToSST:InjectStatus", [&converts](void* arg) { + if (converts.fetch_add(1) >= 1) { + *static_cast(arg) = Status::IOError("second leftover"); + } + }); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("a"), "1"); + ASSERT_EQ(Get("b"), "2"); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); +} + +TEST_F(CrashChild, DISABLED_TruncateInjectFailureFallsBackToWal) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", + [](void*) { ::_exit(1); }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "tr", "ok")); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, TruncateInjectFailureFallsBackToWal) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "TruncateInjectFailureFallsBackToWal"), 1); + SyncPoint::GetInstance()->EnableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + bool truncate_called = false; + SyncPoint::GetInstance()->SetCallBack( + "MemTableRep::ConvertToSST:Truncate", [&truncate_called](void* arg) { + truncate_called = true; + *static_cast(arg) = IOStatus::IOError("inject truncate"); + }); + ASSERT_OK(TryReopen(options)); + ASSERT_TRUE(truncate_called); + ASSERT_EQ(Get("tr"), "ok"); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); +} + +TEST_F(CrashChild, DISABLED_AfterConvertBeforeAddFileKeepsRegisteredFile) { + ASSERT_EQ(arg_.size(), 3U); + Options options = BaseCrashSafeOptions(dbname_, true, arg_[1] == '1'); + if (arg_[0] == '1') SetupOsl(&options, true); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", + [](void*) { ::_exit(1); }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + auto* cfd = static_cast(child_db)->GetVersionSet() + ->GetColumnFamilySet()->GetDefault(); + ASSERT_OK(WriteStringToFile( + options.env, MakeTableFileName(dbname_, cfd->mem()->GetFileNumber()), + dbname_ + "/active-file")); + ASSERT_OK(child_db->Put(WriteOptions(), "or", std::string(128, 'v'))); + ::_exit(0); +} + +TEST_F(CrashChild, DISABLED_AfterConvertBeforeAddFileKeepsRegisteredFileRecover) { + ASSERT_EQ(arg_.size(), 3U); + Options options = BaseCrashSafeOptions(dbname_, true, arg_[1] == '1'); + if (arg_[0] == '1') SetupOsl(&options, true); + const char* points[] = { + "MemTableRep::ConvertToSST:Truncate", + "CrashSafeRecover::AfterConvertBeforeAddFile", + "CrashSafeRecover::AfterConvertBeforeAddFile", + "DBImpl::RegisterMemTableFile:AfterLogAndApply"}; + ASSERT_LT(arg_[2] - '0', 4); + SyncPoint::GetInstance()->SetCallBack( + points[arg_[2] - '0'], + [&](void*) { + if (arg_[2] == '1') { + // Model an incomplete SST tail, without claiming this callback runs + // in the middle of a write. The persisted trie remains intact. + const auto files = ListLeftovers(options, dbname_); + ASSERT_EQ(files.size(), 2U); + std::string active; + ASSERT_OK(ReadFileToString(options.env, dbname_ + "/active-file", &active)); + ASSERT_EQ(std::count(files.begin(), files.end(), active), 1); + const int fd = ::open(active.c_str(), O_RDWR); + ASSERT_GE(fd, 0); + uint64_t structure_size = 0; + if (arg_[0] == '1') { + terark::OSL_MmapHeader hdr{}; + ASSERT_EQ(::pread(fd, &hdr, sizeof(hdr), 0), + static_cast(sizeof(hdr))); + structure_size = hdr.mem_used; + } else { + terark::DFA_MmapHeader hdr{}; + ASSERT_EQ(::pread(fd, &hdr, sizeof(hdr), 0), + static_cast(sizeof(hdr))); + structure_size = hdr.file_size; + } + uint64_t size = 0; + ASSERT_OK(options.env->GetFileSize(active, &size)); + ASSERT_GT(size, structure_size); + ASSERT_EQ(::ftruncate(fd, size - 1), 0); + ::close(fd); + } + ::_exit(1); + }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* recover_db = nullptr; + DB::Open(options, dbname_, &recover_db); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, AfterConvertBeforeAddFileKeepsRegisteredFile) { + Close(); + for (bool osl : {false, true}) { + for (bool log_index : {false, true}) { + for (int window = 0; window < 4; ++window) { + const std::string config = std::to_string(osl) + + std::to_string(log_index) + std::to_string(window); + SCOPED_TRACE(config); + Options options = BaseCrashSafeOptions(dbname_, true, log_index); + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, + "AfterConvertBeforeAddFileKeepsRegisteredFile", config), 1); + const auto before = ListLeftovers(options, dbname_); + ASSERT_EQ(before.size(), 2U); + std::string active; + ASSERT_OK(ReadFileToString(env_, dbname_ + "/active-file", &active)); + ASSERT_EQ(std::count(before.begin(), before.end(), active), 1); + for (const auto& path : before) ASSERT_OK(env_->FileExists(path)); + // Before commit, interrupt the same file twice to exercise footer + // replacement, not just a one-time conversion of the original source. + for (int crash = 0; crash < (window == 3 ? 1 : 2); ++crash) { + ASSERT_EQ(RunCrashChild(dbname_, + "AfterConvertBeforeAddFileKeepsRegisteredFileRecover", config), 1); + ASSERT_OK(env_->FileExists(active)); + if (window != 3) { + ASSERT_EQ(ListLeftovers(options, dbname_), before); + for (const auto& path : before) ASSERT_OK(env_->FileExists(path)); + } + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_EQ(rec.generation & 1, 0U); + } + int converted = 0; + SyncPoint::GetInstance()->SetCallBack( + "MemTableRep::ConvertToSST:After", [&](void*) { ++converted; }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopen(options)); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_EQ(converted, window == 3 ? 0 : 1); + ASSERT_EQ(Get("or"), std::string(128, 'v')); + std::vector files; + db_->GetLiveFilesMetaData(&files); + ASSERT_EQ(files.size(), 1U); + ASSERT_EQ(files.front().level, 0); + ASSERT_EQ(MakeTableFileName(dbname_, files.front().file_number), + active); + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("or"), std::string(128, 'v')); + ASSERT_EQ(CountL0(db_), 1); + Close(); + } + } + } +} + +TEST_F(DBCsppCrashSafeTest, CrashSafeOrLogIndexDisablesWalCompression) { + for (bool crash_safe : {false, true}) { + for (bool log_index : {false, true}) { + SCOPED_TRACE(crash_safe); + SCOPED_TRACE(log_index); + Close(); + Options options = BaseCrashSafeOptions(dbname_, crash_safe, log_index); + options.wal_compression = kZSTD; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(db_->GetDBOptions().wal_compression, + crash_safe || log_index ? kNoCompression : kZSTD); + ASSERT_OK(Put("k", "v")); + Reopen(options); + ASSERT_EQ(Get("k"), "v"); + } + } +} + +TEST_F(CrashChild, DISABLED_AfterWriteToWALGhostCopyHidden) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::WriteImpl:AfterWriteToWALBeforePublish", + [](void*) { ::_exit(1); }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(child_db->Put(WriteOptions(), "tail", "2")); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, AfterWriteToWALGhostCopyHidden) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + { + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("first", "1")); + Close(); + } + ASSERT_EQ(RunCrashChild(dbname_, "AfterWriteToWALGhostCopyHidden"), 1); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("first"), "1"); + ASSERT_EQ(Get("tail"), "2"); + ASSERT_OK(Put("bump", "seq")); + ASSERT_EQ(Get("tail"), "2"); + int n = 0; + std::unique_ptr it(db_->NewIterator(ReadOptions())); + for (it->Seek("tail"); it->Valid() && it->key() == "tail"; it->Next()) { + n++; + } + ASSERT_OK(it->status()); + ASSERT_EQ(n, 1); +} + +TEST_F(CrashChild, DISABLED_WriteCommittedPrepareInDoubt) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + TransactionDBOptions txn_opts; + txn_opts.write_policy = TxnDBWritePolicy::WRITE_COMMITTED; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", + [](void*) { ::_exit(1); }); + SyncPoint::GetInstance()->EnableProcessing(); + TransactionDB* txn_db = nullptr; + ASSERT_OK(TransactionDB::Open(options, txn_opts, dbname_, &txn_db)); + TransactionOptions to; + to.skip_prepare = false; + Transaction* txn = txn_db->BeginTransaction(WriteOptions(), to); + ASSERT_OK(txn->SetName("xid1")); + ASSERT_OK(txn->Put("prep", "v")); + ASSERT_OK(txn->Prepare()); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, WriteCommittedPrepareInDoubt) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + TransactionDBOptions txn_opts; + txn_opts.write_policy = TxnDBWritePolicy::WRITE_COMMITTED; + ASSERT_EQ(RunCrashChild(dbname_, "WriteCommittedPrepareInDoubt"), 1); + TransactionDB* txn_db = nullptr; + ASSERT_OK(TransactionDB::Open(options, txn_opts, dbname_, &txn_db)); + std::vector prepared; + txn_db->GetAllPreparedTransactions(&prepared); + ASSERT_EQ(prepared.size(), 1U); + ASSERT_OK(prepared[0]->Rollback()); + delete prepared[0]; + delete txn_db; +} + +TEST_F(CrashChild, DISABLED_WriteCommittedPrepareAndCommit) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + TransactionDBOptions txn_opts; + txn_opts.write_policy = TxnDBWritePolicy::WRITE_COMMITTED; + std::atomic prepare_done{false}; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", [&prepare_done](void*) { + if (prepare_done.load(std::memory_order_acquire)) { + ::_exit(1); + } + }); + SyncPoint::GetInstance()->EnableProcessing(); + TransactionDB* txn_db = nullptr; + ASSERT_OK(TransactionDB::Open(options, txn_opts, dbname_, &txn_db)); + TransactionOptions to; + to.skip_prepare = false; + Transaction* txn = txn_db->BeginTransaction(WriteOptions(), to); + ASSERT_OK(txn->SetName("xid2")); + ASSERT_OK(txn->Put("pc", "v")); + ASSERT_OK(txn->Prepare()); + prepare_done.store(true, std::memory_order_release); + ASSERT_OK(txn->Commit()); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, WriteCommittedPrepareAndCommit) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + TransactionDBOptions txn_opts; + txn_opts.write_policy = TxnDBWritePolicy::WRITE_COMMITTED; + ASSERT_EQ(RunCrashChild(dbname_, "WriteCommittedPrepareAndCommit"), 1); + TransactionDB* txn_db = nullptr; + ASSERT_OK(TransactionDB::Open(options, txn_opts, dbname_, &txn_db)); + std::vector prepared; + txn_db->GetAllPreparedTransactions(&prepared); + ASSERT_TRUE(prepared.empty()); + std::string v; + ASSERT_OK(txn_db->Get(ReadOptions(), "pc", &v)); + ASSERT_EQ(v, "v"); + delete txn_db; +} + +TEST_F(DBCsppCrashSafeTest, WriteCommittedPrepareTwoQueuesPublishes) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.two_write_queues = true; + Destroy(options); + TransactionDBOptions txn_opts; + txn_opts.write_policy = TxnDBWritePolicy::WRITE_COMMITTED; + TransactionDB* txn_db = nullptr; + ASSERT_OK(TransactionDB::Open(options, txn_opts, dbname_, &txn_db)); + TransactionOptions to; + to.skip_prepare = false; + Transaction* txn = txn_db->BeginTransaction(WriteOptions(), to); + ASSERT_OK(txn->SetName("xidq")); + ASSERT_OK(txn->Put("pq", "1")); + ASSERT_OK(txn->Prepare()); + PublishedSeqOnDisk rec; + ASSERT_TRUE(ReadPublishedSeqFile(dbname_, &rec)); + ASSERT_GT(rec.wal_number, 0U); + ASSERT_GT(rec.wal_offset, 0U); + ASSERT_OK(txn->Rollback()); + delete txn; + delete txn_db; +} + +#if !defined(OS_WIN) +TEST_F(CrashChild, DISABLED_WriteCommittedPrepareTwoQueuesInDoubt) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.two_write_queues = true; + TransactionDBOptions txn_opts; + txn_opts.write_policy = TxnDBWritePolicy::WRITE_COMMITTED; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", + [](void*) { ::_exit(1); }); + SyncPoint::GetInstance()->EnableProcessing(); + TransactionDB* txn_db = nullptr; + ASSERT_OK(TransactionDB::Open(options, txn_opts, dbname_, &txn_db)); + TransactionOptions to; + to.skip_prepare = false; + Transaction* txn = txn_db->BeginTransaction(WriteOptions(), to); + ASSERT_OK(txn->SetName("xidq1")); + ASSERT_OK(txn->Put("prepq", "v")); + ASSERT_OK(txn->Prepare()); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, WriteCommittedPrepareTwoQueuesInDoubt) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.two_write_queues = true; + Destroy(options); + TransactionDBOptions txn_opts; + txn_opts.write_policy = TxnDBWritePolicy::WRITE_COMMITTED; + ASSERT_EQ(RunCrashChild(dbname_, "WriteCommittedPrepareTwoQueuesInDoubt"), 1); + TransactionDB* txn_db = nullptr; + ASSERT_OK(TransactionDB::Open(options, txn_opts, dbname_, &txn_db)); + std::vector prepared; + txn_db->GetAllPreparedTransactions(&prepared); + ASSERT_EQ(prepared.size(), 1U); + ASSERT_OK(prepared[0]->Rollback()); + delete prepared[0]; + delete txn_db; +} + +TEST_F(CrashChild, DISABLED_WriteCommittedPrepareTwoQueuesAndCommit) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.two_write_queues = true; + TransactionDBOptions txn_opts; + txn_opts.write_policy = TxnDBWritePolicy::WRITE_COMMITTED; + std::atomic prepare_done{false}; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", [&prepare_done](void*) { + if (prepare_done.load(std::memory_order_acquire)) { + ::_exit(1); + } + }); + SyncPoint::GetInstance()->EnableProcessing(); + TransactionDB* txn_db = nullptr; + ASSERT_OK(TransactionDB::Open(options, txn_opts, dbname_, &txn_db)); + TransactionOptions to; + to.skip_prepare = false; + Transaction* txn = txn_db->BeginTransaction(WriteOptions(), to); + ASSERT_OK(txn->SetName("xidq2")); + ASSERT_OK(txn->Put("pcq", "v")); + ASSERT_OK(txn->Prepare()); + prepare_done.store(true, std::memory_order_release); + ASSERT_OK(txn->Commit()); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, WriteCommittedPrepareTwoQueuesAndCommit) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.two_write_queues = true; + Destroy(options); + TransactionDBOptions txn_opts; + txn_opts.write_policy = TxnDBWritePolicy::WRITE_COMMITTED; + ASSERT_EQ(RunCrashChild(dbname_, "WriteCommittedPrepareTwoQueuesAndCommit"), 1); + TransactionDB* txn_db = nullptr; + ASSERT_OK(TransactionDB::Open(options, txn_opts, dbname_, &txn_db)); + std::vector prepared; + txn_db->GetAllPreparedTransactions(&prepared); + ASSERT_TRUE(prepared.empty()); + std::string v; + ASSERT_OK(txn_db->Get(ReadOptions(), "pcq", &v)); + ASSERT_EQ(v, "v"); + delete txn_db; +} +#endif + +TEST_F(DBCsppCrashSafeTest, WritePreparedFallsBackToWal) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.two_write_queues = true; + Destroy(options); + TransactionDBOptions txn_opts; + txn_opts.write_policy = TxnDBWritePolicy::WRITE_PREPARED; + TransactionDB* txn_db = nullptr; + ASSERT_OK(TransactionDB::Open(options, txn_opts, dbname_, &txn_db)); + Transaction* txn = txn_db->BeginTransaction(WriteOptions()); + ASSERT_OK(txn->Put("wp", "1")); + ASSERT_OK(txn->Commit()); + delete txn; + delete txn_db; + ASSERT_OK(TransactionDB::Open(options, txn_opts, dbname_, &txn_db)); + std::string v; + ASSERT_OK(txn_db->Get(ReadOptions(), "wp", &v)); + ASSERT_EQ(v, "1"); + delete txn_db; +} + +TEST_F(DBCsppCrashSafeTest, AfterConvertCloseSecondFlushInject) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("a", "1")); + ASSERT_OK(dbfull()->TEST_SwitchMemtable()); + ASSERT_OK(Put("b", "2")); + ASSERT_OK(dbfull()->TEST_SwitchMemtable()); + ASSERT_OK(Put("c", "3")); + std::atomic converts{0}; + SyncPoint::GetInstance()->SetCallBack( + "FlushJob::ConvertToSST:Status", [&converts](void* arg) { + if (converts.fetch_add(1) >= 1) { + *static_cast(arg) = Status::IOError("close second"); + } + }); + SyncPoint::GetInstance()->EnableProcessing(); + Close(); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_OK(env_->FileExists(CrashSafePubSeqFileName(dbname_))); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("a"), "1"); + ASSERT_EQ(Get("b"), "2"); + ASSERT_EQ(Get("c"), "3"); +} + +TEST_F(CrashChild, DISABLED_DualCfLeftoverConvertsBoth) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + std::atomic pubs{0}; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", [&pubs](void*) { + if (pubs.fetch_add(1) >= 1) { + ::_exit(1); + } + }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + std::vector hs; + std::vector cfs = { + {kDefaultColumnFamilyName, options}, {"one", options}}; + ASSERT_OK(DB::Open(options, dbname_, cfs, &hs, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), hs[0], "d", "1")); + ASSERT_OK(child_db->Put(WriteOptions(), hs[1], "c", "2")); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, DualCfLeftoverConvertsBoth) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + ASSERT_OK(TryReopen(options)); + CreateAndReopenWithCF({"one"}, options); + Close(); + ASSERT_EQ(RunCrashChild(dbname_, "DualCfLeftoverConvertsBoth"), 1); + ASSERT_OK(TryReopenWithColumnFamilies({kDefaultColumnFamilyName, "one"}, + options)); + ASSERT_EQ(Get(0, "d"), "1"); + ASSERT_EQ(Get(1, "c"), "2"); + ASSERT_GE(CountL0(db_, kDefaultColumnFamilyName), 1); + ASSERT_GE(CountL0(db_, "one"), 1); +} + +TEST_F(CrashChild, DISABLED_AtomicFlushDualCfLeftoverConvertsBoth) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.atomic_flush = true; + std::atomic pubs{0}; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", [&pubs](void*) { + if (pubs.fetch_add(1) >= 1) { + ::_exit(1); + } + }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + std::vector hs; + std::vector cfs = { + {kDefaultColumnFamilyName, options}, {"one", options}}; + ASSERT_OK(DB::Open(options, dbname_, cfs, &hs, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), hs[0], "d", "1")); + ASSERT_OK(child_db->Put(WriteOptions(), hs[1], "c", "2")); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, AtomicFlushDualCfLeftoverConvertsBoth) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.atomic_flush = true; + Destroy(options); + ASSERT_OK(TryReopen(options)); + CreateAndReopenWithCF({"one"}, options); + Close(); + ASSERT_EQ(RunCrashChild(dbname_, "AtomicFlushDualCfLeftoverConvertsBoth"), 1); + ASSERT_OK(TryReopenWithColumnFamilies({kDefaultColumnFamilyName, "one"}, + options)); + ASSERT_EQ(Get(0, "d"), "1"); + ASSERT_EQ(Get(1, "c"), "2"); + ASSERT_GE(CountL0(db_, kDefaultColumnFamilyName), 1); + ASSERT_GE(CountL0(db_, "one"), 1); +} + +TEST_F(CrashChild, DISABLED_AtomicFlushManifestWindow) { + ASSERT_EQ(arg_.size(), 2U); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.atomic_flush = true; + if (arg_[0] == '1') SetupOsl(&options, true); + DB* child_db = nullptr; + std::vector handles; + const std::vector cfs = { + {kDefaultColumnFamilyName, options}, {"one", options}}; + ASSERT_OK(DB::Open(options, dbname_, cfs, &handles, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), handles[0], "default-key", "one")); + ASSERT_OK(child_db->Put(WriteOptions(), handles[1], "other-key", "two")); + SyncPoint::GetInstance()->SetCallBack( + arg_[1] == '0' ? "FlushJob::BeforeManifest" : "FlushJob::AfterManifest", + [](void*) { ::_exit(42); }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(child_db->Flush(FlushOptions(), handles)); + ::_exit(1); +} + +TEST_F(DBCsppCrashSafeTest, AtomicFlushCrashAcrossManifestCommit) { + Close(); + for (bool osl : {false, true}) { + for (bool committed : {false, true}) { + SCOPED_TRACE(osl); + SCOPED_TRACE(committed); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.atomic_flush = true; + if (osl) SetupOsl(&options, true); + Destroy(options); + ASSERT_OK(TryReopen(options)); + CreateAndReopenWithCF({"one"}, options); + Close(); + const std::string arg = std::string(osl ? "1" : "0") + + (committed ? "1" : "0"); + ASSERT_EQ(RunCrashChild(dbname_, "AtomicFlushManifestWindow", arg), 42); + const auto registered = ListLeftovers(options, dbname_); + ASSERT_GE(registered.size(), 2U); + for (const auto& path : registered) ASSERT_OK(env_->FileExists(path)); + std::atomic converted{0}; + SyncPoint::GetInstance()->SetCallBack( + "MemTableRep::ConvertToSST:After", + [&](void*) { ++converted; }); + SyncPoint::GetInstance()->EnableProcessing(); + ASSERT_OK(TryReopenWithColumnFamilies( + {kDefaultColumnFamilyName, "one"}, options)); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_EQ(converted.load(), committed ? 0 : 2); + ASSERT_EQ(Get(0, "default-key"), "one"); + ASSERT_EQ(Get(1, "other-key"), "two"); + ASSERT_EQ(CountL0(db_, kDefaultColumnFamilyName), 1); + ASSERT_EQ(CountL0(db_, "one"), 1); + std::vector files; + db_->GetLiveFilesMetaData(&files); + ASSERT_EQ(files.size(), 2U); + for (const auto& file : files) { + const std::string path = MakeTableFileName(dbname_, file.file_number); + ASSERT_OK(env_->FileExists(path)); + ASSERT_EQ(std::count(registered.begin(), registered.end(), path), + committed ? 0 : 1); + } + Close(); + ASSERT_OK(TryReopenWithColumnFamilies( + {kDefaultColumnFamilyName, "one"}, options)); + ASSERT_EQ(Get(0, "default-key"), "one"); + ASSERT_EQ(Get(1, "other-key"), "two"); + ASSERT_EQ(CountL0(db_, kDefaultColumnFamilyName), 1); + ASSERT_EQ(CountL0(db_, "one"), 1); + Close(); + } + } +} + +TEST_F(CrashChild, DISABLED_DroppedCfLeftoverSkipped) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + std::atomic pubs{0}; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", [&pubs](void*) { + if (pubs.fetch_add(1) >= 1) { + ::_exit(1); + } + }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + std::vector hs; + std::vector cfs = { + {kDefaultColumnFamilyName, options}, {"one", options}}; + ASSERT_OK(DB::Open(options, dbname_, cfs, &hs, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), hs[0], "keep", "1")); + ASSERT_OK(child_db->Put(WriteOptions(), hs[1], "drop", "2")); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, DroppedCfLeftoverSkipped) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + ASSERT_OK(TryReopen(options)); + CreateAndReopenWithCF({"one"}, options); + Close(); + ASSERT_EQ(RunCrashChild(dbname_, "DroppedCfLeftoverSkipped"), 1); + ASSERT_OK(TryReopenWithColumnFamilies({kDefaultColumnFamilyName, "one"}, + options)); + ASSERT_EQ(Get(0, "keep"), "1"); + ASSERT_EQ(Get(1, "drop"), "2"); + ASSERT_OK(Flush(0)); + const auto registered = dbfull()->GetVersionSet()->GetColumnFamilySet() + ->GetColumnFamily(1)->GetMemTableFiles(); + ASSERT_FALSE(registered.empty()); + const std::string left1 = MakeTableFileName(dbname_, *registered.begin()); + ASSERT_FALSE(left1.empty()); + const std::string bak = left1 + ".bak"; + CopyFile(left1, bak); + ASSERT_OK(db_->DropColumnFamily(handles_[1])); + Close(); + if (::access(left1.c_str(), F_OK) != 0) { + ASSERT_EQ(::rename(bak.c_str(), left1.c_str()), 0); + } else { + ::unlink(bak.c_str()); + } + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("keep"), "1"); + ASSERT_EQ(dbfull()->GetVersionSet()->GetColumnFamilySet() + ->GetColumnFamily(1), nullptr); + const auto leftovers = ListLeftovers(options, dbname_); + ASSERT_EQ(std::count(leftovers.begin(), leftovers.end(), left1), 0); +} + +TEST_F(CrashChild, DISABLED_MultiChunkAfterCommitStillReadable) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.memtable_factory.reset(NewCSPPMemTabForPlain( + R"({"mem_cap":16777216,"convert_to_sst":"kFileMmap","chunk_size":4096})")); + std::atomic pubs{0}; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", [&pubs](void*) { + if (pubs.fetch_add(1) >= 199) { + ::_exit(1); + } + }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + for (int i = 0; i < 200; ++i) { + ASSERT_OK(child_db->Put(WriteOptions(), "k" + std::to_string(i), + std::string(80, 'v'))); + } + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, MultiChunkAfterCommitStillReadable) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.memtable_factory.reset(NewCSPPMemTabForPlain( + R"({"mem_cap":16777216,"convert_to_sst":"kFileMmap","chunk_size":4096})")); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "MultiChunkAfterCommitStillReadable"), 1); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("k0").size(), 80U); + ASSERT_EQ(Get("k199").size(), 80U); +} + +TEST_F(CrashChild, DISABLED_OslMultiChunkAfterCommitStillReadable) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + SetupOsl(&options, true); + std::atomic pubs{0}; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", [&pubs](void*) { + if (pubs.fetch_add(1) >= 399) { + ::_exit(1); + } + }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + for (int i = 0; i < 400; ++i) { + ASSERT_OK(child_db->Put(WriteOptions(), "ok" + std::to_string(i), + std::string(8192, 'o'))); + } + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, OslMultiChunkAfterCommitStillReadable) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + SetupOsl(&options, true); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "OslMultiChunkAfterCommitStillReadable"), 1); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("ok0").size(), 8192U); + ASSERT_EQ(Get("ok399").size(), 8192U); +} + +TEST_F(CrashChild, DISABLED_OslFirstChunkAfterCommitReadable) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + SetupOsl(&options, true); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", + [](void*) { ::_exit(1); }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "osl1", "v")); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, OslFirstChunkAfterCommitReadable) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + SetupOsl(&options, true); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "OslFirstChunkAfterCommitReadable"), 1); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("osl1"), "v"); + ColumnFamilyMetaData metadata; + db_->GetColumnFamilyMetaData(&metadata); + ASSERT_EQ(metadata.levels[0].files.size(), 1U); + ASSERT_TRUE(metadata.levels[0].files[0].marked_for_compaction); +} + +TEST_F(CrashChild, DISABLED_WalSwitchKeepsTail) { + Options options = BaseCrashSafeOptions(dbname_, true, false); + std::atomic pubs{0}; + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", [&pubs](void*) { + if (pubs.fetch_add(1) >= 1) { + ::_exit(1); + } + }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "old", "1")); + ASSERT_OK(static_cast(child_db)->TEST_SwitchWAL()); + ASSERT_OK(child_db->Put(WriteOptions(), "neu", "2")); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, WalSwitchKeepsTail) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "WalSwitchKeepsTail"), 1); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("old"), "1"); + ASSERT_EQ(Get("neu"), "2"); +} + +TEST_F(CrashChild, DISABLED_LeftoverLogRefUnbindFallsBackToWal) { + Options options = BaseCrashSafeOptions(dbname_, true, true); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterCommit", + [](void*) { ::_exit(1); }); + SyncPoint::GetInstance()->EnableProcessing(); + DB* child_db = nullptr; + ASSERT_OK(DB::Open(options, dbname_, &child_db)); + ASSERT_OK(child_db->Put(WriteOptions(), "rb", "ok")); + ::_exit(0); +} + +TEST_F(DBCsppCrashSafeTest, LeftoverLogRefUnbindFallsBackToWal) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, true); + Destroy(options); + ASSERT_EQ(RunCrashChild(dbname_, "LeftoverLogRefUnbindFallsBackToWal"), 1); + auto leftovers = ListLeftovers(options, dbname_); + ASSERT_FALSE(leftovers.empty()); + const int fd = ::open(leftovers[0].c_str(), O_RDWR); + ASSERT_GE(fd, 0); + terark::DFA_MmapHeader hdr{}; + ASSERT_EQ(::pread(fd, &hdr, sizeof(hdr), 0), + static_cast(sizeof(hdr))); + auto* cs = reinterpret_cast(hdr.reserved); + ASSERT_EQ(cs[0], 0x50505343U); + uint64_t fake_fileno = 999999; + const size_t wal0 = offsetof(terark::DFA_MmapHeader, reserved) + 16; + ASSERT_EQ(::pwrite(fd, &fake_fileno, sizeof(fake_fileno), + static_cast(wal0)), + static_cast(sizeof(fake_fileno))); + ::close(fd); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("rb"), "ok"); +} + +// Non-AVX builds hit AfterOddGeneration. +#if !defined(__AVX__) +TEST_F(DBCsppCrashSafeTest, CloseWaitsForStatsPublication) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.stats_dump_period_sec = 0; + options.stats_persist_period_sec = 1; + options.persist_stats_to_disk = true; + options.statistics = CreateDBStatistics(); + Destroy(options); + ASSERT_OK(TryReopen(options)); + SyncPoint::GetInstance()->LoadDependency({ + {"CloseWaitsForStatsPublication:Publishing", + "CloseWaitsForStatsPublication:BeginClose"}, + {"Timer::WaitForTaskCompleteIfNecessary:TaskExecuting", + "CloseWaitsForStatsPublication:Resume"}, + }); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::PersistPublishedSequence:AfterOddGeneration", [](void*) { + TEST_SYNC_POINT("CloseWaitsForStatsPublication:Publishing"); + TEST_SYNC_POINT("CloseWaitsForStatsPublication:Resume"); + }); + SyncPoint::GetInstance()->EnableProcessing(); + TEST_SYNC_POINT("CloseWaitsForStatsPublication:BeginClose"); + ASSERT_OK(db_->Close()); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + SyncPoint::GetInstance()->LoadDependency({}); + Close(); +} +#endif // !__AVX__ + +TEST_F(DBCsppCrashSafeTest, InFlightFlushThenClose) { + Close(); + Options options = BaseCrashSafeOptions(dbname_, true, false); + options.max_background_flushes = 1; + Destroy(options); + ASSERT_OK(TryReopen(options)); + ASSERT_OK(Put("if", "1")); + ASSERT_OK(dbfull()->TEST_SwitchMemtable()); + ASSERT_OK(Put("tail", "2")); + test::SleepingBackgroundTask sleeping_task; + SyncPoint::GetInstance()->SetCallBack( + "FlushJob::WriteLevel0Table", [&](void*) { sleeping_task.DoSleep(); }); + SyncPoint::GetInstance()->EnableProcessing(); + FlushOptions fo; + fo.wait = false; + const Status flush_status = db_->Flush(fo); + const bool paused = !sleeping_task.TimedWaitUntilSleeping(10 * 1000000); + if (!flush_status.ok() || !paused) sleeping_task.WakeUp(); + ASSERT_OK(flush_status); + ASSERT_TRUE(paused); + SyncPoint::GetInstance()->SetCallBack( + "DBImpl::FlushMemTable:AfterScheduleFlush", + [&](void*) { sleeping_task.WakeUp(); }); + std::thread closer([&] { Close(); }); + sleeping_task.WaitUntilDone(); + closer.join(); + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("if"), "1"); + ASSERT_EQ(Get("tail"), "2"); +} +#endif // !OS_WIN + +} // namespace ROCKSDB_NAMESPACE + +int main(int argc, char** argv) { + ROCKSDB_NAMESPACE::port::InstallStackTraceHandler(); + ::testing::InitGoogleTest(&argc, argv); +#if !defined(OS_WIN) + if (argc == 5 && std::strcmp(argv[1], "--crash-child") == 0) { + ROCKSDB_NAMESPACE::crash_child_db = argv[3]; + ROCKSDB_NAMESPACE::crash_child_arg = argv[4]; + ::testing::GTEST_FLAG(filter) = std::string("CrashChild.DISABLED_") + argv[2]; + ::testing::GTEST_FLAG(also_run_disabled_tests) = true; + ::alarm(30); + const int result = RUN_ALL_TESTS(); + // Every child action must reach its explicit _exit, not just finish a test. + return result == 0 ? 126 : 125; + } +#endif + return RUN_ALL_TESTS(); +} diff --git a/db/db_impl/db_impl.cc b/db/db_impl/db_impl.cc index 7765184761..cca68b6be9 100644 --- a/db/db_impl/db_impl.cc +++ b/db/db_impl/db_impl.cc @@ -381,9 +381,10 @@ DBImpl::DBImpl(const DBOptions& options, const std::string& dbname, // Reserve ten files or so for other uses and give the rest to TableCache. // Give a large number for setting of "infinite" open files. + // Small limits must not turn the 10-file reserve into a huge cache capacity. const int table_cache_size = (mutable_db_options_.max_open_files == -1) ? TableCache::kInfiniteCapacity - : mutable_db_options_.max_open_files - 10; + : std::max(0, mutable_db_options_.max_open_files - 10); LRUCacheOptions co; co.capacity = table_cache_size; co.num_shard_bits = immutable_db_options_.table_cache_numshardbits; @@ -619,7 +620,8 @@ void DBImpl::CancelAllBackgroundWork(bool wait) { InstrumentedMutexLock l(&mutex_); if (!shutting_down_.load(std::memory_order_acquire) && - has_unpersisted_data_.load(std::memory_order_relaxed) && + (has_unpersisted_data_.load(std::memory_order_relaxed) || + AllMemtablesSupportConvertToSST()) && !mutable_db_options_.avoid_flush_during_shutdown) { s = DBImpl::FlushAllColumnFamilies(FlushOptions(), FlushReason::kShutDown); s.PermitUncheckedError(); //**TODO: What to do on error? @@ -666,6 +668,8 @@ Status DBImpl::CloseHelper() { // marker. After this we do a variant of the waiting and unschedule work // (to consider: moving all the waiting into CancelAllBackgroundWork(true)) CancelAllBackgroundWork(false); + // Periodic PersistStats writes must finish before unmapping the cursor. + UnmapPublishedSeqFile(); // Cancel manual compaction if there's any if (HasPendingManualCompaction()) { @@ -1517,7 +1521,7 @@ Status DBImpl::SetDBOptions( new_options.delayed_write_rate); table_cache_.get()->SetCapacity(new_options.max_open_files == -1 ? TableCache::kInfiniteCapacity - : new_options.max_open_files - 10); + : std::max(0, new_options.max_open_files - 10)); wal_other_option_changed = mutable_db_options_.wal_bytes_per_sync != new_options.wal_bytes_per_sync; wal_size_option_changed = mutable_db_options_.max_total_wal_size != @@ -2298,7 +2302,6 @@ bool DBImpl::ShouldReferenceSuperVersion(const MergeContext& merge_context) { merge_context.GetOperands().size(); } -ROCKSDB_FLATTEN Status DBImpl::GetImpl(const ReadOptions& read_options, const Slice& key, GetImplOptions& get_impl_options) { #if defined(ROCKSDB_UNIT_TEST) @@ -4071,6 +4074,10 @@ Status DBImpl::CreateColumnFamilyImpl(const ColumnFamilyOptions& cf_options, // LogAndApply will both write the creation in MANIFEST and create // ColumnFamilyData object + auto pending_memtable = pending_outputs_.end(); + if (cf_options.memtable_factory->SupportCrashSafe()) { + pending_memtable = CaptureCurrentFileNumberInPendingOutputs(); + } { // write thread WriteThread::Writer w; write_thread_.EnterUnbatched(&w, &mutex_); @@ -4087,6 +4094,22 @@ Status DBImpl::CreateColumnFamilyImpl(const ColumnFamilyOptions& cf_options, assert(cfd != nullptr); std::map> dummy_created_dirs; s = cfd->AddDirectories(&dummy_created_dirs); + if (s.ok() && immutable_db_options_.memtable_crash_safe_recover) { + cfd->PrepareNewMemtableInBackground(*cfd->GetLatestMutableCFOptions()); + s = RegisterMemTableFiles(cfd); + if (!s.ok()) { + // CF creation is already committed. Keep its in-memory state valid, + // but do not return a writable handle after registration fails. + InstallSuperVersionAndScheduleWork(cfd, &sv_context, + *cfd->GetLatestMutableCFOptions()); + cfd->set_initialized(); + error_handler_.SetBGError(s, BackgroundErrorReason::kManifestWrite) + .PermitUncheckedError(); + } + } + } + if (pending_memtable != pending_outputs_.end()) { + pending_outputs_.erase(pending_memtable); } if (s.ok()) { auto* cfd = @@ -5917,6 +5940,12 @@ Status DestroyDB(const std::string& dbname, const Options& options, } } } + env->DeleteFile(CrashSafePubSeqFileName(dbname)).PermitUncheckedError(); + for (const auto& fname : filenames) { + if (fname.find(".memtab-") != std::string::npos) { + env->DeleteFile(dbname + "/" + fname).PermitUncheckedError(); + } + } std::set paths; for (const DbPath& db_path : options.db_paths) { diff --git a/db/db_impl/db_impl.h b/db/db_impl/db_impl.h index 759fa9b07c..ff8531ba79 100644 --- a/db/db_impl/db_impl.h +++ b/db/db_impl/db_impl.h @@ -78,6 +78,7 @@ class TaskLimiterToken; class Version; class VersionEdit; class VersionSet; +struct PublishedSeqMmapHeader; class WriteCallback; struct JobContext; struct ExternalSstFileInfo; @@ -1076,7 +1077,16 @@ class DBImpl : public DB { static Status Open(const DBOptions& db_options, const std::string& name, const std::vector& column_families, std::vector* handles, DB** dbptr, - const bool seq_per_batch, const bool batch_per_txn); + const bool seq_per_batch, const bool batch_per_txn, + bool options_already_updated = false); + + // memtable_crash_safe_recover: if CSPUBSEQ's wal_offset_kind differs + // from memtable_as_log_index, open with the sidecar kind, flush, close and + // delete CSPUBSEQ. Call before Open() with the same seq/txn flags. + static Status PrepareCrashSafeKindForOpen( + const DBOptions& db_options, const std::string& name, + const std::vector& column_families, + bool seq_per_batch, bool batch_per_txn); static IOStatus CreateAndNewDirectory( FileSystem* fs, const std::string& dirname, @@ -1427,6 +1437,13 @@ class DBImpl : public DB { // such a file's absolute path to its parent directory. std::unordered_map files_to_delete_; bool is_new_db_ = false; + // WAL tail cursor for this Recover. RecoverLogFiles reads it from here. + bool crash_safe_wal_tail_replay_ = false; + uint8_t crash_safe_wal_offset_kind_ = 0; + uint64_t crash_safe_wal_number_ = 0; + uint64_t crash_safe_wal_offset_ = 0; + // WALs numbered below this predate CSPUBSEQ: probe their format. + uint64_t crash_safe_probe_wal_below_ = 0; }; // Persist options to options file. Must be holding options_mutex_. @@ -1466,6 +1483,8 @@ class DBImpl : public DB { void NotifyOnExternalFileIngested( ColumnFamilyData* cfd, const ExternalSstFileIngestionJob& ingestion_job); + bool AllMemtablesSupportConvertToSST() const; + Status FlushAllColumnFamilies(const FlushOptions& flush_options, FlushReason flush_reason); @@ -1929,6 +1948,28 @@ class DBImpl : public DB { bool* corrupted_log_found, RecoveryContext* recovery_ctx); + // Kind-prep first pass under allow_2pc: NotSupported if prepared + // transactions were recovered, else persist min_log_number_to_keep past + // the old-format WALs so the next Open does not replay them. + Status RetireWalsForKindPrep(); + + // *probe_wal_below: WALs numbered below it have unknown format. + Status MapPublishedSeqFile(uint64_t* probe_wal_below); + void UnmapPublishedSeqFile(); + void PersistPublishedSequence(SequenceNumber seq, uint64_t wal_number, + uint64_t wal_offset); + void MaybePersistPublishedSequence(SequenceNumber seq, + const WriteThread::WriteGroup& write_group); + void PersistStagedPublishedWal(); + void StagePublishedWal(SequenceNumber seq, uint64_t wal_number, + uint64_t wal_offset); + void AccountPendingMemtableWrites(size_t n); + Status RegisterMemTableFiles(ColumnFamilyData* cfd); + bool CanConvertLeftoverForCrashSafeRecover( + SequenceNumber mmap_pubseq, std::string* fail_reason); + Status ConvertLeftoverMemtables( + SequenceNumber max_visible_seq, RecoveryContext* recovery_ctx); + // The following two methods are used to flush a memtable to // storage. The first one is used at database RecoveryTime (when the // database is opened) and is heavyweight because it holds the mutex @@ -2742,6 +2783,19 @@ class DBImpl : public DB { // It contains the implementations for each periodic task. std::map periodic_task_functions_; + PublishedSeqMmapHeader* pubseq_mmap_ = nullptr; + size_t pubseq_mmap_size_ = 0; + intptr_t pubseq_fd_ = -1; + SequenceNumber staged_pub_seq_ = 0; + uint64_t staged_pub_wal_number_ = 0; + uint64_t staged_pub_wal_offset_ = 0; + + // Staged WAL cursor shared by unordered_write and pipelined_write. + // After this group's WriteWAL, persist the previously staged cursor if + // pending_memtable_writes_ == 0, then stage this group's cursor (mmap + // trails one WAL). + bool staged_pub_valid_ = false; + // When set, we use a separate queue for writes that don't write to memtable. // In 2PC these are the writes at Prepare phase. const bool two_write_queues_; diff --git a/db/db_impl/db_impl_compaction_flush.cc b/db/db_impl/db_impl_compaction_flush.cc index 2dc47c9045..a6e012bce4 100644 --- a/db/db_impl/db_impl_compaction_flush.cc +++ b/db/db_impl/db_impl_compaction_flush.cc @@ -1982,6 +1982,18 @@ int DBImpl::Level0StopWriteTrigger(ColumnFamilyHandle* column_family) { ->mutable_cf_options.level0_stop_writes_trigger; } +bool DBImpl::AllMemtablesSupportConvertToSST() const { + mutex_.AssertHeld(); + for (auto* cfd : *versions_->GetColumnFamilySet()) { + if (!cfd->IsDropped() && + (cfd->mem() == nullptr || !cfd->mem()->SupportConvertToSST() || + !cfd->imm()->UnflushedMemtablesSupportConvertToSST())) { + return false; + } + } + return true; +} + Status DBImpl::FlushAllColumnFamilies(const FlushOptions& flush_options, FlushReason flush_reason) { mutex_.AssertHeld(); @@ -3310,6 +3322,23 @@ Status DBImpl::BackgroundFlush(bool* made_progress, JobContext* job_context, } #endif /* !NDEBUG */ *reason = bg_flush_args[0].flush_reason_; + if (status.ok() && + (shutdown_initiated_ || *reason == FlushReason::kShutDown)) { + // Conversion picks one memtable at a time; finish the shutdown flush. + FlushRequest remaining{*reason, {}}; + for (const auto& arg : bg_flush_args) { + auto* cfd = arg.cfd_; + if (!cfd->IsDropped() && cfd->imm()->NumNotFlushed() != 0 && + cfd->imm()->GetEarliestMemTableID() <= arg.max_memtable_id_) { + cfd->imm()->FlushRequested(); + if (cfd->imm()->IsFlushPending()) { + remaining.cfd_to_max_mem_id_to_persist.emplace( + cfd, arg.max_memtable_id_); + } + } + } + SchedulePendingFlush(remaining); + } for (auto& arg : bg_flush_args) { ColumnFamilyData* cfd = arg.cfd_; if (cfd->UnrefAndTryDelete()) { @@ -4370,7 +4399,8 @@ Status DBImpl::WaitForCompact( return s; } } else if (wait_for_compact_options.close_db && - has_unpersisted_data_.load(std::memory_order_relaxed) && + (has_unpersisted_data_.load(std::memory_order_relaxed) || + AllMemtablesSupportConvertToSST()) && !mutable_db_options_.avoid_flush_during_shutdown) { Status s = DBImpl::FlushAllColumnFamilies(FlushOptions(), FlushReason::kShutDown); diff --git a/db/db_impl/db_impl_files.cc b/db/db_impl/db_impl_files.cc index 47901f4d2c..e3659ced51 100644 --- a/db/db_impl/db_impl_files.cc +++ b/db/db_impl/db_impl_files.cc @@ -175,6 +175,14 @@ void DBImpl::FindObsoleteFiles(JobContext* job_context, bool force, if (doing_the_full_scan) { versions_->AddLiveFiles(&job_context->sst_live, &job_context->blob_live); + // These share the SST namespace, but are still mutable. Protect them from + // GC without exposing them as immutable SSTs to checkpoints and backups. + for (auto* cfd : *versions_->GetColumnFamilySet()) { + const auto& files = cfd->GetMemTableFiles(); + job_context->sst_live.insert(job_context->sst_live.end(), + files.begin(), files.end()); + cfd->AddMemTableFileNumbers(&job_context->sst_live); + } InfoLogPrefix info_log_prefix(!immutable_db_options_.db_log_dir.empty(), dbname_); std::set paths; diff --git a/db/db_impl/db_impl_open.cc b/db/db_impl/db_impl_open.cc index 8c62470ac0..edaecd6bd5 100644 --- a/db/db_impl/db_impl_open.cc +++ b/db/db_impl/db_impl_open.cc @@ -7,6 +7,13 @@ // Use of this source code is governed by a BSD-style license that can be // found in the LICENSE file. See the AUTHORS file for names of contributors. #include +#ifndef _MSC_VER +#include +#include +#endif +#if defined(__AVX__) +#include +#endif #include "db/builder.h" #include "db/db_impl/db_impl.h" @@ -25,6 +32,9 @@ #include "rocksdb/wal_filter.h" #include "test_util/sync_point.h" #include "util/rate_limiter_impl.h" +#include +#include +#include #include "util/string_util.h" #include "util/udt_util.h" @@ -47,9 +57,25 @@ DBOptions SanitizeOptions(const std::string& dbname, const DBOptions& src, result.env = Env::Default(); } - if (result.memtable_as_log_index) { + if (read_only && result.memtable_crash_safe_recover) { + result.memtable_crash_safe_recover = false; + const char* warning = + "memtable_crash_safe_recover is disabled for read-only Open; full WAL " + "replay may be very slow because using much larger MemTables than in " + "typical RocksDB configurations is encouraged in crash-safe mode"; + if (result.info_log) { + ROCKS_LOG_WARN(result.info_log, "%s", warning); + } else { + fprintf(stderr, "%s WARNING: %s\n", terark::StrDateTimeNow(), warning); + } + } + + if (result.memtable_as_log_index || result.memtable_crash_safe_recover) { result.recycle_log_file_num = 0; result.manual_wal_flush = false; + result.wal_compression = kNoCompression; + } + if (result.memtable_as_log_index) { #if !defined(ROCKSDB_UNIT_TEST) // avoid infrequent CFs reference too many WALs when frequent CFs // writing many data @@ -63,7 +89,9 @@ DBOptions SanitizeOptions(const std::string& dbname, const DBOptions& src, if (max_max_open_files == -1) { max_max_open_files = 0x400000; } - ClipToRange(&result.max_open_files, 20, max_max_open_files); + // Do not raise small limits: turning 0 into 20 would force SST preloading + // despite the caller's request for a cheap open. + terark::minimize(result.max_open_files, max_max_open_files); TEST_SYNC_POINT_CALLBACK("SanitizeOptions::AfterChangeMaxOpenFiles", &result.max_open_files); } @@ -424,11 +452,494 @@ IOStatus Directories::SetDirectories(FileSystem* fs, const std::string& dbname, return IOStatus::OK(); } +struct alignas(32) PublishedSeqRecord { + uint64_t pubseq = 0; + uint64_t wal_number = 0; + uint64_t wal_offset = 0; + uint64_t padding = 0; +}; + +struct alignas(CACHE_LINE_SIZE) PublishedSeqMmapHeader { + uint64_t magic = 0; + uint32_t version = 0; + uint32_t header_size = 0; + // 1 physical, 2 log-index. Written by Stamp, not by each publish. + uint32_t wal_offset_kind = 0; + // First WAL written under wal_offset_kind; 0 until that WAL exists. + // Older WALs predate this sidecar and have unknown format. + uint32_t kind_since_wal = 0; + uint64_t generation = 0; + PublishedSeqRecord rec; + + void Stamp(bool memtable_as_log_index); +}; + +namespace { + +constexpr size_t kPublishedSeqMmapSize = 4096; +constexpr uint64_t kPublishedSeqMagic = 0x5145534255505343ULL; // 'CSPUBSEQ' +constexpr uint32_t kPublishedSeqVersion = 1; + +enum class PublishedWalOffsetKind : uint32_t { + kNone = 0, + kPhysical = 1, + kLogIndex = 2, +}; + +static_assert(std::atomic::is_always_lock_free); +static_assert(sizeof(std::atomic) == sizeof(uint64_t)); +static_assert(sizeof(PublishedSeqMmapHeader) == CACHE_LINE_SIZE); +static_assert(alignof(PublishedSeqMmapHeader) == CACHE_LINE_SIZE); +static_assert(sizeof(PublishedSeqMmapHeader) <= kPublishedSeqMmapSize); +static_assert(sizeof(PublishedSeqRecord) == 32); +static_assert(offsetof(PublishedSeqMmapHeader, rec) % 32 == 0); + +bool PublishedSeqHeaderValid(const PublishedSeqMmapHeader* hdr) { + return hdr->magic == kPublishedSeqMagic && + hdr->version == kPublishedSeqVersion && + hdr->header_size == sizeof(PublishedSeqMmapHeader) && + (hdr->wal_offset_kind == 1 || // kPhysical + hdr->wal_offset_kind == 2); // kLogIndex +} + +uint64_t PublishedWalRecordStart(uint64_t offset, uint64_t kind) { + if (kind == 2) { // PublishedWalOffsetKind::kLogIndex + return offset >= sizeof(log::RawRecHeader) + ? offset - sizeof(log::RawRecHeader) + : 0; + } + return offset; +} + +bool ReadPublishedSeqRecord(const PublishedSeqMmapHeader* hdr, + PublishedSeqRecord* out) { + if ((hdr->generation & 1) == 0) { + *out = hdr->rec; + return true; + } else { + return false; + } +} + +bool PeekPublishedWalKind(const std::string& dbname, uint64_t* kind) { + const std::string path = CrashSafePubSeqFileName(dbname); + PublishedSeqMmapHeader hdr; + int fd = ::open(path.c_str(), O_RDONLY | O_CLOEXEC); + if (fd < 0) { + return false; + } + auto sz = ::pread(fd, &hdr, sizeof(hdr), 0); + ::close(fd); + if (sz != static_cast(sizeof(hdr))) { + return false; + } + // wal_offset_kind is written only by Stamp, so an odd generation from a + // crash mid-publish does not make it stale. + // A Stamp from an unfinished Open has not established a WAL format yet. + bool ok = PublishedSeqHeaderValid(&hdr) && hdr.kind_since_wal != 0; + if (ok) { + *kind = hdr.wal_offset_kind; + } + return ok; +} + +void DestroyKindPrepDb(DB* db, std::vector* handles) { + for (auto* h : *handles) { + delete h; + } + handles->clear(); + if (db != nullptr) { + db->Close().PermitUncheckedError(); + delete db; + } +} + +} // namespace + +Status DBImpl::PrepareCrashSafeKindForOpen( + const DBOptions& db_options, const std::string& dbname, + const std::vector& column_families, + bool seq_per_batch, bool batch_per_txn) { + if (!db_options.memtable_crash_safe_recover) { + return Status::OK(); + } + uint64_t sidecar_kind = 0; + if (!PeekPublishedWalKind(dbname, &sidecar_kind) || sidecar_kind == 0) { + return Status::OK(); + } + const uint64_t opt_kind = db_options.memtable_as_log_index ? 2 : 1; + if (sidecar_kind == opt_kind) { + return Status::OK(); + } + DBOptions cheap = db_options; + cheap.memtable_as_log_index = sidecar_kind == 2; // kLogIndex + cheap.memtable_crash_safe_recover = true; + cheap.max_open_files = 0; + cheap.persist_stats_to_disk = false; + std::vector cheap_cfs = column_families; + for (auto& cf : cheap_cfs) { + cf.options.disable_auto_compactions = true; + } + std::vector handles; + DB* db = nullptr; + Status s = DBImpl::Open(cheap, dbname, cheap_cfs, &handles, &db, + seq_per_batch, batch_per_txn, + true /* options_already_updated */); + if (!s.ok()) { + return s; + } + TEST_SYNC_POINT("DBImpl::Open::KindPrep:AfterOpenBeforeFlush"); + FlushOptions fo; + fo.wait = true; + s = db->Flush(fo, handles); + if (s.ok() && cheap.allow_2pc) { + s = static_cast(db)->RetireWalsForKindPrep(); + } + if (!s.ok()) { + DestroyKindPrepDb(db, &handles); + return s; + } + TEST_SYNC_POINT("DBImpl::Open::KindPrep:AfterFlushBeforeClose"); + for (auto* h : handles) { + delete h; + } + handles.clear(); + s = db->Close(); + delete db; + db = nullptr; + if (!s.ok()) { + return s; + } + TEST_SYNC_POINT("DBImpl::Open::KindPrep:AfterCloseBeforeDeleteSidecar"); + Env* env = db_options.env != nullptr ? db_options.env : Env::Default(); + s = env->DeleteFile(CrashSafePubSeqFileName(dbname)); + if (!s.ok() && !s.IsNotFound()) { + return s; + } + return Status::OK(); +} + +Status DBImpl::RetireWalsForKindPrep() { + InstrumentedMutexLock l(&mutex_); + if (!recovered_transactions_.empty()) { + return Status::NotSupported( + "memtable_as_log_index change with prepared transactions", + "reopen with the previous memtable_as_log_index, commit or roll back " + "them, then switch"); + } + const uint64_t min_log = versions_->MinLogNumberWithUnflushedData(); + if (min_log <= versions_->min_log_number_to_keep()) { + return Status::OK(); + } + VersionEdit edit; + if (immutable_db_options_.track_and_verify_wals_in_manifest) { + edit.DeleteWalsBefore(min_log); + } + edit.SetMinLogNumberToKeep(min_log); + auto* cfd = versions_->GetColumnFamilySet()->GetDefault(); + const ReadOptions read_options(Env::IOActivity::kDBOpen); + return versions_->LogAndApply(cfd, *cfd->GetLatestMutableCFOptions(), + read_options, &edit, &mutex_, + directories_.GetDbDir()); +} + +void PublishedSeqMmapHeader::Stamp(bool memtable_as_log_index) { + magic = kPublishedSeqMagic; + version = kPublishedSeqVersion; + header_size = static_cast(sizeof(PublishedSeqMmapHeader)); + wal_offset_kind = memtable_as_log_index ? 2 : 1; // kLogIndex : kPhysical + generation = 0; + rec = PublishedSeqRecord{}; + kind_since_wal = 0; +} + +Status DBImpl::MapPublishedSeqFile(uint64_t* probe_wal_below) { + *probe_wal_below = std::numeric_limits::max(); + if (!immutable_db_options_.memtable_crash_safe_recover) { + return Status::OK(); + } + if (pubseq_mmap_ != nullptr) { + return Status::OK(); + } + const std::string path = CrashSafePubSeqFileName(dbname_); + size_t sz = 0; // new or empty file: 4K; existing file: mapped whole + intptr_t fd = -1; + void* p = nullptr; + try { + p = terark::mmap_write(path, &sz, &fd); + if (sz < kPublishedSeqMmapSize) { + terark::mmap_close(p, sz, fd); + sz = kPublishedSeqMmapSize; + p = terark::mmap_write(path, &sz, &fd); + } + } catch (const std::exception& ex) { + return Status::IOError(path, ex.what()); + } + pubseq_mmap_ = static_cast(p); + pubseq_mmap_size_ = sz; + pubseq_fd_ = fd; + if (!PublishedSeqHeaderValid(pubseq_mmap_) || + pubseq_mmap_->kind_since_wal == 0) { + TEST_SYNC_POINT("DBImpl::MapPublishedSeqFile:AfterMmapBeforeStamp"); + pubseq_mmap_->Stamp(immutable_db_options_.memtable_as_log_index); + return Status::OK(); + } + const uint32_t opt_kind = immutable_db_options_.memtable_as_log_index + ? uint32_t(PublishedWalOffsetKind::kLogIndex) + : uint32_t(PublishedWalOffsetKind::kPhysical); + if (pubseq_mmap_->wal_offset_kind != opt_kind) { + // DB::Open / TransactionDB::Open switch kind before getting here. + UnmapPublishedSeqFile(); + return Status::InvalidArgument( + path, "wal_offset_kind differs from memtable_as_log_index"); + } + if (pubseq_mmap_->kind_since_wal != 0) { + *probe_wal_below = pubseq_mmap_->kind_since_wal; + } + return Status::OK(); +} + +void DBImpl::UnmapPublishedSeqFile() { + if (pubseq_mmap_ != nullptr) { + terark::mmap_close(pubseq_mmap_, pubseq_mmap_size_, pubseq_fd_); + pubseq_mmap_ = nullptr; + pubseq_mmap_size_ = 0; + pubseq_fd_ = -1; + } +} + +void DBImpl::PersistPublishedSequence(SequenceNumber seq, uint64_t wal_number, + uint64_t wal_offset) { + if (pubseq_mmap_ == nullptr) { + return; + } + auto* rec = &pubseq_mmap_->rec; + // DO NOT REWRITE. Do not split this back into + // `if (seq < pubseq) return; if (seq == pubseq) SameSeq;`. + // That form compares twice on the hot path. Here the hot path is one + // UNLIKELY(seq <= pubseq). seq == pubseq is not an error: it falls + // through and publishes. Only seq < pubseq logs and returns. Do not + // return on <=. + if (UNLIKELY(seq <= rec->pubseq)) { + if (UNLIKELY(seq < rec->pubseq)) { + ROCKS_LOG_ERROR(immutable_db_options_.info_log, + "PersistPublishedSequence seq %" PRIu64 + " < published %" PRIu64 " wal #%" PRIu64 + " offset %" PRIu64, + seq, rec->pubseq, wal_number, wal_offset); + return; + } + TEST_SYNC_POINT("DBImpl::PersistPublishedSequence:SameSeq"); + } +#if defined(__AVX__) + // One aligned 32-byte store. A process crash falls between instructions, + // so the record is all old or all new. + // PublishedSeqRecord next; + // next.pubseq = seq; + // next.wal_number = wal_number; + // next.wal_offset = wal_offset; + // next.padding = 0; + // _mm256_store_si256((__m256i*)rec, _mm256_load_si256((const __m256i*)&next)); + // Lowest 64 bits are pubseq, matching PublishedSeqRecord field order. + __m256i packed = _mm256_set_epi64x(0, wal_offset, wal_number, seq); + // Clang/LLVM specifies that the backend must not split or merge target-legal + // volatile loads/stores. This keeps the AVX store a single 32-byte instruction. + // https://llvm.org/docs/LangRef.html#volatile-memory-accesses + *(volatile __m256i*)rec = packed; + pubseq_mmap_->generation += 2; +#else + // Use the odd/even bracket also for files moved from an AVX host. + // A failed publication may leave an odd generation. + const uint64_t g = pubseq_mmap_->generation | 1; + // Keep generation odd throughout the field stores, even if already odd. + // Acquire keeps the field stores after this exchange; the release below + // keeps them before publication of the next even generation. + terark::as_atomic(pubseq_mmap_->generation) + .exchange(g, std::memory_order_acquire); + TEST_SYNC_POINT("DBImpl::PersistPublishedSequence:AfterOddGeneration"); + rec->wal_number = wal_number; + rec->wal_offset = wal_offset; + rec->pubseq = seq; + terark::as_atomic(pubseq_mmap_->generation) + .store(g + 1, std::memory_order_release); +#endif + TEST_SYNC_POINT("DBImpl::PersistPublishedSequence:AfterCommit"); +} + +void DBImpl::MaybePersistPublishedSequence( + SequenceNumber seq, const WriteThread::WriteGroup& write_group) { + if (pubseq_mmap_ == nullptr) { + return; + } + TEST_SYNC_POINT("DBImpl::WriteImpl:AfterWriteToWALBeforePublish"); + PersistPublishedSequence(seq, write_group.wal_number, write_group.wal_offset); +} + +void DBImpl::PersistStagedPublishedWal() { + if (pubseq_mmap_ == nullptr || !staged_pub_valid_) { + return; + } + if (pending_memtable_writes_.load(std::memory_order_acquire) != 0) { + return; + } + PersistPublishedSequence(staged_pub_seq_, staged_pub_wal_number_, + staged_pub_wal_offset_); +} + +void DBImpl::StagePublishedWal(SequenceNumber seq, uint64_t wal_number, + uint64_t wal_offset) { + PersistStagedPublishedWal(); + staged_pub_seq_ = seq; + staged_pub_wal_number_ = wal_number; + staged_pub_wal_offset_ = wal_offset; + staged_pub_valid_ = true; +} + +void DBImpl::AccountPendingMemtableWrites(size_t n) { + if (n == 0) { + return; + } + const size_t pending_cnt = pending_memtable_writes_.fetch_sub(n) - n; + if (pending_cnt == 0) { + std::lock_guard lck(switch_mutex_); + switch_cv_.notify_all(); + } +} + +Status DBImpl::RegisterMemTableFiles(ColumnFamilyData* cfd) { + mutex_.AssertHeld(); + VersionEdit edit; + cfd->AddMemTableFileEdits(&edit); + edit.SetColumnFamily(cfd->GetID()); + TEST_SYNC_POINT("DBImpl::RegisterMemTableFile:BeforeLogAndApply"); + Status s = versions_->LogAndApply(cfd, *cfd->GetLatestMutableCFOptions(), + ReadOptions(), &edit, &mutex_, + directories_.GetDbDir()); + TEST_SYNC_POINT_CALLBACK("DBImpl::RegisterMemTableFile:AfterLogAndApply", &s); + if (s.ok()) { + cfd->PublishRegisteredMemTables(); + TEST_SYNC_POINT("DBImpl::RegisterMemTableFile:BeforeInstall"); + } + return s; +} + +bool DBImpl::CanConvertLeftoverForCrashSafeRecover( + SequenceNumber mmap_pubseq, std::string* fail_reason) { + mutex_.AssertHeld(); + auto fail = [&](const std::string& reason) { + *fail_reason = reason; + return false; + }; + if (!versions_->HasMemTableFileTracking()) { + return fail("MANIFEST predates MemTable file tracking"); + } + if (immutable_db_options_.wal_filter != nullptr) { + return fail("wal_filter requires full WAL replay"); + } + if (!last_seq_same_as_publish_seq_) { + return fail("seq_per_batch && two_write_queues"); + } + if (immutable_db_options_.best_efforts_recovery) { + return fail("best_efforts_recovery"); + } + if (mmap_pubseq == 0) { + for (auto* cfd : *versions_->GetColumnFamilySet()) { + if (!cfd->GetMemTableFiles().empty()) { + return fail("CSPUBSEQ pubseq 0 with leftover present"); + } + } + } + return true; +} + +Status DBImpl::ConvertLeftoverMemtables( + SequenceNumber max_visible_seq, RecoveryContext* recovery_ctx) { + mutex_.AssertHeld(); + struct ConvertedLeftover { + ColumnFamilyData* cfd; + FileMetaData meta; + std::vector blobs; + }; + std::vector converted; + // MANIFEST is the complete inventory, including empty, precreated tables. + // A directory scan cannot detect one missing file among several in a CF. + for (auto* cfd : *versions_->GetColumnFamilySet()) { + for (uint64_t file_num : cfd->GetMemTableFiles()) { + auto& one = converted.emplace_back(); + one.cfd = cfd; + one.meta.fd = FileDescriptor(file_num, 0, 0); + } + } + if (converted.empty()) return Status::OK(); + Status s; + mutex_.Unlock(); + for (auto& one : converted) { + auto* cfd = one.cfd; + const uint64_t file_num = one.meta.fd.GetNumber(); + const auto leftover_path = TableFileName(cfd->ioptions()->cf_paths, file_num, 0); + one.meta.fd.smallest_seqno = 0; + one.meta.fd.largest_seqno = max_visible_seq; + one.meta.epoch_number = cfd->NewEpochNumber(); + const MutableCFOptions moptions = *cfd->GetLatestMutableCFOptions(); + TableBuilderOptions tboptions( + *cfd->ioptions(), moptions, cfd->internal_comparator(), + cfd->int_tbl_prop_collector_factories(), + GetCompressionFlush(*cfd->ioptions(), moptions), + moptions.compression_opts, cfd->GetID(), cfd->GetName(), 0 /* level */, + false /* is_bottommost */, TableFileCreationReason::kRecovery, + 0 /* oldest_key_time */, 0 /* file_creation_time */, db_id_, + db_session_id_, 0 /* target_file_size */, file_num); + tboptions.generate_file_no = [this]() { return versions_->NewFileNumber(); }; + tboptions.add_blob_file = [&one](BlobFileAddition b) { + one.blobs.push_back(std::move(b)); + }; + s = cfd->ioptions()->memtable_factory->RecoverCrashSafeMemTableToSST( + leftover_path, &one.meta, tboptions); + if (!s.ok()) { + break; + } + } + mutex_.Lock(); + if (!s.ok()) { + // Do not retire registered sources until full WAL recovery commits. + for (auto& one : converted) { + for (const auto& blob : one.blobs) { + env_->DeleteFile(BlobFileName(one.cfd->ioptions()->cf_paths.front().path, + blob.GetBlobFileNumber())) + .PermitUncheckedError(); + } + } + return s; + } + for (auto& one : converted) { + VersionEdit edit; + edit.SetColumnFamily(one.cfd->GetID()); + edit.DeleteMemTableFile(one.meta.fd.GetNumber()); + if (one.meta.fd.GetFileSize() != 0) { + one.meta.marked_for_compaction = true; + edit.AddFile(0, one.meta); + } + for (const auto& blob : one.blobs) { + edit.AddBlobFile(blob); + } + recovery_ctx->UpdateVersionEdits(one.cfd, edit); + } + return Status::OK(); +} + Status DBImpl::Recover( const std::vector& column_families, bool read_only, bool error_if_wal_file_exists, bool error_if_data_exists_in_wals, uint64_t* recovered_seq, RecoveryContext* recovery_ctx) { mutex_.AssertHeld(); + if (read_only) { + for (const auto& cf : column_families) { + if (cf.options.memtable_factory->SupportCrashSafe()) { + return Status::InvalidArgument( + "FileMmap memtable is not supported in read-only mode", cf.name); + } + } + } bool tmp_is_new_db = false; bool& is_new_db = recovery_ctx ? recovery_ctx->is_new_db_ : tmp_is_new_db; @@ -556,6 +1067,25 @@ Status DBImpl::Recover( if (!s.ok()) { return s; } + if (!read_only && immutable_db_options_.memtable_crash_safe_recover) { + s = MapPublishedSeqFile(&recovery_ctx->crash_safe_probe_wal_below_); + if (!s.ok()) { + return s; + } + } else if (!read_only) { + // This session writes WALs without updating CSPUBSEQ, so an old + // one would misstate their format to a later crash-safe Open. + const std::string path = CrashSafePubSeqFileName(dbname_); + if (env_->FileExists(path).ok()) { + s = env_->DeleteFile(path); + if (!s.ok()) { + return s; + } + ROCKS_LOG_WARN(immutable_db_options_.info_log, + "memtable_crash_safe_recover is off, deleted %s", + path.c_str()); + } + } if (s.ok() && !read_only) { for (auto cfd : *versions_->GetColumnFamilySet()) { // Try to trivially move files down the LSM tree to start from bottommost @@ -656,10 +1186,60 @@ Status DBImpl::Recover( } s = SetupDBId(read_only, recovery_ctx); ROCKS_LOG_INFO(immutable_db_options_.info_log, "DB ID: %s\n", db_id_.c_str()); + bool crash_safe_convert = false; + PublishedSeqRecord mmap_rec; + bool mmap_valid = false; + if (s.ok() && !read_only && + immutable_db_options_.memtable_crash_safe_recover) { + mmap_valid = ReadPublishedSeqRecord(pubseq_mmap_, &mmap_rec); + std::string fail_reason; + if (mmap_valid) { + crash_safe_convert = CanConvertLeftoverForCrashSafeRecover( + mmap_rec.pubseq, &fail_reason); + } else { + fail_reason = "invalid CSPUBSEQ generation"; + } + if (!crash_safe_convert) { + ROCKS_LOG_WARN(immutable_db_options_.info_log, + "Crash-safe recover check failed (%s), fallback to full " + "WAL RecoverLogFiles", + fail_reason.c_str()); + } + } if (s.ok() && !read_only) { s = DeleteUnreferencedSstFiles(recovery_ctx); } - + if (s.ok()) { // CrashSafe needs to create an sst with next_file_number_. + // next_file_number_ must be restored before creating MemTables. + // WAL replay may contain sequences below LastSequence(). + for (auto* cfd : *versions_->GetColumnFamilySet()) { + if (cfd->mem() == nullptr) { + assert(cfd->ioptions()->memtable_factory->SupportCrashSafe()); + cfd->CreateNewMemtable(*cfd->GetLatestMutableCFOptions(), 0); + } + } + } + if (s.ok() && crash_safe_convert) { + const Status cs = ConvertLeftoverMemtables(mmap_rec.pubseq, recovery_ctx); + if (!cs.ok()) { + ROCKS_LOG_WARN(immutable_db_options_.info_log, + "Crash-safe leftover Convert failed (%s), fallback to " + "full WAL RecoverLogFiles", + cs.ToString().c_str()); + crash_safe_convert = false; + } else if (mmap_rec.wal_number != 0 || mmap_rec.wal_offset != 0) { + recovery_ctx->crash_safe_wal_tail_replay_ = true; + recovery_ctx->crash_safe_wal_number_ = mmap_rec.wal_number; + recovery_ctx->crash_safe_wal_offset_ = mmap_rec.wal_offset; + recovery_ctx->crash_safe_wal_offset_kind_ = + static_cast(pubseq_mmap_->wal_offset_kind); + ROCKS_LOG_INFO(immutable_db_options_.info_log, + "Crash-safe recover Convert leftover, WAL tail from " + "#%" PRIu64 " offset %" PRIu64 " kind %u seq %" PRIu64, + mmap_rec.wal_number, mmap_rec.wal_offset, + pubseq_mmap_->wal_offset_kind, mmap_rec.pubseq); + } + } if (immutable_db_options_.paranoid_checks && s.ok()) { s = CheckConsistency(); } @@ -795,6 +1375,21 @@ Status DBImpl::Recover( } } + if (s.ok() && !read_only && !crash_safe_convert) { + // Retire the old inventory only together with the successful WAL recovery. + // Missing files are allowed here: their absence is what forced the replay. + for (auto* cfd : *versions_->GetColumnFamilySet()) { + const auto& files = cfd->GetMemTableFiles(); + if (files.empty()) continue; + VersionEdit edit; + edit.SetColumnFamily(cfd->GetID()); + for (uint64_t number : files) { + edit.DeleteMemTableFile(number); + } + recovery_ctx->UpdateVersionEdits(cfd, edit); + } + } + if (read_only) { // If we are opening as read-only, we need to update options_file_number_ // to reflect the most recent OPTIONS file. It does not matter for regular @@ -834,6 +1429,14 @@ Status DBImpl::Recover( versions_->options_file_size_ = options_file_size; } } + if (s.ok() && !read_only && + immutable_db_options_.memtable_crash_safe_recover) { + if (mmap_valid && mmap_rec.pubseq > versions_->LastSequence()) { + versions_->SetLastAllocatedSequence(mmap_rec.pubseq); + versions_->SetLastPublishedSequence(mmap_rec.pubseq); + versions_->SetLastSequence(mmap_rec.pubseq); + } + } return s; } @@ -949,9 +1552,38 @@ Status DBImpl::LogAndApplyForRecovery(const RecoveryContext& recovery_ctx) { mutex_.AssertHeld(); assert(versions_->descriptor_log_ == nullptr); const ReadOptions read_options(Env::IOActivity::kDBOpen); + // Recovery retires the old inventory and installs all recovered CFs at once. + // A truncated MANIFEST tail must not install just part of that transition. + uint32_t remaining = 0; + bool tracks_memtable_files = false; + for (const auto& edits : recovery_ctx.edit_lists_) { + remaining += static_cast(edits.size()); + for (const auto* edit : edits) { + tracks_memtable_files |= edit->HasMemTableFileTracking() || + !edit->GetMemTableFileAdditions().empty() || + !edit->GetMemTableFileDeletions().empty(); + } + } + if (tracks_memtable_files && remaining > 1) { + for (const auto& edits : recovery_ctx.edit_lists_) { + for (auto* edit : edits) { + edit->MarkAtomicGroup(--remaining); + } + } + } + TEST_SYNC_POINT("DBImpl::RegisterMemTableFile:BeforeLogAndApply"); Status s = versions_->LogAndApply( recovery_ctx.cfds_, recovery_ctx.mutable_cf_opts_, read_options, recovery_ctx.edit_lists_, &mutex_, directories_.GetDbDir()); + TEST_SYNC_POINT_CALLBACK("DBImpl::RegisterMemTableFile:AfterLogAndApply", &s); + if (s.ok()) { + if (immutable_db_options_.memtable_crash_safe_recover) { + for (auto* cfd : *versions_->GetColumnFamilySet()) { + cfd->PublishRegisteredMemTables(); + } + } + TEST_SYNC_POINT("DBImpl::RegisterMemTableFile:BeforeInstall"); + } if (s.ok() && !(recovery_ctx.files_to_delete_.empty())) { mutex_.Unlock(); for (const auto& stale_sst_file : recovery_ctx.files_to_delete_) { @@ -965,6 +1597,12 @@ Status DBImpl::LogAndApplyForRecovery(const RecoveryContext& recovery_ctx) { } mutex_.Lock(); } + if (s.ok() && pubseq_mmap_ != nullptr && (pubseq_mmap_->generation & 1)) { + // Discard a torn cursor before enabling publication. + pubseq_mmap_->rec = PublishedSeqRecord{}; + terark::as_atomic(pubseq_mmap_->generation) + .store(0, std::memory_order_release); + } return s; } @@ -1131,6 +1769,16 @@ Status DBImpl::RecoverLogFiles(const std::vector& wal_numbers, bool flushed = false; uint64_t corrupted_wal_number = kMaxSequenceNumber; uint64_t min_wal_number = MinLogNumberToKeep(); + const bool crash_safe_wal_tail_replay = + recovery_ctx != nullptr && recovery_ctx->crash_safe_wal_tail_replay_; + const uint64_t crash_safe_wal_number = + crash_safe_wal_tail_replay ? recovery_ctx->crash_safe_wal_number_ : 0; + const uint64_t crash_safe_wal_offset = + crash_safe_wal_tail_replay ? recovery_ctx->crash_safe_wal_offset_ : 0; + const uint8_t crash_safe_wal_offset_kind = + crash_safe_wal_tail_replay ? recovery_ctx->crash_safe_wal_offset_kind_ : 0; + const uint64_t crash_safe_probe_wal_below = + recovery_ctx != nullptr ? recovery_ctx->crash_safe_probe_wal_below_ : 0; if (!allow_2pc()) { // In non-2pc mode, we skip WALs that do not back unflushed data. min_wal_number = @@ -1148,6 +1796,14 @@ Status DBImpl::RecoverLogFiles(const std::vector& wal_numbers, // records after allocating this log number. So we manually // update the file number allocation counter in VersionSet. versions_->MarkFileNumberUsed(wal_number); + if (crash_safe_wal_tail_replay && !allow_2pc() && + wal_number < crash_safe_wal_number) { + ROCKS_LOG_INFO(immutable_db_options_.info_log, + "Skipping InsertInto for log #%" PRIu64 + " before crash-safe WAL cursor #%" PRIu64, + wal_number, crash_safe_wal_number); + continue; + } // Open the log file std::string fname = LogFileName(immutable_db_options_.GetWalDir(), wal_number); @@ -1168,7 +1824,8 @@ Status DBImpl::RecoverLogFiles(const std::vector& wal_numbers, continue; } bool wal_memtable_format = immutable_db_options_.memtable_as_log_index; - if (immutable_db_options_.check_wal_format) { + if (immutable_db_options_.check_wal_format || wal_number < crash_safe_probe_wal_below) { + TEST_SYNC_POINT_CALLBACK("DBImpl::RecoverLogFiles:ProbeWalFormat", &wal_number); if (IOStatus ios = log::Reader::IsMemTableAsLogIndexFile (*fs_, fname, &wal_memtable_format); !ios.ok()) { auto info_log = immutable_db_options_.info_log.get(); @@ -1177,6 +1834,8 @@ Status DBImpl::RecoverLogFiles(const std::vector& wal_numbers, } } + bool prefix_compare = false; + reopen_wal: std::unique_ptr file_reader; { std::unique_ptr file; @@ -1225,6 +1884,44 @@ Status DBImpl::RecoverLogFiles(const std::vector& wal_numbers, fmap->tail_pos = std::make_shared(fmap->size_); } + if (crash_safe_wal_tail_replay && !prefix_compare) { + const uint64_t cursor_wal = crash_safe_wal_number; + bool has_udt = false; + for (auto* cfd : *versions_->GetColumnFamilySet()) { + if (cfd->user_comparator()->timestamp_size() > 0) { + has_udt = true; + break; + } + } + if (allow_2pc()) { + prefix_compare = true; + } else if (wal_number == cursor_wal) { + if (has_udt) { + ROCKS_LOG_WARN(immutable_db_options_.info_log, + "Crash-safe WAL Seek skipped (UDT), " + "compare LastRecordOffset from start of log #%" PRIu64, + wal_number); + prefix_compare = true; + } else { + const uint64_t record_start = PublishedWalRecordStart( + crash_safe_wal_offset, crash_safe_wal_offset_kind); + IOStatus seek_s = reader.SeekToFileOffset(record_start); + TEST_SYNC_POINT_CALLBACK( + "CrashSafeRecover::SeekToFileOffset:InjectStatus", &seek_s); + if (!seek_s.ok()) { + ROCKS_LOG_WARN(immutable_db_options_.info_log, + "Crash-safe SeekToFileOffset(%" PRIu64 + ") failed (%s), compare LastRecordOffset from start", + record_start, seek_s.ToString().c_str()); + prefix_compare = true; + // A failed seek may have consumed bytes. Recreate both readers + // before replaying from the beginning of this WAL. + goto reopen_wal; + } + } + } + } + // Determine if we should tolerate incomplete records at the tail end of the // Read all the records and add to a memtable std::string scratch; @@ -1325,11 +2022,26 @@ Status DBImpl::RecoverLogFiles(const std::vector& wal_numbers, if (fmap) { batch_to_use->SetWAL({fmap, wal_number, reader.LastRecordOffset()}); } + column_family_memtables_->skip_memtable_data_ = false; + if (crash_safe_wal_tail_replay && prefix_compare) { + if (wal_number < crash_safe_wal_number) { + column_family_memtables_->skip_memtable_data_ = true; + } else if (wal_number == crash_safe_wal_number && + reader.LastRecordOffset() < + PublishedWalRecordStart(crash_safe_wal_offset, + crash_safe_wal_offset_kind)) { + column_family_memtables_->skip_memtable_data_ = true; + } + } status = WriteBatchInternal::InsertInto( batch_to_use, column_family_memtables_.get(), &flush_scheduler_, &trim_history_scheduler_, true, wal_number, this, false /* concurrent_memtable_writes */, next_sequence, &has_valid_writes, seq_per_batch_, batch_per_txn_); + column_family_memtables_->skip_memtable_data_ = false; + if (status.IsNotSupported()) { + return status; + } MaybeIgnoreError(&status); if (!status.ok()) { // We are treating this as a failure while reading since we read valid @@ -1849,12 +2561,23 @@ Status DB::Open(const Options& options, const std::string& dbname, DB** dbptr) { Status DB::Open(const DBOptions& db_options, const std::string& dbname, const std::vector& column_families, std::vector* handles, DB** dbptr) { + *dbptr = nullptr; + MaybeOptionsUpdateFrom(const_cast(&db_options), + const_cast*>(&column_families), + dbname); const bool kSeqPerBatch = true; const bool kBatchPerTxn = true; ThreadStatusUtil::SetEnableTracking(db_options.enable_thread_tracking); ThreadStatusUtil::SetThreadOperation(ThreadStatus::OperationType::OP_DBOPEN); + Status s0 = DBImpl::PrepareCrashSafeKindForOpen( + db_options, dbname, column_families, !kSeqPerBatch, kBatchPerTxn); + if (!s0.ok()) { + ThreadStatusUtil::ResetThreadStatus(); + return s0; + } Status s = DBImpl::Open(db_options, dbname, column_families, handles, dbptr, - !kSeqPerBatch, kBatchPerTxn); + !kSeqPerBatch, kBatchPerTxn, + true /* options_already_updated */); ThreadStatusUtil::ResetThreadStatus(); return s; } @@ -1988,10 +2711,14 @@ IOStatus DBImpl::CreateWAL(uint64_t log_file_num, uint64_t recycle_log_number, Status DBImpl::Open(const DBOptions& db_options, const std::string& dbname, const std::vector& column_families, std::vector* handles, DB** dbptr, - const bool seq_per_batch, const bool batch_per_txn) { - MaybeOptionsUpdateFrom(const_cast(&db_options), - const_cast*>(&column_families), - dbname); + const bool seq_per_batch, const bool batch_per_txn, + bool options_already_updated) { + // Kind-prep supplies the old WAL kind after applying the external config. + if (!options_already_updated) { + MaybeOptionsUpdateFrom(const_cast(&db_options), + const_cast*>(&column_families), + dbname); + } *dbptr = nullptr; ROCKSDB_SCOPE_EXIT(MaybeRetainDB(*dbptr, *handles)); @@ -2116,7 +2843,20 @@ Status DBImpl::Open(const DBOptions& db_options, const std::string& dbname, } } } + if (s.ok() && impl->pubseq_mmap_ != nullptr && + impl->pubseq_mmap_->kind_since_wal == 0) { + impl->pubseq_mmap_->kind_since_wal = impl->logfile_number_; + } if (s.ok()) { + if (impl->immutable_db_options_.memtable_crash_safe_recover) { + for (auto* cfd : *impl->versions_->GetColumnFamilySet()) { + VersionEdit edit; + edit.SetColumnFamily(cfd->GetID()); + cfd->PrepareNewMemtableInBackground(*cfd->GetLatestMutableCFOptions()); + cfd->AddMemTableFileEdits(&edit); + recovery_ctx.UpdateVersionEdits(cfd, edit); + } + } s = impl->LogAndApplyForRecovery(recovery_ctx); } diff --git a/db/db_impl/db_impl_secondary.cc b/db/db_impl/db_impl_secondary.cc index 8f4d2e6415..0d2af8b4ab 100644 --- a/db/db_impl/db_impl_secondary.cc +++ b/db/db_impl/db_impl_secondary.cc @@ -36,6 +36,12 @@ Status DBImplSecondary::Recover( bool /*error_if_data_exists_in_wals*/, uint64_t*, RecoveryContext* /*recovery_ctx*/) { mutex_.AssertHeld(); + for (const auto& cf : column_families) { + if (cf.options.memtable_factory->SupportCrashSafe()) { + return Status::InvalidArgument( + "FileMmap memtable is not supported in secondary mode", cf.name); + } + } JobContext job_context(0); Status s; diff --git a/db/db_impl/db_impl_write.cc b/db/db_impl/db_impl_write.cc index 4ad662e1f3..1675d30969 100644 --- a/db/db_impl/db_impl_write.cc +++ b/db/db_impl/db_impl_write.cc @@ -190,7 +190,8 @@ Status DBImpl::WriteImpl(const WriteOptions& write_options, write_options.protection_bytes_per_key == 0 || write_options.protection_bytes_per_key == my_batch->GetProtectionBytesPerKey()); - if (immutable_db_options_.memtable_as_log_index) { + if (immutable_db_options_.memtable_as_log_index || + immutable_db_options_.memtable_crash_safe_recover) { const_cast(write_options.disableWAL) = false; } @@ -368,8 +369,16 @@ Status DBImpl::WriteImpl(const WriteOptions& write_options, } } versions_->SetLastSequence(last_sequence); + MaybePersistPublishedSequence(last_sequence, *w.write_group); + if (pubseq_mmap_ != nullptr && two_write_queues_ && + !immutable_db_options_.unordered_write) { + AccountPendingMemtableWrites(1); + } MemTableInsertStatusCheck(w.status); write_thread_.ExitAsBatchGroupFollower(&w); + } else if (pubseq_mmap_ != nullptr && two_write_queues_ && + !immutable_db_options_.unordered_write) { + AccountPendingMemtableWrites(1); } assert(w.state == WriteThread::STATE_COMPLETED); // STATE_COMPLETED conditional below handles exit @@ -426,6 +435,7 @@ Status DBImpl::WriteImpl(const WriteOptions& write_options, write_thread_.EnterAsBatchGroupLeader(&w, &write_group); IOStatus io_s; + bool account_two_q_pending = false; Status pre_release_cb_status; if (status.ok()) { // Rules for when we can update the memtable concurrently @@ -529,6 +539,9 @@ Status DBImpl::WriteImpl(const WriteOptions& write_options, // wal_write_mutex_ to ensure ordered events in WAL io_s = ConcurrentWriteToWAL(write_group, log_used, &last_sequence, seq_inc); + account_two_q_pending = + io_s.ok() && pubseq_mmap_ != nullptr && + !immutable_db_options_.unordered_write; } else { // Otherwise we inc seq number for memtable writes last_sequence = versions_->FetchAddLastAllocatedSequence(seq_inc); @@ -671,9 +684,15 @@ Status DBImpl::WriteImpl(const WriteOptions& write_options, // Note: if we are to resume after non-OK statuses we need to revisit how // we reacts to non-OK statuses here. versions_->SetLastSequence(last_sequence); + MaybePersistPublishedSequence(last_sequence, write_group); + } + if (account_two_q_pending) { + AccountPendingMemtableWrites(in_parallel_group ? 1 : write_group.size); } MemTableInsertStatusCheck(w.status); write_thread_.ExitAsBatchGroupLeader(write_group, status); + } else if (in_parallel_group && account_two_q_pending) { + AccountPendingMemtableWrites(1); } if (status.ok()) { @@ -690,6 +709,7 @@ Status DBImpl::PipelinedWriteImpl(const WriteOptions& write_options, StopWatch write_sw(immutable_db_options_.clock, stats_, DB_WRITE); WriteContext write_context; + WriteThread::WriteGroup memtable_write_group; WriteThread::Writer w(write_options, my_batch, callback, log_ref, disable_memtable, /*_batch_cnt=*/0, @@ -806,15 +826,21 @@ Status DBImpl::PipelinedWriteImpl(const WriteOptions& write_options, const ReadOptions read_options; w.status = ApplyWALToManifest(read_options, &synced_wals); } + if (w.status.ok() && pubseq_mmap_ != nullptr) { + size_t memtable_write_cnt = 0; + for (auto* writer : wal_write_group) { + if (writer->ShouldWriteToMemtable()) { + memtable_write_cnt++; + } + } + StagePublishedWal(current_sequence + total_count - 1, + wal_write_group.wal_number, + wal_write_group.wal_offset); + pending_memtable_writes_ += memtable_write_cnt; + } write_thread_.ExitAsBatchGroupLeader(wal_write_group, w.status); } - // NOTE: the memtable_write_group is declared before the following - // `if` statement because its lifetime needs to be longer - // that the inner context of the `if` as a reference to it - // may be used further below within the outer _write_thread - WriteThread::WriteGroup memtable_write_group; - if (w.state == WriteThread::STATE_MEMTABLE_WRITER_LEADER) { PERF_TIMER_WITH_HISTOGRAM(write_memtable_time, MEMTAB_WRITE_KV_NANOS, stats_); assert(w.ShouldWriteToMemtable()); @@ -829,6 +855,9 @@ Status DBImpl::PipelinedWriteImpl(const WriteOptions& write_options, write_options.ignore_missing_column_families, 0 /*log_number*/, this, false /*concurrent_memtable_writes*/, seq_per_batch_, batch_per_txn_); versions_->SetLastSequence(memtable_write_group.last_sequence); + if (pubseq_mmap_ != nullptr) { + AccountPendingMemtableWrites(memtable_write_group.size); + } write_thread_.ExitAsMemTableWriter(&w, memtable_write_group); } } else { @@ -847,6 +876,9 @@ Status DBImpl::PipelinedWriteImpl(const WriteOptions& write_options, 0 /*log_number*/, this, true /*concurrent_memtable_writes*/, false /*seq_per_batch*/, 0 /*batch_cnt*/, true /*batch_per_txn*/, write_options.memtable_insert_hint_per_batch); + if (pubseq_mmap_ != nullptr) { + AccountPendingMemtableWrites(1); + } if (write_thread_.CompleteParallelMemTableWriter(&w)) { MemTableInsertStatusCheck(w.status); versions_->SetLastSequence(w.write_group->last_sequence); @@ -892,15 +924,7 @@ Status DBImpl::UnorderedWriteMemtable(const WriteOptions& write_options, } } - size_t pending_cnt = pending_memtable_writes_.fetch_sub(1) - 1; - if (pending_cnt == 0) { - // switch_cv_ waits until pending_memtable_writes_ = 0. Locking its mutex - // before notify ensures that cv is in waiting state when it is notified - // thus not missing the update to pending_memtable_writes_ even though it is - // not modified under the mutex. - std::lock_guard lck(switch_mutex_); - switch_cv_.notify_all(); - } + AccountPendingMemtableWrites(1); WriteStatusCheck(w.status); if (!w.FinalStatus().ok()) { @@ -1111,8 +1135,13 @@ Status DBImpl::WriteImplWALOnly( // Currently we only use kDoPublishLastSeq in unordered_write assert(immutable_db_options_.unordered_write); } - if (immutable_db_options_.unordered_write && status.ok()) { + if (immutable_db_options_.unordered_write && status.ok() && + pubseq_mmap_ == nullptr) { pending_memtable_writes_ += memtable_write_cnt; + } else if (immutable_db_options_.unordered_write && !status.ok() && + pubseq_mmap_ != nullptr) { + // ConcurrentWriteToWAL counted this group, but it will skip memtable writes. + AccountPendingMemtableWrites(memtable_write_cnt); } write_thread->ExitAsBatchGroupLeader(write_group, status); if (status.ok()) { @@ -1488,6 +1517,12 @@ IOStatus DBImpl::DoWriteWAL(const WriteBatch& merged_batch, total_log_size_.fetch_add(log_size_sum, std::memory_order_relaxed); log_file_number_size.AddSize(log_size_sum); log_empty_ = false; + if (io_s.ok() && pubseq_mmap_ != nullptr) { + write_group.wal_number = logfile_number_; + write_group.wal_offset = immutable_db_options_.memtable_as_log_index + ? log_writer->get_log_offset() + : log_writer->file()->GetFileSize(); + } return io_s; } @@ -1626,6 +1661,37 @@ IOStatus DBImpl::ConcurrentWriteToWAL( cached_recoverable_state_ = *to_be_cached_state; cached_recoverable_state_empty_ = false; } + if (io_s.ok() && pubseq_mmap_ != nullptr) { + const SequenceNumber pub_seq = *last_sequence + seq_inc; + if (write_group.leader->disable_memtable) { + StagePublishedWal(pub_seq, write_group.wal_number, + write_group.wal_offset); + PersistStagedPublishedWal(); + } else if (immutable_db_options_.unordered_write) { + StagePublishedWal(pub_seq, write_group.wal_number, + write_group.wal_offset); + } + } + if (io_s.ok() && pubseq_mmap_ != nullptr && + immutable_db_options_.unordered_write) { + size_t n = 0; + for (auto* writer : write_group) { + if (!writer->CallbackFailed() && !writer->disable_memtable) { + n++; + } + } + pending_memtable_writes_ += n; + } else if (io_s.ok() && pubseq_mmap_ != nullptr && two_write_queues_ && + !immutable_db_options_.unordered_write && + !write_group.leader->disable_memtable) { + size_t n = 0; + for (auto* writer : write_group) { + if (!writer->disable_memtable) { + n++; + } + } + pending_memtable_writes_ += n; + } log_write_mutex_.Unlock(); if (io_s.ok()) { @@ -2256,6 +2322,22 @@ Status DBImpl::SwitchMemtable(ColumnFamilyData* cfd, WriteContext* context) { return s; } + auto pending_memtable = pending_outputs_.end(); + if (cfd->ioptions()->memtable_factory->SupportCrashSafe()) { + auto* cached = cfd->PeekPrecreatedMemtable(); + uint64_t number = cached ? cached->GetFileNumber() + : versions_->current_next_file_number(); + // A background-created file may enter the cache while the DB mutex is free. + if (!pending_outputs_.empty()) { + terark::minimize(number, pending_outputs_.front()); + } + pending_memtable = pending_outputs_.insert(pending_outputs_.begin(), number); + } + ROCKSDB_SCOPE_EXIT(if (pending_memtable != pending_outputs_.end()) { + mutex_.AssertHeld(); + pending_outputs_.erase(pending_memtable); + }); + // Attempt to switch to a new memtable and trigger flush of old. // Do this without holding the dbmutex lock. assert(versions_->prev_log_number() == 0); @@ -2302,6 +2384,7 @@ Status DBImpl::SwitchMemtable(ColumnFamilyData* cfd, WriteContext* context) { new_mem = cfd->ConstructNewMemtable(mutable_cf_options, seq); context->superversion_context.NewSuperVersion(); } + TEST_SYNC_POINT("DBImpl::SwitchMemtable:BeforeInstallMemTable"); ROCKS_LOG_INFO(immutable_db_options_.info_log, "[%s] New memtable created with log file: #%" PRIu64 ". Immutable memtables: %d.\n", @@ -2319,6 +2402,39 @@ Status DBImpl::SwitchMemtable(ColumnFamilyData* cfd, WriteContext* context) { assert(log_recycle_files_.front() == recycle_log_number); log_recycle_files_.pop_front(); } + if (s.ok() && immutable_db_options_.memtable_crash_safe_recover && + !new_mem->IsFileRegistered()) { + TEST_SYNC_POINT_CALLBACK("DBImpl::SwitchMemtable:MemTableCacheMiss", new_mem); + s = error_handler_.GetBGError(); + // A flush may have registered this file after it left the cache. + if (s.ok() && !cfd->GetMemTableFiles().count(new_mem->GetFileNumber())) { + VersionEdit edit; + edit.SetColumnFamily(cfd->GetID()); + edit.AddMemTableFile(new_mem->GetFileNumber()); + edit.SetMemTableFileTracking(); + TEST_SYNC_POINT("DBImpl::RegisterMemTableFile:BeforeLogAndApply"); + s = versions_->LogAndApply(cfd, mutable_cf_options, ReadOptions(), &edit, + &mutex_, directories_.GetDbDir()); + TEST_SYNC_POINT_CALLBACK("DBImpl::RegisterMemTableFile:AfterLogAndApply", &s); + if (!s.ok()) { + error_handler_.SetBGError(s, BackgroundErrorReason::kManifestWrite) + .PermitUncheckedError(); + } else { + s = error_handler_.GetBGError(); + } + } + if (s.ok() && shutting_down_.load(std::memory_order_acquire)) { + s = Status::ShutdownInProgress(); + } + if (!s.ok()) { + delete new_mem; + delete new_log; + context->superversion_context.new_superversion.reset(); + return s; + } + new_mem->MarkFileRegistered(); + TEST_SYNC_POINT("DBImpl::RegisterMemTableFile:BeforeInstall"); + } if (s.ok() && creating_new_log) { InstrumentedMutexLock l(&log_write_mutex_); assert(new_log != nullptr); diff --git a/db/db_iter.cc b/db/db_iter.cc index 2df4c83487..8a3ba99836 100644 --- a/db/db_iter.cc +++ b/db/db_iter.cc @@ -307,7 +307,6 @@ Slice DBIter::NextWithKey() { return Slice(nullptr, 0); } -ROCKSDB_FLATTEN Slice DBIter::PrevWithKey() { return IterPrevWithKeyImpl(this); } bool DBIter::SetBlobValueIfNeeded(const Slice& user_key, @@ -1166,6 +1165,7 @@ void DBIter::Prev() { } } +terark_no_inline bool DBIter::ReverseToForward() { assert(iter_.status().ok()); @@ -1208,6 +1208,7 @@ bool DBIter::ReverseToForward() { } // Move iter_ to the key before saved_key_. +terark_no_inline bool DBIter::ReverseToBackward() { assert(iter_.status().ok()); @@ -1939,7 +1940,6 @@ void DBIter::SetSavedKeyToSeekForPrevTarget(const Slice& target) { } } -ROCKSDB_FLATTEN void DBIter::Seek(const Slice& target) { PERF_COUNTER_ADD(iter_seek_count, 1); PERF_CPU_TIMER_GUARD(iter_seek_cpu_nanos, clock_); diff --git a/db/db_memtable_convert_test.cc b/db/db_memtable_convert_test.cc new file mode 100644 index 0000000000..5b31be6220 --- /dev/null +++ b/db/db_memtable_convert_test.cc @@ -0,0 +1,428 @@ +// Copyright (c) 2026-present, Topling Inc. +// Flush/Close conversion, independently of crash-safe recovery. + +#include +#include +#include +#include +#include + +#include +#include + +#include "db/db_impl/db_impl.h" +#include "db/db_test_util.h" +#include "db/memtable.h" +#include "db/version_set.h" +#include "env/env_chroot.h" +#include "file/filename.h" +#include "memory/arena.h" +#include "port/stack_trace.h" +#include "test_util/sync_point.h" +#include "table/table_builder.h" + +namespace ROCKSDB_NAMESPACE { + +class DBMemtableConvertTest + : public DBTestBase, + public ::testing::WithParamInterface> { + public: + DBMemtableConvertTest() + : DBTestBase("db_memtable_convert_test", /*env_do_fsync=*/false) {} + ~DBMemtableConvertTest() override { + SyncPoint::GetInstance()->DisableProcessing(); + SyncPoint::GetInstance()->ClearAllCallBacks(); + } + + protected: + Options ConvertOptions() { + Options options = CurrentOptions(); + options.disable_auto_compactions = true; + options.avoid_flush_during_recovery = true; + options.write_buffer_size = 64 << 20; + options.max_write_buffer_number = 8; + options.min_write_buffer_number_to_merge = 4; + options.atomic_flush = std::get<2>(GetParam()); + const bool osl = std::get<0>(GetParam()); + const json params = {{"mem_cap", 16777216}, + {"convert_to_sst", std::get<1>(GetParam())}}; + auto& repo = repo_; + options.memtable_factory = PluginFactorySP::AcquirePlugin( + osl ? "OffsetSkipList" : "CSPPMemTab", params, repo); + repo.Put("default", options.table_factory); + repo.Put("converted", PluginFactorySP::AcquirePlugin( + osl ? "OffsetSkipListTable" : "CSPPMemTabTable", params, repo)); + options.table_factory = PluginFactorySP::AcquirePlugin( + "Dispatch", {{"default", "$default"}}, repo); + DispatcherTableBackPatch(options.table_factory.get(), repo); + return options; + } + + void ObserveConversion(bool fail = false) { + const auto caller = std::this_thread::get_id(); + SyncPoint::GetInstance()->SetCallBack( + "FlushJob::ConvertToSST:Status", [this, caller, fail](void* arg) { + EXPECT_NE(std::this_thread::get_id(), caller); + EXPECT_TRUE(static_cast(arg)->ok()); + ++converts_; + if (fail) { + *static_cast(arg) = Status::IOError("injected convert"); + } + }); + SyncPoint::GetInstance()->SetCallBack( + "FlushJob::WriteLevel0Table:s", [this](void*) { ++builds_; }); + SyncPoint::GetInstance()->EnableProcessing(); + } + + void CheckClose(bool wait_for_compact) { + Options options = ConvertOptions(); + DestroyAndReopen(options); + ASSERT_OK(Put("a", "1")); + ASSERT_OK(dbfull()->TEST_SwitchMemtable()); + ASSERT_OK(Put("b", "2")); + ASSERT_OK(dbfull()->TEST_SwitchMemtable()); + ASSERT_OK(Put("c", "3")); + ObserveConversion(); + if (wait_for_compact) { + WaitForCompactOptions wait; + wait.close_db = true; + ASSERT_OK(db_->WaitForCompact(wait)); + } else { + ASSERT_OK(db_->Close()); + } + Close(); + ASSERT_EQ(converts_.load(), 3); + ASSERT_EQ(builds_.load(), 0); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("a"), "1"); + ASSERT_EQ(Get("b"), "2"); + ASSERT_EQ(Get("c"), "3"); + Close(); + } + + std::atomic converts_{0}; + std::atomic builds_{0}; + SidePluginRepo repo_; +}; + +TEST_P(DBMemtableConvertTest, ManualFlushConverts) { + Options options = ConvertOptions(); + DestroyAndReopen(options); + ASSERT_OK(Put("k", "v")); + ObserveConversion(); + ASSERT_OK(db_->Flush(FlushOptions())); + ASSERT_EQ(converts_.load(), 1); + ASSERT_EQ(builds_.load(), 0); + ASSERT_EQ(Get("k"), "v"); + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("k"), "v"); + Close(); +} + +TEST_P(DBMemtableConvertTest, ConversionSanitizesMergeThreshold) { + // Avoid the existing atomic-flush sanitizer masking the conversion rule. + if (std::get<2>(GetParam())) return; + Options options = ConvertOptions(); + ASSERT_GT(options.min_write_buffer_number_to_merge, 1); + DestroyAndReopen(options); + ASSERT_EQ(db_->GetOptions().min_write_buffer_number_to_merge, 1); + ColumnFamilyHandle* cf = nullptr; + ASSERT_OK(db_->CreateColumnFamily(options, "converted", &cf)); + ASSERT_EQ(db_->GetOptions(cf).min_write_buffer_number_to_merge, 1); + ASSERT_OK(db_->DestroyColumnFamilyHandle(cf)); + Close(); +} + +TEST_P(DBMemtableConvertTest, NonConversionKeepsMergeThreshold) { + // One run per plugin suffices for the disabled conversion and default + // factory controls; the converting modes are covered independently above. + if (std::get<2>(GetParam()) || + std::string(std::get<1>(GetParam())) != "kDumpMem") return; + for (bool skip_list : {false, true}) { + SCOPED_TRACE(skip_list); + Options options = CurrentOptions(); + options.max_write_buffer_number = 8; + options.min_write_buffer_number_to_merge = 2; + options.atomic_flush = false; + if (skip_list) { + options.memtable_factory = std::make_shared(); + } else { + options.memtable_factory = PluginFactorySP::AcquirePlugin( + std::get<0>(GetParam()) ? "OffsetSkipList" : "CSPPMemTab", + {{"mem_cap", 16777216}, {"convert_to_sst", "kDontConvert"}}, repo_); + } + DestroyAndReopen(options); + ASSERT_EQ(db_->GetOptions().min_write_buffer_number_to_merge, 2); + ColumnFamilyHandle* cf = nullptr; + ASSERT_OK(db_->CreateColumnFamily(options, "plain", &cf)); + ASSERT_EQ(db_->GetOptions(cf).min_write_buffer_number_to_merge, 2); + ASSERT_OK(db_->DestroyColumnFamilyHandle(cf)); + Close(); + } +} + +TEST_P(DBMemtableConvertTest, FileMmapConversionKeepsFileNumberAndPath) { + if (std::string(std::get<1>(GetParam())) != "kFileMmap") return; + Options options = ConvertOptions(); + DestroyAndReopen(options); + ASSERT_OK(Put("same-file", "value")); + for (auto* cfd : *dbfull()->GetVersionSet()->GetColumnFamilySet()) { + ASSERT_TRUE(cfd->GetMemTableFiles().empty()); + } + const uint64_t number = dbfull()->GetVersionSet()->GetColumnFamilySet() + ->GetDefault()->mem()->GetFileNumber(); + const std::string path = MakeTableFileName(dbname_, number); + ASSERT_OK(env_->FileExists(path)); + ObserveConversion(); + ASSERT_OK(db_->Flush(FlushOptions())); + std::vector files; + db_->GetLiveFilesMetaData(&files); + ASSERT_EQ(files.size(), 1U); + ASSERT_EQ(files[0].file_number, number); + ASSERT_OK(env_->FileExists(path)); + for (auto* cfd : *dbfull()->GetVersionSet()->GetColumnFamilySet()) { + ASSERT_TRUE(cfd->GetMemTableFiles().empty()); + } + Close(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("same-file"), "value"); +} + +#if !defined(OS_WIN) +TEST_P(DBMemtableConvertTest, FileMmapExclusiveCreationAndChrootConversion) { + if (std::string(std::get<1>(GetParam())) != "kFileMmap") return; + Options options = ConvertOptions(); + DestroyAndReopen(options); + Close(); + const std::string logical_path = MakeTableFileName("", 900000); + const std::string physical_path = dbname_ + logical_path; + auto factory = PluginFactorySP::AcquirePlugin( + std::get<0>(GetParam()) ? "OffsetSkipList" : "CSPPMemTab", + {{"mem_cap", 16777216}, {"convert_to_sst", "kFileMmap"}, + {"chroot_dir", dbname_}}, repo_); + InternalKeyComparator icmp(options.comparator); + MemTable::KeyComparator cmp(icmp); + MutableCFOptions moptions(options); + moptions.write_buffer_size = 1 << 20; + Arena arena; + if (std::get<0>(GetParam())) { + const std::string sentinel = "existing SST must remain intact"; + ASSERT_OK(WriteStringToFile(env_, sentinel, physical_path)); + ASSERT_THROW({ + std::unique_ptr collision(factory->CreateMemTableRep( + logical_path, moptions, cmp, &arena, nullptr, nullptr, 0)); + }, std::runtime_error); + std::string after; + ASSERT_OK(ReadFileToString(env_, physical_path, &after)); + ASSERT_EQ(after, sentinel); + ASSERT_OK(env_->DeleteFile(physical_path)); + } + + std::unique_ptr chroot_env(NewChrootEnv(env_, dbname_)); + options.env = chroot_env.get(); + options.cf_paths = {{"", 0}}; + ImmutableOptions ioptions(options); + IntTblPropCollectorFactories collectors; + const std::string cf_name = "default"; + TableBuilderOptions tbo(ioptions, moptions, icmp, &collectors, + options.compression, options.compression_opts, 0, + cf_name, 0); + std::unique_ptr rep(factory->CreateMemTableRep( + logical_path, moptions, cmp, &arena, nullptr, nullptr, 0)); + ASSERT_TRUE(rep->SupportCrashSafe()); + rep->InitSetMemTableAsLogIndex(false); + ASSERT_TRUE(rep->InsertKeyValue(PackSequenceAndType(1, kTypeValue), "k", "v")); + rep->MarkReadOnly(); + FileMetaData meta; + meta.fd = FileDescriptor(900001, 0, 0); + meta.num_entries = 1; + ASSERT_TRUE(rep->ConvertToSST(&meta, tbo).IsInvalidArgument()); + rep.reset(); + ASSERT_OK(env_->FileExists(physical_path)); + meta.fd = FileDescriptor(900000, 0, 0); + meta.fd.largest_seqno = kMaxSequenceNumber; + ASSERT_OK(factory->RecoverCrashSafeMemTableToSST(logical_path, &meta, tbo)); + ASSERT_GT(meta.fd.GetFileSize(), 0U); + ASSERT_OK(env_->FileExists(physical_path)); + ASSERT_OK(env_->DeleteFile(physical_path)); + + // Retry conversion on the same live rep after a completed footer exists. + rep.reset(factory->CreateMemTableRep( + logical_path, moptions, cmp, &arena, nullptr, nullptr, 0)); + rep->InitSetMemTableAsLogIndex(false); + ASSERT_TRUE(rep->InsertKeyValue(PackSequenceAndType(1, kTypeValue), "k", "v")); + rep->MarkReadOnly(); + meta.fd = FileDescriptor(900000, 0, 0); + ASSERT_OK(rep->ConvertToSST(&meta, tbo)); + const uint64_t first_size = meta.fd.GetFileSize(); + ASSERT_GT(first_size, 0U); + ASSERT_OK(rep->ConvertToSST(&meta, tbo)); + ASSERT_EQ(meta.fd.GetFileSize(), first_size); + uint64_t actual_size = 0; + ASSERT_OK(env_->GetFileSize(physical_path, &actual_size)); + ASSERT_EQ(actual_size, first_size); + std::string value; + rep->GetPIK(ReadOptions(), ParsedInternalKey("k", 1, kTypeValue), &value, + [](void* arg, const MemTableRep::KeyValuePair& kv) { + *static_cast(arg) = kv.value.ToString(); + return false; + }); + ASSERT_EQ(value, "v"); + rep.reset(); + meta.fd = FileDescriptor(900000, 0, 0); + meta.fd.largest_seqno = kMaxSequenceNumber; + ASSERT_OK(factory->RecoverCrashSafeMemTableToSST(logical_path, &meta, tbo)); + ASSERT_OK(env_->DeleteFile(physical_path)); +} +#endif + +TEST_P(DBMemtableConvertTest, FileMmapCloseKeepsEveryFileNumber) { + if (std::string(std::get<1>(GetParam())) != "kFileMmap") return; + Options options = ConvertOptions(); + DestroyAndReopen(options); + ASSERT_OK(Put("head", "1")); + auto* cfd = dbfull()->GetVersionSet()->GetColumnFamilySet()->GetDefault(); + const uint64_t head_number = cfd->mem()->GetFileNumber(); + ASSERT_OK(dbfull()->TEST_SwitchMemtable()); + ASSERT_OK(Put("tail", "2")); + const uint64_t tail_number = cfd->mem()->GetFileNumber(); + const std::set numbers{head_number, tail_number}; + ASSERT_EQ(numbers.size(), 2U); + for (auto* cfd : *dbfull()->GetVersionSet()->GetColumnFamilySet()) { + ASSERT_TRUE(cfd->GetMemTableFiles().empty()); + } + ObserveConversion(); + ASSERT_OK(db_->Close()); + Close(); + ASSERT_EQ(converts_.load(), 2); + for (uint64_t number : numbers) { + ASSERT_OK(env_->FileExists(MakeTableFileName(dbname_, number))); + } + ASSERT_OK(TryReopen(options)); + std::vector files; + db_->GetLiveFilesMetaData(&files); + ASSERT_EQ(files.size(), 2U); + for (const auto& file : files) { + ASSERT_EQ(numbers.count(file.file_number), 1U); + } + ASSERT_EQ(Get("head"), "1"); + ASSERT_EQ(Get("tail"), "2"); +} + +TEST_P(DBMemtableConvertTest, CloseConvertsAllMemtables) { + CheckClose(false); +} + +TEST_P(DBMemtableConvertTest, WaitForCompactConvertsBeforeClose) { + CheckClose(true); +} + +TEST_P(DBMemtableConvertTest, AvoidCloseFlush) { + Options options = ConvertOptions(); + options.avoid_flush_during_shutdown = true; + DestroyAndReopen(options); + ASSERT_OK(Put("k", "v")); + ObserveConversion(); + ASSERT_OK(db_->Close()); + Close(); + ASSERT_EQ(converts_.load(), 0); + ASSERT_EQ(builds_.load(), 0); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("k"), "v"); + Close(); +} + +TEST_P(DBMemtableConvertTest, MixedColumnFamilies) { + Options options = ConvertOptions(); + Options plain = CurrentOptions(); + const std::vector cf_options = {options, plain}; + DestroyAndReopen(options); + CreateColumnFamilies({"plain"}, plain); + ASSERT_OK(TryReopenWithColumnFamilies({"default", "plain"}, cf_options)); + ASSERT_OK(Put(0, "convert", "v1")); + ASSERT_OK(Put(1, "plain", "v2")); + ObserveConversion(); + Close(); + ASSERT_EQ(converts_.load(), 0); + ASSERT_EQ(builds_.load(), 0); + ASSERT_OK(TryReopenWithColumnFamilies({"default", "plain"}, cf_options)); + ASSERT_EQ(Get(0, "convert"), "v1"); + ASSERT_EQ(Get(1, "plain"), "v2"); + Close(); +} + +TEST_P(DBMemtableConvertTest, UnpersistedDataStillFlushes) { + Options options = ConvertOptions(); + options.memtable_factory = CurrentOptions().memtable_factory; + DestroyAndReopen(options); + WriteOptions write; + write.disableWAL = true; + ASSERT_OK(db_->Put(write, "k", "v")); + ObserveConversion(); + ASSERT_OK(db_->Close()); + Close(); + ASSERT_EQ(converts_.load(), 0); + ASSERT_EQ(builds_.load(), 1); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("k"), "v"); + Close(); +} + +TEST_P(DBMemtableConvertTest, CloseConvertFailureDoesNotBuildTable) { + Options options = ConvertOptions(); + DestroyAndReopen(options); + ASSERT_OK(Put("k", "v")); + ObserveConversion(true); + Close(); + ASSERT_EQ(converts_.load(), 1); + ASSERT_EQ(builds_.load(), 0); + SyncPoint::GetInstance()->DisableProcessing(); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("k"), "v"); + Close(); +} + +TEST_P(DBMemtableConvertTest, InFlightFlushThenClose) { + Options options = ConvertOptions(); + options.max_background_flushes = 1; + DestroyAndReopen(options); + ASSERT_OK(Put("head", "1")); + ASSERT_OK(dbfull()->TEST_SwitchMemtable()); + ASSERT_OK(Put("tail", "2")); + test::SleepingBackgroundTask sleeping; + env_->Schedule(&test::SleepingBackgroundTask::DoSleepTask, &sleeping, + Env::Priority::HIGH); + sleeping.WaitUntilSleeping(); + ObserveConversion(); + FlushOptions flush; + flush.wait = false; + ASSERT_OK(db_->Flush(flush)); + SyncPoint::GetInstance()->SetCallBack( + options.atomic_flush ? "DBImpl::AtomicFlushMemTables:AfterScheduleFlush" + : "DBImpl::FlushMemTable:AfterScheduleFlush", + [&](void*) { sleeping.WakeUp(); }); + std::thread closer([&] { Close(); }); + sleeping.WaitUntilDone(); + closer.join(); + ASSERT_EQ(converts_.load(), 2); + ASSERT_EQ(builds_.load(), 0); + ASSERT_OK(TryReopen(options)); + ASSERT_EQ(Get("head"), "1"); + ASSERT_EQ(Get("tail"), "2"); + Close(); +} + +INSTANTIATE_TEST_CASE_P( + Formats, DBMemtableConvertTest, + ::testing::Combine(::testing::Bool(), + ::testing::Values("kDumpMem", "kFileMmap"), + ::testing::Bool())); + +} // namespace ROCKSDB_NAMESPACE + +int main(int argc, char** argv) { + ROCKSDB_NAMESPACE::port::InstallStackTraceHandler(); + ::testing::InitGoogleTest(&argc, argv); + return RUN_ALL_TESTS(); +} diff --git a/db/db_memtable_test.cc b/db/db_memtable_test.cc index bd17b7c047..833c8156a9 100644 --- a/db/db_memtable_test.cc +++ b/db/db_memtable_test.cc @@ -361,6 +361,30 @@ TEST_F(DBMemTableTest, InsertWithHint) { ASSERT_EQ("vvv", Get("NotInPrefixDomain")); } +TEST_F(DBMemTableTest, ShouldFlushNowWithoutRangeDel) { + Options options; + options.memtable_as_log_index = false; + options.write_buffer_size = 64 * 1024; + InternalKeyComparator cmp(BytewiseComparator()); + auto factory = std::make_shared(); + options.memtable_factory = factory; + ImmutableOptions ioptions(options); + WriteBufferManager wb(options.db_write_buffer_size); + MemTable* mem = new MemTable(cmp, ioptions, MutableCFOptions(options), &wb, + kMaxSequenceNumber, 0 /* column_family_id */); + std::string value(128, 'v'); + int i = 0; + for (; i < 2000; ++i) { + ASSERT_OK(mem->Add(i + 1, kTypeValue, "k" + std::to_string(i), value, + nullptr /* kv_prot_info */)); + if (mem->ShouldFlushNow()) { + break; + } + } + ASSERT_LT(i, 2000); + delete mem; +} + TEST_F(DBMemTableTest, ColumnFamilyId) { // Verifies MemTableRepFactory is told the right column family id. Options options; diff --git a/db/db_options_test.cc b/db/db_options_test.cc index 8ccf940ee0..b217e26cee 100644 --- a/db/db_options_test.cc +++ b/db/db_options_test.cc @@ -28,6 +28,10 @@ namespace ROCKSDB_NAMESPACE { +#ifdef HAS_TOPLING_CSPP_MEMTABLE +std::shared_ptr EasyNewMemTableRep(Slice cls, Slice js); +#endif + class DBOptionsTest : public DBTestBase { public: DBOptionsTest() : DBTestBase("db_options_test", /*env_do_fsync=*/true) {} @@ -80,6 +84,130 @@ class DBOptionsTest : public DBTestBase { } }; +TEST_F(DBOptionsTest, SanitizeManualWalFlushAndAtomicFlush) { + DBOptions src; + src.memtable_crash_safe_recover = true; + src.manual_wal_flush = true; + src.recycle_log_file_num = 2; + const DBOptions out = SanitizeOptions(dbname_, src); + ASSERT_FALSE(out.manual_wal_flush); + ASSERT_EQ(out.recycle_log_file_num, 0U); + ASSERT_FALSE(out.atomic_flush); + src.atomic_flush = true; + const DBOptions keep = SanitizeOptions(dbname_, src); + ASSERT_TRUE(keep.atomic_flush); +} + +TEST_F(DBOptionsTest, CrashSafeOrLogIndexDisablesWalCompression) { + if (!StreamingCompressionTypeSupported(kZSTD)) { + return; + } + for (bool crash_safe : {false, true}) { + for (bool log_index : {false, true}) { + SCOPED_TRACE(crash_safe); + SCOPED_TRACE(log_index); + DBOptions src; + src.memtable_crash_safe_recover = crash_safe; + src.memtable_as_log_index = log_index; + src.wal_compression = kZSTD; + const DBOptions out = SanitizeOptions(dbname_, src); + ASSERT_EQ(out.wal_compression, + crash_safe || log_index ? kNoCompression : kZSTD); + } + } +} + +TEST_F(DBOptionsTest, ReadOnlyDisablesCrashSafeAndWarns) { + for (bool use_logger : {false, true}) { + for (bool crash_safe : {false, true}) { + SCOPED_TRACE(use_logger); + SCOPED_TRACE(crash_safe); + DBOptions src; + src.memtable_crash_safe_recover = crash_safe; + src.manual_wal_flush = true; + src.recycle_log_file_num = 2; + src.wal_recovery_mode = WALRecoveryMode::kSkipAnyCorruptedRecords; + if (StreamingCompressionTypeSupported(kZSTD)) { + src.wal_compression = kZSTD; + } + const std::string log_path = dbname_ + "/readonly-warning.log"; + if (use_logger) { + ASSERT_OK(env_->NewLogger(log_path, &src.info_log)); + src.info_log->SetInfoLogLevel(InfoLogLevel::WARN_LEVEL); + } + testing::internal::CaptureStderr(); + const DBOptions out = SanitizeOptions(dbname_, src, /*read_only=*/true); + const std::string stderr_output = testing::internal::GetCapturedStderr(); + ASSERT_FALSE(out.memtable_crash_safe_recover); + ASSERT_EQ(out.info_log, src.info_log); + ASSERT_TRUE(out.manual_wal_flush); + ASSERT_EQ(out.recycle_log_file_num, 2U); + ASSERT_EQ(out.wal_compression, src.wal_compression); + std::string warning = stderr_output; + if (use_logger) { + ASSERT_TRUE(stderr_output.empty()); + src.info_log->Flush(); + ASSERT_OK(ReadFileToString(env_, log_path, &warning)); + } + if (crash_safe) { + ASSERT_NE(warning.find("WARN"), std::string::npos); + ASSERT_NE(warning.find("disabled for read-only Open"), std::string::npos); + ASSERT_NE(warning.find("replay may be very slow"), std::string::npos); + ASSERT_NE(warning.find("much larger MemTables"), std::string::npos); + } else { + ASSERT_TRUE(warning.empty()); + } + } + } +} + +TEST_F(DBOptionsTest, ReadOnlyReplaysWalWithCrashSafeDisabled) { + Options options = CurrentOptions(); + options.memtable_crash_safe_recover = false; + options.avoid_flush_during_shutdown = true; + DestroyAndReopen(options); + ASSERT_OK(Put("k", "v")); + Close(); + options.memtable_crash_safe_recover = true; + DB* ro = nullptr; + ASSERT_OK(DB::OpenForReadOnly(options, dbname_, &ro)); + std::unique_ptr read_only_db(ro); + ASSERT_FALSE(ro->GetDBOptions().memtable_crash_safe_recover); + std::string value; + ASSERT_OK(ro->Get(ReadOptions(), "k", &value)); + ASSERT_EQ(value, "v"); +} + +TEST_F(DBOptionsTest, CrashSafeForcesWal) { + for (bool crash_safe : {false, true}) { + SCOPED_TRACE(crash_safe); + Options options = CurrentOptions(); + if (crash_safe) { +#ifdef HAS_TOPLING_CSPP_MEMTABLE + options.memtable_factory = EasyNewMemTableRep( + "CSPPMemTab", R"({"mem_cap":16777216,"convert_to_sst":"kFileMmap"})"); + options.avoid_flush_during_shutdown = true; +#else + continue; +#endif + } + options.memtable_as_log_index = false; + options.memtable_crash_safe_recover = crash_safe; + options.statistics = CreateDBStatistics(); + Reopen(options); + WriteOptions write; + write.disableWAL = true; + const uint64_t before = options.statistics->getTickerCount(WAL_FILE_BYTES); + ASSERT_OK(db_->Put(write, "k", "v")); + const uint64_t after = options.statistics->getTickerCount(WAL_FILE_BYTES); + if (crash_safe) { + ASSERT_GT(after, before); + } else { + ASSERT_EQ(after, before); + } + } +} + TEST_F(DBOptionsTest, ImmutableTrackAndVerifyWalsInManifest) { Options options; options.env = env_; @@ -876,6 +1004,39 @@ TEST_F(DBOptionsTest, MaxOpenFilesChange) { Close(); } +TEST_F(DBOptionsTest, SmallMaxOpenFiles) { + Options options = CurrentOptions(); + options.table_cache_numshardbits = 0; + options.statistics = CreateDBStatistics(); + options.disable_auto_compactions = true; + options.skip_stats_update_on_db_open = true; + Reopen(options); + ASSERT_OK(Put("k", "v")); + ASSERT_OK(Flush()); + Close(); + for (int max_open_files : {0, 1, 9, 10, 11}) { + SCOPED_TRACE(max_open_files); + options.max_open_files = max_open_files; + options.statistics->getAndResetTickerCount(NO_FILE_OPENS); + Reopen(options); + ASSERT_EQ(options.statistics->getTickerCount(NO_FILE_OPENS), 0); + Cache* tc = dbfull()->TEST_table_cache(); + const size_t capacity = std::max(0, max_open_files - 10); + ASSERT_EQ(db_->GetDBOptions().max_open_files, max_open_files); + ASSERT_EQ(tc->GetCapacity(), capacity); + ASSERT_OK(db_->SetDBOptions({{"max_background_jobs", "4"}})); + ASSERT_EQ(tc->GetCapacity(), capacity); + ASSERT_OK(db_->SetDBOptions({{"max_open_files", "1024"}})); + ASSERT_EQ(tc->GetCapacity(), 1014U); + ASSERT_OK(db_->SetDBOptions( + {{"max_open_files", std::to_string(max_open_files)}})); + ASSERT_EQ(tc->GetCapacity(), capacity); + ASSERT_EQ(Get("k"), "v"); + ASSERT_GT(options.statistics->getTickerCount(NO_FILE_OPENS), 0); + Close(); + } +} + TEST_F(DBOptionsTest, SanitizeDelayedWriteRate) { Options options; options.env = CurrentOptions().env; diff --git a/db/db_range_del_test.cc b/db/db_range_del_test.cc index a84cad4c56..8000ecd663 100644 --- a/db/db_range_del_test.cc +++ b/db/db_range_del_test.cc @@ -3484,6 +3484,10 @@ TEST_F(DBRangeDelTest, NonBottommostCompactionDropRangetombstone) { } TEST_F(DBRangeDelTest, MemtableMaxRangeDeletions) { + SyncPoint::GetInstance()->SetCallBack( + "ColumnFamilyData::ConstructNewMemtable:UseCache", + [](void* arg) { *static_cast(arg) = false; }); + SyncPoint::GetInstance()->EnableProcessing(); // Tests option `memtable_max_range_deletions`. Options options = CurrentOptions(); options.level_compaction_dynamic_file_size = false; diff --git a/db/db_test.cc b/db/db_test.cc index ca7e2716a9..f02bcab133 100644 --- a/db/db_test.cc +++ b/db/db_test.cc @@ -4497,6 +4497,10 @@ TEST_F(DBTest, ManualFlushWalAndWriteRace) { } TEST_F(DBTest, DynamicMemtableOptions) { + SyncPoint::GetInstance()->SetCallBack( + "ColumnFamilyData::ConstructNewMemtable:UseCache", + [](void* arg) { *static_cast(arg) = false; }); + SyncPoint::GetInstance()->EnableProcessing(); const uint64_t k64KB = 1 << 16; const uint64_t k128KB = 1 << 17; const uint64_t k5KB = 5 * 1024; diff --git a/db/db_test2.cc b/db/db_test2.cc index 4ca59a5ec7..020fa41788 100644 --- a/db/db_test2.cc +++ b/db/db_test2.cc @@ -324,6 +324,9 @@ TEST_P(DBTestSharedWriteBufferAcrossCFs, SharedWriteBufferAcrossCFs) { static_cast*>(arg); *std::get<0>(*pair) = *std::get<1>(*pair); }); + SyncPoint::GetInstance()->SetCallBack( + "ColumnFamilyData::PrepareNewMemtableInBackground:UseCache", + [](void* arg) { *static_cast(arg) = false; }); ROCKSDB_NAMESPACE::SyncPoint::GetInstance()->EnableProcessing(); // The total soft write buffer size is about 105000 @@ -5770,6 +5773,10 @@ TEST_F(DBTest2, SeekFileRangeDeleteTail) { } TEST_F(DBTest2, BackgroundPurgeTest) { + SyncPoint::GetInstance()->SetCallBack( + "ColumnFamilyData::PrepareNewMemtableInBackground:UseCache", + [](void* arg) { *static_cast(arg) = false; }); + SyncPoint::GetInstance()->EnableProcessing(); Options options = CurrentOptions(); options.write_buffer_manager = std::make_shared(1 << 20); diff --git a/db/db_write_test.cc b/db/db_write_test.cc index 4de24dcb7f..92b810d14d 100644 --- a/db/db_write_test.cc +++ b/db/db_write_test.cc @@ -39,6 +39,20 @@ class DBWriteTestUnparameterized : public DBTestBase { : DBTestBase("pipelined_write_test", /*env_do_fsync=*/false) {} }; +TEST_F(DBWriteTestUnparameterized, UnorderedWriteWithoutWALCanFlush) { + Options options = CurrentOptions(); + options.unordered_write = true; + Reopen(options); + WriteOptions write; + for (bool disable_wal : {false, true}) { + write.disableWAL = disable_wal; + ASSERT_OK(db_->Put(write, "key", disable_wal ? "no-wal" : "wal")); + ASSERT_OK(Flush()); + } + Reopen(options); + ASSERT_EQ(Get("key"), "no-wal"); +} + // It is invalid to do sync write while disabling WAL. TEST_P(DBWriteTest, SyncAndDisableWAL) { if (GetOptions().memtable_as_log_index) { diff --git a/db/flush_job.cc b/db/flush_job.cc index 1adb729d23..9294e4cc61 100644 --- a/db/flush_job.cc +++ b/db/flush_job.cc @@ -209,7 +209,15 @@ void FlushJob::PickMemTable() { edit_->SetColumnFamily(cfd_->GetID()); // path 0 for level 0 file. - meta_.fd = FileDescriptor(versions_->NewFileNumber(), 0, 0); + const uint64_t file_number = [&]() { + if (m->SupportCrashSafe()) { + ROCKSDB_ASSERT_EQ(mems_.size(), 1); + return m->GetFileNumber(); + } else { + return versions_->NewFileNumber(); + } + }(); + meta_.fd = FileDescriptor(file_number, 0, 0); meta_.epoch_number = cfd_->NewEpochNumber(); base_ = cfd_->current(); @@ -229,7 +237,8 @@ Status FlushJob::Run(LogsWithPrepTracker* prep_tracker, FileMetaData* file_meta, double mempurge_threshold = mutable_cf_options_.experimental_mempurge_threshold; - if (db_options_.memtable_as_log_index) { + if (db_options_.memtable_as_log_index || + cfd_->ioptions()->memtable_factory->SupportCrashSafe()) { mempurge_threshold = 0; // not supported } @@ -963,7 +972,8 @@ Status FlushJob::WriteLevel0Table() { TableFileCreationReason::kFlush, oldest_key_time, current_time, db_id_, db_session_id_, 0 /* target_file_size */, meta_.fd.GetNumber()); - if (mems_.size() == 1 && mems_.front()->SupportConvertToSST()) { + if (mems_.front()->SupportConvertToSST()) { + ROCKSDB_ASSERT_EQ(mems_.size(), 1); // convert MemTable to sst MemTable* memtable = mems_.front(); // pass these fields to ConvertToSST, to fill TableProperties @@ -980,6 +990,7 @@ Status FlushJob::WriteLevel0Table() { blob_file_additions.push_back(std::move(b)); }; s = memtable->ConvertToSST(&meta_, tboptions); + TEST_SYNC_POINT_CALLBACK("FlushJob::ConvertToSST:Status", &s); if (!s.ok()) { ROCKS_LOG_BUFFER(log_buffer_, "[%s] [JOB %d] Level-0 ConvertToSST #%" PRIu64 ": ApproximateMemoryUsage %" PRIu64 @@ -987,19 +998,18 @@ Status FlushJob::WriteLevel0Table() { cfd_->GetName().c_str(), job_context_->job_id, meta_.fd.GetNumber(), memtable->ApproximateMemoryUsage(), s.ToString().c_str()); - goto UseBuildTable; + } else { + meta_.fd.smallest_seqno = std::min(memtable->GetEarliestSequenceNumber(), + memtable->GetFirstSequenceNumber()); + meta_.fd.largest_seqno = memtable->largest_seqno(); + meta_.marked_for_compaction = true; } - meta_.fd.smallest_seqno = std::min(memtable->GetEarliestSequenceNumber(), - memtable->GetFirstSequenceNumber()); - meta_.fd.largest_seqno = memtable->largest_seqno(); - meta_.marked_for_compaction = true; for (auto* p_iter : memtables) { // memtables is vec of memtab iters std::destroy_at(p_iter); // Attention!!! must! } memtables.clear(); } else { // call BuildTable -UseBuildTable: uint64_t num_input_entries = 0; uint64_t memtable_payload_bytes = 0; uint64_t memtable_garbage_bytes = 0; @@ -1076,6 +1086,11 @@ Status FlushJob::WriteLevel0Table() { // should not be added to the manifest. const bool has_output = meta_.fd.GetFileSize() > 0; + if (s.ok() && mems_.front()->IsFileRegistered()) { + ROCKSDB_ASSERT_EQ(mems_.size(), 1); + edit_->DeleteMemTableFile(mems_.front()->GetFileNumber()); + } + if (s.ok() && has_output) { TEST_SYNC_POINT("DBImpl::FlushJob:SSTFileCreated"); // if we have more than 1 background thread, then we cannot diff --git a/db/log_reader.cc b/db/log_reader.cc index 309720b3eb..6beaa929de 100644 --- a/db/log_reader.cc +++ b/db/log_reader.cc @@ -69,6 +69,53 @@ void Reader::InitSetMemTableAsLogIndex(FileSystem& fs) { backing_store_ = nullptr; } +IOStatus Reader::SeekToFileOffset(uint64_t file_offset) { + TEST_SYNC_POINT("CrashSafeRecover::SeekToFileOffset:Before"); + IOStatus io_s; + buffer_ = Slice(); + eof_ = false; + read_error_ = false; + last_record_offset_ = 0; + eof_offset_ = 0; + if (memtable_as_log_index_) { + if (file_offset > 0) { + io_s = file_->Skip(file_offset); + if (!io_s.ok()) { + return io_s; + } + } + end_of_buffer_offset_ = file_offset; + return IOStatus::OK(); + } + const uint64_t frame_start = file_offset - (file_offset % kBlockSize); + if (frame_start > 0) { + io_s = file_->Skip(frame_start); + if (!io_s.ok()) { + return io_s; + } + } + io_s = file_->Read(kBlockSize, &buffer_, backing_store_, Env::IO_TOTAL); + if (!io_s.ok()) { + buffer_.clear(); + end_of_buffer_offset_ = frame_start; + return io_s; + } + end_of_buffer_offset_ = frame_start + buffer_.size(); + const size_t skip_in_frame = static_cast(file_offset - frame_start); + if (buffer_.size() < skip_in_frame) { + eof_ = true; + eof_offset_ = buffer_.size(); + buffer_.clear(); + return IOStatus::OK(); + } + buffer_.remove_prefix(skip_in_frame); + if (end_of_buffer_offset_ - frame_start < kBlockSize) { + eof_ = true; + eof_offset_ = static_cast(end_of_buffer_offset_ - frame_start); + } + return IOStatus::OK(); +} + IOStatus Reader::IsMemTableAsLogIndexFile (FileSystem& fs, const std::string& fname, bool* result) { FileOptions fopt; diff --git a/db/log_reader.h b/db/log_reader.h index ac694eaa9a..acd7c288bc 100644 --- a/db/log_reader.h +++ b/db/log_reader.h @@ -89,9 +89,15 @@ class Reader { // Undefined before the first call to ReadRecord. uint64_t LastRecordOffset(); + // Position a fresh reader whose file is still at offset 0. Only for + // uncompressed, non-recycled WALs with no required metadata in the skipped + // prefix. file_offset must be the physical start of a logical record or EOF. + // Subsequent record offsets remain absolute. Discard the reader on failure. + IOStatus SeekToFileOffset(uint64_t file_offset); + // Returns the first physical offset after the last record returned by - // ReadRecord, or zero before first call to ReadRecord. This can also be - // thought of as the "current" position in processing the file bytes. + // ReadRecord. Before the first ReadRecord, returns the SeekToFileOffset + // position, or zero if not positioned. This is the current processing offset. uint64_t LastRecordEnd(); // returns true if the reader has encountered an eof condition. diff --git a/db/log_test.cc b/db/log_test.cc index 0bf3bf5aec..9b7309c3fe 100644 --- a/db/log_test.cc +++ b/db/log_test.cc @@ -278,6 +278,95 @@ class LogTest } }; +class LogSeekTest : public LogTest {}; + +TEST_P(LogSeekTest, SeekToFileOffsetAtStart) { + Write("first"); + ASSERT_OK(reader_->SeekToFileOffset(0)); + ASSERT_EQ("first", Read()); + ASSERT_EQ(reader_->LastRecordOffset(), 0U); + ASSERT_EQ(reader_->LastRecordEnd(), WrittenBytes()); +} + +TEST_P(LogSeekTest, SeekToFileOffsetAfterFragmentedRecord) { + Write(BigString("B", kBlockSize * 2 + 100)); + const uint64_t tail_offset = WrittenBytes(); + const std::string tail = BigString("tail", kBlockSize + 100); + Write(tail); + const uint64_t tail_end = WrittenBytes(); + Write("last"); + ASSERT_OK(reader_->SeekToFileOffset(tail_offset)); + ASSERT_EQ(tail, Read()); + ASSERT_EQ(reader_->LastRecordOffset(), tail_offset); + ASSERT_EQ(reader_->LastRecordEnd(), tail_end); + ASSERT_EQ("last", Read()); + ASSERT_EQ(reader_->LastRecordOffset(), tail_end); + ASSERT_EQ(reader_->LastRecordEnd(), WrittenBytes()); + ASSERT_EQ("EOF", Read()); +} + +TEST_P(LogSeekTest, SeekToFileOffsetAtBlockBoundary) { + Write(BigString("B", kBlockSize - kHeaderSize)); + ASSERT_EQ(WrittenBytes(), kBlockSize); + Write("tail"); + ASSERT_OK(reader_->SeekToFileOffset(kBlockSize)); + ASSERT_EQ("tail", Read()); + ASSERT_EQ(reader_->LastRecordOffset(), kBlockSize); + ASSERT_EQ(reader_->LastRecordEnd(), WrittenBytes()); +} + +TEST_P(LogSeekTest, SeekToFileOffsetAtEOF) { + Write("first"); + ASSERT_OK(reader_->SeekToFileOffset(WrittenBytes())); + ASSERT_EQ("EOF", Read()); + ASSERT_TRUE(IsEOF()); + ASSERT_EQ(reader_->LastRecordEnd(), WrittenBytes()); +} + +TEST_P(LogSeekTest, SeekToFileOffsetReadError) { + Write("first"); + ForceError(3); + ASSERT_TRUE(reader_->SeekToFileOffset(0).IsCorruption()); +} + +TEST_P(LogSeekTest, SeekToFileOffsetSkipError) { + Write("first"); + ASSERT_TRUE(reader_->SeekToFileOffset(kBlockSize).IsNotFound()); +} + +INSTANTIATE_TEST_CASE_P( + Log, LogSeekTest, + ::testing::Combine(::testing::Values(0), ::testing::Bool(), + ::testing::Values(CompressionType::kNoCompression))); + +TEST(LogReaderSeekTest, SeekToFileOffsetLogIndex) { + const auto fs = Env::Default()->GetFileSystem(); + const std::string fname = test::PerThreadDBPath("log_seek_index"); + { + std::unique_ptr dest; + ASSERT_OK(WritableFileWriter::Create(fs, fname, FileOptions(), &dest, nullptr)); + Writer writer(std::move(dest), 1, false); + writer.InitReaderMmap(*fs, kBlockSize * 4); + ASSERT_OK(writer.AddRecord(BigString("B", kBlockSize * 2 + 100))); + const uint64_t tail_offset = writer.get_log_offset(); + ASSERT_OK(writer.AddRecord("tail")); + std::unique_ptr file; + ASSERT_OK(SequentialFileReader::Create(fs, fname, FileOptions(), &file, + nullptr, nullptr)); + Reader reader(nullptr, std::move(file), nullptr, true, 1); + reader.InitSetMemTableAsLogIndex(*fs); + ASSERT_OK(reader.SeekToFileOffset(tail_offset - sizeof(RawRecHeader))); + Slice record; + std::string scratch; + ASSERT_TRUE(reader.ReadRecord(&record, &scratch)); + ASSERT_EQ(record.ToString(), "tail"); + ASSERT_EQ(reader.LastRecordOffset(), tail_offset); + ASSERT_EQ(reader.LastRecordEnd(), writer.file()->GetFileSize()); + ASSERT_FALSE(reader.ReadRecord(&record, &scratch)); + } + ASSERT_OK(fs->DeleteFile(fname, IOOptions(), nullptr)); +} + TEST_P(LogTest, Empty) { ASSERT_EQ("EOF", Read()); } TEST_P(LogTest, ReadWrite) { diff --git a/db/memtable.cc b/db/memtable.cc index 6c8bbfab61..4167c0102a 100644 --- a/db/memtable.cc +++ b/db/memtable.cc @@ -22,6 +22,7 @@ #include "db/range_tombstone_fragmenter.h" #include "db/read_callback.h" #include "db/wide/wide_column_serialization.h" +#include "file/filename.h" #include "logging/logging.h" #include "memory/arena.h" #include "memory/memory_usage.h" @@ -72,7 +73,8 @@ MemTable::MemTable(const InternalKeyComparator& cmp, const ImmutableOptions& ioptions, const MutableCFOptions& mutable_cf_options, WriteBufferManager* write_buffer_manager, - SequenceNumber latest_seq, uint32_t column_family_id) + SequenceNumber latest_seq, uint32_t column_family_id, + uint64_t file_number) : comparator_(cmp), moptions_(ioptions, mutable_cf_options), refs_(0), @@ -86,7 +88,9 @@ MemTable::MemTable(const InternalKeyComparator& cmp, : nullptr, mutable_cf_options.memtable_huge_page_size), table_(ioptions.memtable_factory->CreateMemTableRep( - ioptions.cf_paths[0].path, // level0_dir + file_number == 0 + ? std::string() + : TableFileName(ioptions.cf_paths, file_number, 0), mutable_cf_options, comparator_, &arena_, mutable_cf_options.prefix_extractor.get(), ioptions.logger, column_family_id)), @@ -105,7 +109,8 @@ MemTable::MemTable(const InternalKeyComparator& cmp, write_buffer_size_(mutable_cf_options.write_buffer_size), flush_in_progress_(false), flush_completed_(false), - file_number_(0), + file_registered_(false), + file_number_(file_number), first_seqno_(0), earliest_seqno_(latest_seq), creation_seq_(latest_seq), @@ -214,8 +219,10 @@ bool MemTable::ShouldFlushNow() { // If arena still have room for new block allocation, we can safely say it // shouldn't flush. auto allocated_memory = table_->ApproximateMemoryUsage() + - range_del_table_->ApproximateMemoryUsage() + arena_.MemoryAllocatedBytes(); + if (!is_range_del_table_empty_.load(std::memory_order_relaxed)) { + allocated_memory += range_del_table_->ApproximateMemoryUsage(); + } approximate_memory_usage_.store(allocated_memory, std::memory_order_relaxed); @@ -603,6 +610,7 @@ FragmentedRangeTombstoneIterator* MemTable::NewRangeTombstoneIterator( immutable_memtable); } +terark_no_inline FragmentedRangeTombstoneIterator* MemTable::NewRangeTombstoneIteratorInternal( const ReadOptions& read_options, SequenceNumber read_seq, bool immutable_memtable) { @@ -784,8 +792,8 @@ Status MemTable::Add(SequenceNumber s, ValueType type, size_t encoded_len = MemTableRep::EncodeKeyValueSize(key, real_value); if (!allow_concurrent) { - // Extract prefix for insert with hint. if (insert_with_hint_prefix_extractor_ != nullptr && + needs_user_key_cmp_in_get_ && // disable per-prefix hint for Topling insert_with_hint_prefix_extractor_->InDomain(key)) { Slice prefix = insert_with_hint_prefix_extractor_->Transform(key); hint = &insert_hints_[prefix]; // overwrite hint? @@ -793,6 +801,14 @@ Status MemTable::Add(SequenceNumber s, ValueType type, if (UNLIKELY(!res)) { return Status::TryAgain("key+seq exists"); } + } else if (hint && !needs_user_key_cmp_in_get_) { + // Borrow NeedsUserKeyCompareInGet: false on Topling (hint pins the + // writer token; pair with FinishHint), true on upstream RocksDB + // (ignore caller hint except the prefix-extractor path above). + bool res = table->InsertKeyValueWithHint(tag, key, value, hint); + if (UNLIKELY(!res)) { + return Status::TryAgain("key+seq exists"); + } } else { bool res = table->InsertKeyValue(tag, key, value); if (UNLIKELY(!res)) { @@ -826,7 +842,7 @@ Status MemTable::Add(SequenceNumber s, ValueType type, largest_seqno_.store(s, std::memory_order_relaxed); } - if (bloom_filter_) { + if (UNLIKELY(nullptr != bloom_filter_)) { #if defined(TOPLINGDB_WITH_TIMESTAMP) size_t ts_sz = GetInternalKeyComparator().user_comparator()->timestamp_size(); Slice key_without_ts = StripTimestampFromUserKey(key, ts_sz); @@ -843,7 +859,7 @@ Status MemTable::Add(SequenceNumber s, ValueType type, // The first sequence number inserted into the memtable assert(first_seqno_ == 0 || s >= first_seqno_); - if (first_seqno_ == 0) { + if (UNLIKELY(first_seqno_.load(std::memory_order_relaxed) == 0)) { first_seqno_.store(s, std::memory_order_relaxed); if (earliest_seqno_ == kMaxSequenceNumber) { @@ -856,7 +872,9 @@ Status MemTable::Add(SequenceNumber s, ValueType type, // TODO(yuzhangyu): support updating newest UDT for when `allow_concurrent` // is true. MaybeUpdateNewestUDT(key); // user key - UpdateFlushState(); + if (!hint || needs_user_key_cmp_in_get_) { + UpdateFlushState(); + } } else { bool res = (hint == nullptr) @@ -882,7 +900,7 @@ Status MemTable::Add(SequenceNumber s, ValueType type, post_process_info->largest_seqno = s; } - if (bloom_filter_) { + if (UNLIKELY(nullptr != bloom_filter_)) { #if defined(TOPLINGDB_WITH_TIMESTAMP) size_t ts_sz = GetInternalKeyComparator().user_comparator()->timestamp_size(); Slice key_without_ts = StripTimestampFromUserKey(key, ts_sz); diff --git a/db/memtable.h b/db/memtable.h index 04549a3187..d291fa46ae 100644 --- a/db/memtable.h +++ b/db/memtable.h @@ -154,7 +154,8 @@ class MemTable : public CacheAlignedNewDelete { const ImmutableOptions& ioptions, const MutableCFOptions& mutable_cf_options, WriteBufferManager* write_buffer_manager, - SequenceNumber earliest_seq, uint32_t column_family_id); + SequenceNumber earliest_seq, uint32_t column_family_id, + uint64_t file_number = 0); // No copying allowed MemTable(const MemTable&) = delete; MemTable& operator=(const MemTable&) = delete; @@ -235,6 +236,9 @@ class MemTable : public CacheAlignedNewDelete { std::memory_order_relaxed); } + // Updates flush_state_ using ShouldFlushNow() + void UpdateFlushState(); + // Return an iterator that yields the contents of the memtable. // // The caller must ensure that the underlying MemTable remains live @@ -547,7 +551,7 @@ class MemTable : public CacheAlignedNewDelete { void FinishHint(void* hint) const { table_->FinishHint(hint); } bool SupportConvertToSST() const { - return table_->SupportConvertToSST() && is_range_del_table_empty_; + return support_convert_to_sst_ && is_range_del_table_empty_; } Status ConvertToSST(struct FileMetaData*, const struct TableBuilderOptions&); @@ -582,6 +586,9 @@ class MemTable : public CacheAlignedNewDelete { void SetFlushCompleted(bool completed) { flush_completed_ = completed; } uint64_t GetFileNumber() const { return file_number_; } + bool SupportCrashSafe() const { return table_->SupportCrashSafe(); } + bool IsFileRegistered() const { return file_registered_; } + void MarkFileRegistered() { file_registered_ = true; } void SetFileNumber(uint64_t file_num) { file_number_ = file_num; } @@ -658,10 +665,11 @@ class MemTable : public CacheAlignedNewDelete { // These are used to manage memtable flushes to storage bool flush_in_progress_; // started the flush bool flush_completed_; // finished the flush + bool file_registered_; bool needs_user_key_cmp_in_get_; bool support_convert_to_sst_; bool reject_memtable_as_log_index_; - uint64_t file_number_; // filled up after flush is complete + uint64_t file_number_; // FileMmap backing file or flush output // The updates to be applied to the transaction log when this // memtable is flushed to storage. @@ -738,9 +746,6 @@ class MemTable : public CacheAlignedNewDelete { terark::minimal_sso<32> newest_udt_; #endif - // Updates flush_state_ using ShouldFlushNow() - void UpdateFlushState(); - void UpdateOldestKeyTime(); diff --git a/db/memtable_list.cc b/db/memtable_list.cc index 9eacca84e9..7a62ef2f42 100644 --- a/db/memtable_list.cc +++ b/db/memtable_list.cc @@ -453,7 +453,7 @@ void MemTableList::RollbackMemtableFlush(const autovector& mems, #ifndef NDEBUG for (MemTable* m : mems) { assert(m->flush_in_progress_); - assert(m->file_number_ == 0); + assert(m->file_number_ == 0 || m->SupportCrashSafe()); } #endif @@ -475,7 +475,9 @@ void MemTableList::RollbackMemtableFlush(const autovector& mems, m->flush_in_progress_ = false; m->flush_completed_ = false; m->edit_.Clear(); - m->file_number_ = 0; + if (!m->SupportCrashSafe()) { + m->file_number_ = 0; + } num_flush_not_started_++; ++it; } else { @@ -486,8 +488,7 @@ void MemTableList::RollbackMemtableFlush(const autovector& mems, for (MemTable* m : mems) { if (m->flush_in_progress_) { - assert(m->file_number_ == 0); - m->file_number_ = 0; + assert(m->file_number_ == 0 || m->SupportCrashSafe()); m->flush_in_progress_ = false; m->flush_completed_ = false; m->edit_.Clear(); @@ -621,10 +622,16 @@ Status MemTableList::TryInstallMemtableFlushResults( const auto manifest_write_cb = [this, cfd, batch_count, log_buffer, to_delete, mu](const Status& status) { + if (status.ok() && !cfd->IsDropped()) { + cfd->PublishRegisteredMemTables(); + TEST_SYNC_POINT("FlushJob::AfterManifest"); + } RemoveMemTablesOrRestoreFlags(status, cfd, batch_count, log_buffer, to_delete, mu); }; if (write_edits) { + cfd->AddMemTableFileEdits(edit_list.front()); + TEST_SYNC_POINT("FlushJob::BeforeManifest"); // this can release and reacquire the mutex. s = vset->LogAndApply(cfd, mutable_cf_options, read_options, edit_list, mu, db_directory, /*new_descriptor_log=*/false, @@ -688,6 +695,21 @@ size_t MemTableList::ApproximateUnflushedMemTablesMemoryUsage() { return total_size; } +bool MemTableList::UnflushedMemtablesSupportConvertToSST() const { + for (MemTable* m : current_->memlist_) { + if (!m->SupportConvertToSST()) { + return false; + } + } + return true; +} + +void MemTableList::AddMemTableFileNumbers(std::vector* live) const { + for (MemTable* mem : current_->memlist_) { + live->push_back(mem->GetFileNumber()); + } +} + size_t MemTableList::ApproximateMemoryUsage() { return current_memory_usage_; } size_t MemTableList::MemoryAllocatedBytesExcludingLast() const { @@ -804,7 +826,9 @@ void MemTableList::RemoveMemTablesOrRestoreFlags( m->flush_in_progress_ = false; m->edit_.Clear(); num_flush_not_started_++; - m->file_number_ = 0; + if (!m->SupportCrashSafe()) { + m->file_number_ = 0; + } imm_flush_needed.store(true, std::memory_order_release); ++mem_id; } @@ -878,6 +902,7 @@ Status InstallMemtableAtomicFlushResults( (*mems_list[k])[0]->ReleaseFlushJobInfo(); committed_flush_jobs_info[k]->push_back(std::move(flush_job_info)); } + cfds[k]->AddMemTableFileEdits((*mems_list[k])[0]->GetEdits()); } Status s; @@ -924,11 +949,16 @@ Status InstallMemtableAtomicFlushResults( assert(0 == num_entries); } + TEST_SYNC_POINT("FlushJob::BeforeManifest"); // this can release and reacquire the mutex. s = vset->LogAndApply(cfds, mutable_cf_options_list, read_options, edit_lists, mu, db_directory); for (size_t k = 0; k != cfds.size(); ++k) { + if (s.ok() && !cfds[k]->IsDropped()) { + cfds[k]->PublishRegisteredMemTables(); + TEST_SYNC_POINT("FlushJob::AfterManifest"); + } auto* imm = (imm_lists == nullptr) ? cfds[k]->imm() : imm_lists->at(k); imm->InstallNewVersion(); } @@ -993,7 +1023,9 @@ Status InstallMemtableAtomicFlushResults( m->SetFlushCompleted(false); m->SetFlushInProgress(false); m->GetEdits()->Clear(); - m->SetFileNumber(0); + if (!m->SupportCrashSafe()) { + m->SetFileNumber(0); + } imm->num_flush_not_started_++; } imm->imm_flush_needed.store(true, std::memory_order_release); diff --git a/db/memtable_list.h b/db/memtable_list.h index 6630046b68..2e66b446bb 100644 --- a/db/memtable_list.h +++ b/db/memtable_list.h @@ -332,6 +332,9 @@ class MemTableList { // the unflushed mem-tables. size_t ApproximateUnflushedMemTablesMemoryUsage(); + bool UnflushedMemtablesSupportConvertToSST() const; + void AddMemTableFileNumbers(std::vector* live) const; + // Returns an estimate of the timestamp of the earliest key. uint64_t ApproximateOldestKeyTime() const; diff --git a/db/range_tombstone_fragmenter.cc b/db/range_tombstone_fragmenter.cc index 7e7cedeca4..07c2571055 100644 --- a/db/range_tombstone_fragmenter.cc +++ b/db/range_tombstone_fragmenter.cc @@ -467,6 +467,7 @@ bool FragmentedRangeTombstoneIterator::Valid() const { return tombstones_ != nullptr && pos_ != tombstones_->end(); } +terark_no_inline SequenceNumber FragmentedRangeTombstoneIterator::MaxCoveringTombstoneSeqnum( const Slice& target_user_key) { SeekToCoveringTombstone(target_user_key); diff --git a/db/version_edit.cc b/db/version_edit.cc index b0af0d87aa..29d9c80905 100644 --- a/db/version_edit.cc +++ b/db/version_edit.cc @@ -86,6 +86,9 @@ void VersionEdit::Clear() { wal_additions_.clear(); wal_deletion_.Reset(); column_family_ = 0; + memtable_file_additions_.clear(); + memtable_file_deletions_.clear(); + has_memtable_file_tracking_ = false; is_column_family_add_ = false; is_column_family_drop_ = false; column_family_name_.clear(); @@ -288,6 +291,17 @@ bool VersionEdit::EncodeTo(std::string* dst, } // 0 is default and does not need to be explicitly written + if (has_memtable_file_tracking_) { + PutVarint32(dst, kMemTableFileTracking); + } + for (uint64_t number : memtable_file_additions_) { + if (number == 0 || number > kFileNumberMask) return false; + PutVarint32Varint64(dst, kMemTableFileAddition, number); + } + for (uint64_t number : memtable_file_deletions_) { + if (number == 0 || number > kFileNumberMask) return false; + PutVarint32Varint64(dst, kMemTableFileDeletion, number); + } if (column_family_ != 0) { PutVarint32Varint32(dst, kColumnFamily, column_family_); } @@ -768,6 +782,24 @@ Status VersionEdit::DecodeFrom(const Slice& src) { is_column_family_drop_ = true; break; + case kMemTableFileTracking: + has_memtable_file_tracking_ = true; + break; + + case kMemTableFileAddition: + case kMemTableFileDeletion: { + uint64_t number; + if (!GetVarint64(&input, &number) || number == 0 || + number > kFileNumberMask) { + msg = "invalid MemTable file number"; + } else if (tag == kMemTableFileAddition) { + memtable_file_additions_.insert(number); + } else { + memtable_file_deletions_.insert(number); + } + break; + } + case kInAtomicGroup: is_in_atomic_group_ = true; if (!GetVarint32(&input, &remaining_entries_)) { @@ -948,6 +980,15 @@ std::string VersionEdit::DebugString(bool hex_key) const { r.append("\n ColumnFamily: "); AppendNumberTo(&r, column_family_); + if (has_memtable_file_tracking_) r.append("\n MemTableFileTracking: true"); + for (uint64_t number : memtable_file_additions_) { + r.append("\n AddMemTableFile: "); + AppendNumberTo(&r, number); + } + for (uint64_t number : memtable_file_deletions_) { + r.append("\n DeleteMemTableFile: "); + AppendNumberTo(&r, number); + } if (is_column_family_add_) { r.append("\n ColumnFamilyAdd: "); r.append(column_family_name_); @@ -1098,6 +1139,19 @@ std::string VersionEdit::DebugJSON(int edit_num, bool hex_key) const { } jw << "ColumnFamily" << column_family_; + if (has_memtable_file_tracking_) jw << "MemTableFileTracking" << true; + if (!memtable_file_additions_.empty()) { + jw << "MemTableFileAdditions"; + jw.StartArray(); + for (uint64_t number : memtable_file_additions_) jw << number; + jw.EndArray(); + } + if (!memtable_file_deletions_.empty()) { + jw << "MemTableFileDeletions"; + jw.StartArray(); + for (uint64_t number : memtable_file_deletions_) jw << number; + jw.EndArray(); + } if (is_column_family_add_) { jw << "ColumnFamilyAdd" << column_family_name_; diff --git a/db/version_edit.h b/db/version_edit.h index ce35758d41..aa6954cfe3 100644 --- a/db/version_edit.h +++ b/db/version_edit.h @@ -63,6 +63,11 @@ enum Tag : uint32_t { kBlobFileAddition = 400, kBlobFileGarbage, + // Required recovery metadata: older readers must reject these tags. + kMemTableFileAddition = 500, + kMemTableFileDeletion = 501, + kMemTableFileTracking = 502, + // Mask for an unidentified tag from the future which can be safely ignored. kTagSafeIgnoreMask = 1 << 13, @@ -652,6 +657,7 @@ class VersionEdit { size_t NumEntries() const { return new_files_.size() + deleted_files_.size() + blob_file_additions_.size() + blob_file_garbages_.size() + + memtable_file_additions_.size() + memtable_file_deletions_.size() + wal_additions_.size() + !wal_deletion_.IsEmpty(); } @@ -660,6 +666,17 @@ class VersionEdit { } uint32_t GetColumnFamily() const { return column_family_; } + void AddMemTableFile(uint64_t number) { memtable_file_additions_.insert(number); } + void DeleteMemTableFile(uint64_t number) { memtable_file_deletions_.insert(number); } + const std::set& GetMemTableFileAdditions() const { + return memtable_file_additions_; + } + const std::set& GetMemTableFileDeletions() const { + return memtable_file_deletions_; + } + void SetMemTableFileTracking() { has_memtable_file_tracking_ = true; } + bool HasMemTableFileTracking() const { return has_memtable_file_tracking_; } + const std::string& GetColumnFamilyName() const { return column_family_name_; } // set column family ID by calling SetColumnFamily() @@ -774,6 +791,9 @@ class VersionEdit { // Each version edit record should have column_family_ set // If it's not set, it is default (0) uint32_t column_family_ = 0; + std::set memtable_file_additions_; + std::set memtable_file_deletions_; + bool has_memtable_file_tracking_ = false; // a version edit can be either column_family add or // column_family drop. If it's column family add, // it also includes column family name. diff --git a/db/version_edit_handler.cc b/db/version_edit_handler.cc index 395d066d74..06c92dcb26 100644 --- a/db/version_edit_handler.cc +++ b/db/version_edit_handler.cc @@ -213,6 +213,7 @@ Status VersionEditHandler::ApplyVersionEdit(VersionEdit& edit, if (s.ok()) { assert(cfd != nullptr); s = ExtractInfoFromVersionEdit(*cfd, edit); + if (s.ok()) version_set_->ApplyMemTableFileEdit(edit); } return s; } @@ -528,6 +529,7 @@ ColumnFamilyData* VersionEditHandler::DestroyCfAndCleanup( ColumnFamilyData* ret = version_set_->GetColumnFamilySet()->GetColumnFamily(cf_id); assert(ret != nullptr); + ret->ApplyMemTableFileEdit(edit); ret->SetDropped(); ret->UnrefAndTryDelete(); ret = nullptr; diff --git a/db/version_edit_test.cc b/db/version_edit_test.cc index 2523520699..8767ef49ee 100644 --- a/db/version_edit_test.cc +++ b/db/version_edit_test.cc @@ -40,6 +40,56 @@ static void TestEncodeDecode(const VersionEdit& edit) { class VersionEditTest : public testing::Test {}; +TEST_F(VersionEditTest, MemTableFilesEncodeDecodeAndClear) { + VersionEdit edit; + edit.SetColumnFamily(7); + edit.SetMemTableFileTracking(); + edit.AddMemTableFile(123); + edit.AddMemTableFile(kFileNumberMask); + edit.DeleteMemTableFile(42); + edit.MarkAtomicGroup(0); + std::string encoded; + ASSERT_TRUE(edit.EncodeTo(&encoded)); + VersionEdit parsed; + ASSERT_OK(parsed.DecodeFrom(encoded)); + ASSERT_EQ(parsed.GetColumnFamily(), 7U); + ASSERT_TRUE(parsed.HasMemTableFileTracking()); + ASSERT_EQ(parsed.GetMemTableFileAdditions(), edit.GetMemTableFileAdditions()); + ASSERT_EQ(parsed.GetMemTableFileDeletions(), edit.GetMemTableFileDeletions()); + ASSERT_TRUE(parsed.IsInAtomicGroup()); + ASSERT_EQ(parsed.GetRemainingEntries(), 0U); + parsed.Clear(); + ASSERT_FALSE(parsed.HasMemTableFileTracking()); + ASSERT_TRUE(parsed.GetMemTableFileAdditions().empty()); + ASSERT_TRUE(parsed.GetMemTableFileDeletions().empty()); + ASSERT_FALSE(parsed.IsInAtomicGroup()); +} + +TEST_F(VersionEditTest, MemTableFilesRejectInvalidNumbers) { + for (uint32_t tag : {uint32_t(kMemTableFileAddition), + uint32_t(kMemTableFileDeletion)}) { + for (uint64_t number : {uint64_t(0), kFileNumberMask + 1}) { + std::string encoded; + PutVarint32Varint64(&encoded, tag, number); + VersionEdit parsed; + ASSERT_TRUE(parsed.DecodeFrom(encoded).IsCorruption()); + VersionEdit invalid; + if (tag == kMemTableFileAddition) invalid.AddMemTableFile(number); + else invalid.DeleteMemTableFile(number); + encoded.clear(); + ASSERT_FALSE(invalid.EncodeTo(&encoded)); + } + std::string truncated; + PutVarint32(&truncated, tag); + truncated.push_back(char(0x80)); + VersionEdit parsed; + ASSERT_TRUE(parsed.DecodeFrom(truncated).IsCorruption()); + } + ASSERT_EQ(kMemTableFileAddition & kTagSafeIgnoreMask, 0U); + ASSERT_EQ(kMemTableFileDeletion & kTagSafeIgnoreMask, 0U); + ASSERT_EQ(kMemTableFileTracking & kTagSafeIgnoreMask, 0U); +} + TEST_F(VersionEditTest, EncodeDecode) { static const uint64_t kBig = 1ull << 50; static const uint32_t kBig32Bit = 1ull << 30; diff --git a/db/version_set.cc b/db/version_set.cc index 573c57e970..afdb0c4f7a 100644 --- a/db/version_set.cc +++ b/db/version_set.cc @@ -129,6 +129,7 @@ inline uint64_t HostPrefixCache(const ParsedInternalKey& ikey) { } template +__attribute__((always_inline)) size_t FindFileInRangeTmpl(Cmp cmp, const LevelFilesBrief& brief, const ParsedInternalKey& key, size_t lo, size_t hi) { const uint64_t* pxcache = brief.prefix_cache; @@ -5505,6 +5506,8 @@ void VersionSet::Reset() { obsolete_files_.clear(); obsolete_manifests_.clear(); wals_.Reset(); + has_memtable_file_tracking_ = false; + replaying_manifest_ = true; } void VersionSet::AppendVersion(ColumnFamilyData* column_family_data, @@ -5743,6 +5746,7 @@ Status VersionSet::ProcessManifestWrites( std::unordered_map curr_state; VersionEdit wal_additions; if (new_descriptor_log) { + if (has_memtable_file_tracking_) wal_additions.SetMemTableFileTracking(); pending_manifest_file_number_ = NewFileNumber(); batch_edits.back()->SetNextFile(next_file_number_.load()); @@ -5956,6 +5960,40 @@ Status VersionSet::ProcessManifestWrites( // Install the new versions if (s.ok()) { + // Only a committed live batch can retire backing files. MANIFEST replay + // applies registry edits without scheduling historical deletions. + std::map retired_memtable_files; + for (const auto* edit : batch_edits) { + auto* cfd = column_family_set_->GetColumnFamily(edit->GetColumnFamily()); + if (cfd != nullptr) { + const auto& files = cfd->GetMemTableFiles(); + const auto& path = cfd->ioptions()->cf_paths[0].path; + if (edit->IsColumnFamilyDrop()) { + for (uint64_t number : files) { + retired_memtable_files.emplace(number, path); + } + } else { + for (uint64_t number : edit->GetMemTableFileDeletions()) { + assert(files.count(number)); + retired_memtable_files.emplace(number, path); + } + } + } + ApplyMemTableFileEdit(*edit); + } + if (!retired_memtable_files.empty()) { + // In-place conversion transfers ownership to the installed SST version. + for (const auto* edit : batch_edits) { + for (const auto& file : edit->GetNewFiles()) { + retired_memtable_files.erase(file.second.fd.GetNumber()); + } + } + for (const auto& file : retired_memtable_files) { + auto* metadata = new FileMetaData(); + metadata->fd = FileDescriptor(file.first, 0, 0); + obsolete_files_.emplace_back(metadata, file.second); + } + } if (first_writer.edit_list.front()->IsColumnFamilyAdd()) { assert(batch_edits.size() == 1); assert(new_cf_options != nullptr); @@ -6346,6 +6384,7 @@ Status VersionSet::Recover( if (s.ok()) { manifest_file_size_ = current_manifest_file_size; + replaying_manifest_ = false; ROCKS_LOG_INFO( db_options_->info_log, "Recovered from manifest file:%s succeeded," @@ -6515,6 +6554,7 @@ Status VersionSet::TryRecoverFromOneManifest( s = handler_pit.status(); if (s.ok()) { RecoverEpochNumbers(); + replaying_manifest_ = false; } return s; } @@ -6850,7 +6890,8 @@ Status VersionSet::WriteCurrentStateToManifest( } // Save WALs. - if (!wal_additions.GetWalAdditions().empty()) { + if (!wal_additions.GetWalAdditions().empty() || + wal_additions.HasMemTableFileTracking()) { TEST_SYNC_POINT_CALLBACK("VersionSet::WriteCurrentStateToManifest:SaveWal", const_cast(&wal_additions)); std::string record; @@ -6916,6 +6957,11 @@ Status VersionSet::WriteCurrentStateToManifest( VersionEdit edit; edit.SetColumnFamily(cfd->GetID()); + // Only the serialized MANIFEST writer changes this registry. + for (uint64_t number : cfd->GetMemTableFiles()) { + edit.AddMemTableFile(number); + } + const auto* current = cfd->current(); assert(current); @@ -7568,6 +7614,14 @@ uint64_t VersionSet::GetObsoleteSstFilesSize() const { return ret; } +void VersionSet::ApplyMemTableFileEdit(const VersionEdit& edit) { + has_memtable_file_tracking_ |= edit.HasMemTableFileTracking(); + // Recovery may skip CFs that were not requested. + if (auto* cfd = column_family_set_->GetColumnFamily(edit.GetColumnFamily())) { + cfd->ApplyMemTableFileEdit(edit); + } +} + ColumnFamilyData* VersionSet::CreateColumnFamily( const ColumnFamilyOptions& cf_options, const ReadOptions& read_options, const VersionEdit* edit) { @@ -7593,10 +7647,12 @@ ColumnFamilyData* VersionSet::CreateColumnFamily( update_stats); AppendVersion(new_cfd, v); - // GetLatestMutableCFOptions() is safe here without mutex since the - // cfd is not available to client - new_cfd->CreateNewMemtable(*new_cfd->GetLatestMutableCFOptions(), - LastSequence()); + if (!cf_options.memtable_factory->SupportCrashSafe() || !replaying_manifest_) { + // GetLatestMutableCFOptions() is safe here without mutex since the + // cfd is not available to client + new_cfd->CreateNewMemtable(*new_cfd->GetLatestMutableCFOptions(), + LastSequence()); + } new_cfd->SetLogNumber(edit->GetLogNumber()); return new_cfd; } diff --git a/db/version_set.h b/db/version_set.h index 2c4b3bf7e5..0ea12bd89b 100644 --- a/db/version_set.h +++ b/db/version_set.h @@ -1370,6 +1370,10 @@ class VersionSet { // Allocate and return a new file number uint64_t NewFileNumber() { return next_file_number_.fetch_add(1); } + // Access requires the DB mutex, like other MANIFEST state. + bool HasMemTableFileTracking() const { return has_memtable_file_tracking_; } + void ApplyMemTableFileEdit(const VersionEdit& edit); + // Fetch And Add n new file number uint64_t FetchAddFileNumber(uint64_t n) { return next_file_number_.fetch_add(n); @@ -1678,6 +1682,8 @@ class VersionSet { // Protected by DB mutex. WalSet wals_; + bool has_memtable_file_tracking_ = false; + bool replaying_manifest_ = true; std::unique_ptr column_family_set_; Cache* table_cache_; diff --git a/db/version_set_test.cc b/db/version_set_test.cc index 4703b1f44c..a2a4b269c5 100644 --- a/db/version_set_test.cc +++ b/db/version_set_test.cc @@ -1505,6 +1505,156 @@ TEST_F(VersionSetTest, SameColumnFamilyGroupCommit) { EXPECT_EQ(kGroupSize - 1, count); } +TEST_F(VersionSetTest, MemTableRegistryManifestRollover) { + NewDB(); + const uint64_t first = versions_->NewFileNumber(); + const uint64_t second = versions_->NewFileNumber(); + ASSERT_FALSE(versions_->HasMemTableFileTracking()); + VersionEdit addition; + addition.SetMemTableFileTracking(); + addition.AddMemTableFile(first); + addition.AddMemTableFile(second); + ASSERT_OK(LogAndApplyToDefaultCF(addition)); + ASSERT_EQ(versions_->GetColumnFamilySet()->GetDefault()->GetMemTableFiles(), + std::set({first, second})); + CreateNewManifest(); + ReopenDB(); + ASSERT_TRUE(versions_->HasMemTableFileTracking()); + ASSERT_EQ(versions_->GetColumnFamilySet()->GetDefault()->GetMemTableFiles(), + std::set({first, second})); + ASSERT_GT(versions_->current_next_file_number(), second); + VersionEdit deletion; + deletion.DeleteMemTableFile(first); + deletion.DeleteMemTableFile(second); + ASSERT_OK(LogAndApplyToDefaultCF(deletion)); + CreateNewManifest(); + ReopenDB(); + ASSERT_TRUE(versions_->HasMemTableFileTracking()); + for (auto* cfd : *versions_->GetColumnFamilySet()) { + ASSERT_TRUE(cfd->GetMemTableFiles().empty()); + } +} + +TEST_F(VersionSetTest, MemTableRegistryFailedCommitDoesNotEnableTracking) { + NewDB(); + VersionEdit invalid; + invalid.SetMemTableFileTracking(); + invalid.AddMemTableFile(0); + ASSERT_TRUE(LogAndApplyToDefaultCF(invalid).IsCorruption()); + ASSERT_FALSE(versions_->HasMemTableFileTracking()); + for (auto* cfd : *versions_->GetColumnFamilySet()) { + ASSERT_TRUE(cfd->GetMemTableFiles().empty()); + } +} + +TEST_F(VersionSetTest, MemTableRegistryReadOnlySkipsUnopenedCf) { + NewDB(); + const uint64_t first = versions_->NewFileNumber(); + const uint64_t second = versions_->NewFileNumber(); + VersionEdit addition; + addition.SetMemTableFileTracking(); + addition.AddMemTableFile(first); + ASSERT_OK(LogAndApplyToDefaultCF(addition)); + auto* cfd = versions_->GetColumnFamilySet()->GetColumnFamily(1); + addition.Clear(); + addition.SetColumnFamily(1); + addition.AddMemTableFile(second); + mutex_.Lock(); + Status s = versions_->LogAndApply(cfd, mutable_cf_options_, read_options_, + &addition, &mutex_, nullptr); + mutex_.Unlock(); + ASSERT_OK(s); + ASSERT_EQ(cfd->GetMemTableFiles(), std::set({second})); + ASSERT_EQ(versions_->GetColumnFamilySet()->GetDefault()->GetMemTableFiles(), + std::set({first})); + VersionSet read_only(dbname_, &db_options_, env_options_, table_cache_.get(), + &write_buffer_manager_, &write_controller_, nullptr, + nullptr, "", "", "", nullptr); + ASSERT_OK(read_only.Recover({column_families_.front()}, true)); + ASSERT_EQ(read_only.GetColumnFamilySet()->GetColumnFamily(1), nullptr); + ASSERT_EQ(read_only.GetColumnFamilySet()->GetDefault()->GetMemTableFiles(), + std::set({first})); + ASSERT_TRUE(read_only.HasMemTableFileTracking()); + ASSERT_GT(read_only.current_next_file_number(), second); +} + +TEST_F(VersionSetTest, MemTableRegistryRetirementQueuesObsoleteFile) { + NewDB(); + const uint64_t first = versions_->NewFileNumber(); + VersionEdit addition; + addition.SetMemTableFileTracking(); + addition.AddMemTableFile(first); + ASSERT_OK(LogAndApplyToDefaultCF(addition)); + VersionEdit deletion; + deletion.DeleteMemTableFile(first); + ASSERT_OK(LogAndApplyToDefaultCF(deletion)); + std::vector tables; + std::vector blobs; + std::vector manifests; + versions_->GetObsoleteFiles(&tables, &blobs, &manifests, first); + ASSERT_TRUE(tables.empty()); + manifests.clear(); + versions_->GetObsoleteFiles(&tables, &blobs, &manifests, first + 1); + ASSERT_EQ(tables.size(), 1U); + ASSERT_EQ(tables[0].metadata->fd.GetNumber(), first); + ASSERT_EQ(tables[0].path, + versions_->GetColumnFamilySet()->GetDefault()->ioptions()->cf_paths[0].path); + ASSERT_EQ(tables[0].metadata->table_reader_handle, nullptr); + tables[0].DeleteMetadata(); +} + +TEST_F(VersionSetTest, MemTableRegistryDropAndDiscardedEdit) { + NewDB(); + const uint64_t first = versions_->NewFileNumber(); + auto* cfd = versions_->GetColumnFamilySet()->GetColumnFamily(1); + ASSERT_NE(cfd, nullptr); + cfd->Ref(); + VersionEdit addition; + addition.SetColumnFamily(1); + addition.SetMemTableFileTracking(); + addition.AddMemTableFile(first); + mutex_.Lock(); + Status status = versions_->LogAndApply( + cfd, mutable_cf_options_, read_options_, &addition, &mutex_, nullptr); + mutex_.Unlock(); + ASSERT_OK(status); + ASSERT_EQ(cfd->GetMemTableFiles(), std::set({first})); + VersionEdit drop; + drop.SetColumnFamily(1); + drop.DropColumnFamily(); + mutex_.Lock(); + status = versions_->LogAndApply( + cfd, mutable_cf_options_, read_options_, &drop, &mutex_, nullptr); + mutex_.Unlock(); + ASSERT_OK(status); + ASSERT_TRUE(cfd->GetMemTableFiles().empty()); + std::vector retired_tables; + std::vector retired_blobs; + std::vector retired_manifests; + versions_->GetObsoleteFiles(&retired_tables, &retired_blobs, + &retired_manifests, first + 1); + ASSERT_EQ(retired_tables.size(), 1U); + ASSERT_EQ(retired_tables[0].metadata->fd.GetNumber(), first); + ASSERT_EQ(retired_tables[0].path, cfd->ioptions()->cf_paths[0].path); + retired_tables[0].DeleteMetadata(); + VersionEdit discarded; + discarded.SetColumnFamily(1); + discarded.AddMemTableFile(versions_->NewFileNumber()); + mutex_.Lock(); + status = versions_->LogAndApply( + cfd, mutable_cf_options_, read_options_, &discarded, &mutex_, nullptr); + EXPECT_TRUE(cfd->GetMemTableFiles().empty()); + cfd->UnrefAndTryDelete(); + mutex_.Unlock(); + ASSERT_TRUE(status.IsColumnFamilyDropped()); + ASSERT_EQ(versions_->GetColumnFamilySet()->GetColumnFamily(1), nullptr); + ASSERT_TRUE(versions_->HasMemTableFileTracking()); + CreateNewManifest(); + ReopenDB(); + ASSERT_TRUE(versions_->HasMemTableFileTracking()); + ASSERT_EQ(versions_->GetColumnFamilySet()->GetColumnFamily(1), nullptr); +} + TEST_F(VersionSetTest, PersistBlobFileStateInNewManifest) { // Initialize the database and add a couple of blob files, one with some // garbage in it, and one without any garbage. @@ -2624,6 +2774,64 @@ TEST_F(VersionSetAtomicGroupTest, EXPECT_EQ(num_initial_edits_ + kAtomicGroupSize, num_recovered_edits_); } +TEST_F(VersionSetAtomicGroupTest, MemTableRegistryAtomicTransition) { + SetupValidAtomicGroup(3); + const uint64_t first = versions_->NewFileNumber(); + const uint64_t second = versions_->NewFileNumber(); + edits_[0].SetMemTableFileTracking(); + edits_[0].AddMemTableFile(first); + edits_[1].DeleteMemTableFile(first); + edits_[1].AddMemTableFile(second); + edits_[2].SetNextFile(versions_->current_next_file_number()); + AddNewEditsToLog(3); + ASSERT_OK(versions_->Recover(column_families_, false)); + ASSERT_TRUE(versions_->HasMemTableFileTracking()); + ASSERT_EQ(versions_->GetColumnFamilySet()->GetDefault()->GetMemTableFiles(), + std::set({second})); + ASSERT_GT(versions_->current_next_file_number(), second); + auto* cfd = versions_->GetColumnFamilySet()->GetColumnFamily(0); + for (int level = 0; level < cfd->NumberLevels(); ++level) { + ASSERT_TRUE(cfd->current()->storage_info()->LevelFiles(level).empty()); + } + std::vector tables; + std::vector blobs; + std::vector manifests; + versions_->GetObsoleteFiles(&tables, &blobs, &manifests, second + 1); + ASSERT_TRUE(tables.empty()); // Historical replay must never schedule GC. +} + +TEST_F(VersionSetAtomicGroupTest, MemTableRegistryIncompleteAtomicGroup) { + SetupIncompleteTrailingAtomicGroup(3); + const uint64_t first = versions_->NewFileNumber(); + const uint64_t second = versions_->NewFileNumber(); + edits_[0].SetMemTableFileTracking(); + edits_[0].AddMemTableFile(first); + edits_[1].DeleteMemTableFile(first); + edits_[1].AddMemTableFile(second); + edits_[1].SetNextFile(versions_->current_next_file_number()); + AddNewEditsToLog(2); + ASSERT_OK(versions_->Recover(column_families_, false)); + ASSERT_FALSE(versions_->HasMemTableFileTracking()); + for (auto* cfd : *versions_->GetColumnFamilySet()) { + ASSERT_TRUE(cfd->GetMemTableFiles().empty()); + } +} + +TEST_F(VersionSetAtomicGroupTest, MemTableRegistryTrackingSurvivesEmptyList) { + SetupValidAtomicGroup(3); + const uint64_t first = versions_->NewFileNumber(); + edits_[0].SetMemTableFileTracking(); + edits_[0].AddMemTableFile(first); + edits_[1].DeleteMemTableFile(first); + edits_[2].SetNextFile(versions_->current_next_file_number()); + AddNewEditsToLog(3); + ASSERT_OK(versions_->Recover(column_families_, false)); + ASSERT_TRUE(versions_->HasMemTableFileTracking()); + ASSERT_TRUE(versions_->GetColumnFamilySet()->GetDefault() + ->GetMemTableFiles().empty()); + ASSERT_GT(versions_->current_next_file_number(), first); +} + TEST_F(VersionSetAtomicGroupTest, HandleValidAtomicGroupWithReactiveVersionSetReadAndApply) { const int kAtomicGroupSize = 3; @@ -3681,6 +3889,31 @@ TEST_F(VersionSetTestMissingFiles, NoFileMissing) { } } +TEST_F(VersionSetTestMissingFiles, MemTableRegistryConversionKeepsSameNumberSst) { + NewDB(); + const uint64_t first = versions_->NewFileNumber(); + VersionEdit addition; + addition.SetMemTableFileTracking(); + addition.AddMemTableFile(first); + ASSERT_OK(LogAndApplyToDefaultCF(addition)); + SstInfo sst(first, kDefaultColumnFamilyName, "a", 0, 100); + std::vector file_metas; + CreateDummyTableFiles({sst}, &file_metas); + VersionEdit conversion; + conversion.DeleteMemTableFile(first); + conversion.AddFile(0, file_metas[0]); + ASSERT_OK(LogAndApplyToDefaultCF(conversion)); + ASSERT_TRUE(versions_->GetColumnFamilySet()->GetDefault() + ->GetMemTableFiles().empty()); + std::vector tables; + std::vector blobs; + std::vector manifests; + versions_->GetObsoleteFiles(&tables, &blobs, &manifests, first + 1); + ASSERT_TRUE(tables.empty()); + ASSERT_EQ(versions_->GetColumnFamilySet()->GetDefault()->current() + ->storage_info()->LevelFiles(0).size(), 1U); +} + TEST_F(VersionSetTestMissingFiles, MinLogNumberToKeep2PC) { db_options_.allow_2pc = true; NewDB(); diff --git a/db/write_batch.cc b/db/write_batch.cc index 165571a24c..1542c751c1 100644 --- a/db/write_batch.cc +++ b/db/write_batch.cc @@ -2300,6 +2300,7 @@ class MemTableInserter : public WriteBatch::Handler { hint_per_batch_(hint_per_batch) { assert(cf_mems_); memtable_as_log_index_ = cf_mems->GetImmutableDBOptions()->memtable_as_log_index; + skip_memtable_data_ = cf_mems->skip_memtable_data_; } bool memtable_as_log_index() const { return memtable_as_log_index_; } void SetBeginPrepareNextPtr(const char* curr) final { @@ -2313,16 +2314,20 @@ class MemTableInserter : public WriteBatch::Handler { } const WriteBatch* src_batch_ = nullptr; const char* prepare_content_begin_ = nullptr; + bool skip_memtable_data_ = false; ~MemTableInserter() override { if (dup_dectector_on_) { reinterpret_cast(&duplicate_detector_) ->~DuplicateDetector(); } - GetHintMap().for_each([](auto& iter) { + GetHintMap().for_each([this](auto& iter) { // In base MemTableRep, FinishHint do delete [] (char*)(hint). // In ToplingDB CSPP PatriciaTrie, FinishHint idle/release token. iter.first->FinishHint(iter.second); + if (!concurrent_memtable_writes_) { + iter.first->UpdateFlushState(); + } }); delete rebuilding_trx_; } @@ -2394,6 +2399,10 @@ class MemTableInserter : public WriteBatch::Handler { *s = Status::OK(); return false; } + if (skip_memtable_data_) { + *s = Status::OK(); + return false; + } if (has_valid_writes_ != nullptr) { *has_valid_writes_ = true; diff --git a/db/write_batch_internal.h b/db/write_batch_internal.h index ba6f5c5d7d..979da2f805 100644 --- a/db/write_batch_internal.h +++ b/db/write_batch_internal.h @@ -41,6 +41,7 @@ class ColumnFamilyMemTables { virtual ColumnFamilyHandle* GetColumnFamilyHandle() = 0; virtual ColumnFamilyData* current() { return nullptr; } virtual const ImmutableDBOptions* GetImmutableDBOptions() = 0; + bool skip_memtable_data_ = false; }; class ColumnFamilyMemTablesDefault : public ColumnFamilyMemTables { diff --git a/db/write_thread.h b/db/write_thread.h index 633d8bc2df..df2ec2e072 100644 --- a/db/write_thread.h +++ b/db/write_thread.h @@ -83,6 +83,9 @@ class WriteThread { Status status; std::atomic running; size_t size = 0; + // WAL cursor of this group after WriteToWAL (not log_file_number_size). + mutable uint64_t wal_number = 0; + mutable uint64_t wal_offset = 0; struct Iterator { Writer* writer; diff --git a/file/filename.cc b/file/filename.cc index fb7d254721..c78cc8129e 100644 --- a/file/filename.cc +++ b/file/filename.cc @@ -178,6 +178,10 @@ std::string CurrentFileName(const std::string& dbname) { std::string LockFileName(const std::string& dbname) { return dbname + "/LOCK"; } +std::string CrashSafePubSeqFileName(const std::string& dbname) { + return dbname + "/CSPUBSEQ"; +} + std::string TempFileName(const std::string& dbname, uint64_t number) { return MakeFileName(dbname, number, kTempFileNameSuffix.c_str()); } diff --git a/file/filename.h b/file/filename.h index 2eb125b6a1..be747e7c9b 100644 --- a/file/filename.h +++ b/file/filename.h @@ -100,6 +100,9 @@ extern std::string CurrentFileName(const std::string& dbname); // "dbname". The result will be prefixed with "dbname". extern std::string LockFileName(const std::string& dbname); +// Crash-safe recover publish-seq mmap under dbname. +extern std::string CrashSafePubSeqFileName(const std::string& dbname); + // Return the name of a temporary file owned by the db named "dbname". // The result will be prefixed with "dbname". extern std::string TempFileName(const std::string& dbname, uint64_t number); diff --git a/include/rocksdb/memtablerep.h b/include/rocksdb/memtablerep.h index 5ddfc6d76a..32e87d1dfa 100644 --- a/include/rocksdb/memtablerep.h +++ b/include/rocksdb/memtablerep.h @@ -42,6 +42,7 @@ #include #include "rocksdb/customizable.h" +#include "rocksdb/enum_reflection.h" #include "rocksdb/slice.h" namespace ROCKSDB_NAMESPACE { @@ -63,6 +64,8 @@ extern const char* EncodeKey(std::string* scratch, const Slice& target); class MemTableRep : public CacheAlignedNewDelete { public: + ROCKSDB_ENUM_CLASS_INCLASS(ConvertKind, uint8_t, + kDontConvert, kDumpMem, kFileMmap); // KeyComparator provides a means to compare keys, which are internal keys // concatenated with values. class KeyComparator { @@ -308,21 +311,48 @@ class MemTableRep : public CacheAlignedNewDelete { virtual void FinishHint(void*); virtual void InitSetMemTableAsLogIndex(bool) {} virtual bool SupportMemTableAsLogIndex() const { return false; } - virtual bool SupportConvertToSST() const { return false; } + ConvertKind GetConvertKind() const { return m_convert_to_sst; } + bool SupportConvertToSST() const { + return m_convert_to_sst != ConvertKind::kDontConvert; + } + bool SupportCrashSafe() const { + return m_convert_to_sst == ConvertKind::kFileMmap; + } virtual Status ConvertToSST(struct FileMetaData*, const struct TableBuilderOptions&); protected: + // Leftover / converted SST visibility: packed tag exists only when + // seq <= min(lookup_seq, max_visible_seq). Used by crash-safe MemTableRep. + static bool VisibleTag(uint64_t tag, uint64_t lookup_seq, + uint64_t max_visible_seq) { + const uint64_t seq = tag >> 8; + const uint64_t effective_max = + lookup_seq < max_visible_seq ? lookup_seq : max_visible_seq; + return seq <= effective_max; + } + + static uint64_t CapFindTag(uint64_t find_tag, uint64_t max_visible_seq) { + const uint64_t seq = find_tag >> 8; + if (seq <= max_visible_seq) { + return find_tag; + } + return (max_visible_seq << 8) | (find_tag & 0xff); + } + // When *key is an internal key concatenated with the value, returns the // user key. virtual Slice UserKey(const char* key) const; Allocator* allocator_; + ConvertKind m_convert_to_sst = ConvertKind::kDontConvert; }; // This is the base class for all factories that are used by RocksDB to create // new MemTableRep objects class MemTableRepFactory : public Customizable { public: + using ConvertKind = MemTableRep::ConvertKind; + ~MemTableRepFactory() override {} static const char* Type() { return "MemTableRepFactory"; } @@ -342,8 +372,10 @@ class MemTableRepFactory : public Customizable { uint32_t /* column_family_id */) { return CreateMemTableRep(key_cmp, allocator, slice_transform, logger); } + // The DB supplies the final SST path for a file-backed memtable. + // The DB owns the backing file; the Rep must not delete it. virtual MemTableRep* CreateMemTableRep( - const std::string& /*level0_dir*/, + const std::string& /*memtable_file_path*/, const MutableCFOptions&, const MemTableRep::KeyComparator& key_cmp, Allocator* allocator, const SliceTransform* slice_transform, Logger* logger, @@ -363,6 +395,30 @@ class MemTableRepFactory : public Customizable { // false when if the already exists. // Default: false virtual bool CanHandleDuplicatedKey() const { return false; } + + bool SupportConvertToSST() const { + return convert_to_sst != ConvertKind::kDontConvert; + } + + // Return true if leftover mmap memtables created by this factory can be + // RO-loaded after a crash and ConvertToSST during Recover. + // Default: false + bool SupportCrashSafe() const { + return convert_to_sst == ConvertKind::kFileMmap; + } + + // Load leftover, cap visible entries, truncate, ConvertToSST. The caller + // supplies the visibility bound in meta->fd.largest_seqno; preserve it for + // subsequent table readers. + // Default: NotSupported. + virtual Status RecoverCrashSafeMemTableToSST( + const std::string& /*leftover_path*/, struct FileMetaData* /*meta*/, + const struct TableBuilderOptions&) { + return Status::NotSupported("RecoverCrashSafeMemTableToSST"); + } + + protected: + ConvertKind convert_to_sst = ConvertKind::kDontConvert; }; // This uses a skip list to store keys. It is the default. diff --git a/include/rocksdb/options.h b/include/rocksdb/options.h index eb3f17909c..0b6ffa2950 100644 --- a/include/rocksdb/options.h +++ b/include/rocksdb/options.h @@ -625,6 +625,14 @@ struct DBOptions { // on target_file_size_base and target_file_size_multiplier for level-based // compaction. For universal-style compaction, you can usually set it to -1. // + // Value 0 disables SST reader caching and eager SST opening during DB::Open(). + // Also set skip_stats_update_on_db_open=true to avoid opening SSTs for stats; + // reads still open SSTs on demand. + // Previously, values below 20 were silently raised to 20, so even 0 caused + // SSTs to be opened during DB::Open(). This defeated the user's intent to + // open without opening any SSTs and violated the principle of least surprise. + // Small nonnegative values are now honored instead of imposing that minimum. + // // A high value or -1 for this option can cause high memory usage. // See BlockBasedTableOptions::cache_usage_options to constrain // memory usage in case of block based table format. @@ -925,9 +933,30 @@ struct DBOptions { bool memtable_as_log_index = false; + // Enable crash-safe recovery using file mmap MemTables (CSPP / OffsetSkipList) + // and ConvertToSST to reduce WAL replay on the next open. + // Only process crashes (kill -9 / _exit) are covered, not power loss. + // DeleteRange is not covered. + // Reuse leftover MemTable files up to the published sequence, then replay + // the WAL tail. All CFs must support crash-safe file mmap MemTables. + // Unusable leftovers or cursor metadata, wal_filter, best_efforts_recovery, + // and unsupported sequence publication modes fall back to full WAL replay. + // Independent of memtable_as_log_index and Flush/Close ConvertToSST. + // Requires WAL: disableWAL is ignored; manual_wal_flush and WAL recycling + // are disabled, and wal_compression is forced to kNoCompression. + // Read-only Open disables this option and warns that full WAL replay may be + // very slow with the much larger MemTables encouraged in crash-safe mode. + // If a valid saved WAL kind differs from memtable_as_log_index, Open first + // recovers and flushes using the old kind, then opens in the requested kind. + // TransactionDB requires resolving recovered prepared transactions before + // switching kinds. + // Default: false. Not dynamically changeable through SetDBOptions(). + bool memtable_crash_safe_recover = false; + // If true, each WAL file is probed on DB open to auto-detect its on-disk - // format, so recovery works even when memtable_as_log_index was changed - // between runs. + // format. This selects the reader format, not the memtable format. Replaying + // a classic WAL directly into log-index memtables returns NotSupported; + // reopen with memtable_as_log_index=false and flush before switching. // // Defaults to false because the probe relies on CRC32 self-consistency // rather than a magic number to distinguish the two formats, which carries @@ -1254,8 +1283,15 @@ struct DBOptions { bool avoid_flush_during_recovery = false; // By default RocksDB will flush all memtables on DB close if there are - // unpersisted data (i.e. with WAL disabled) The flush can be skip to speedup - // DB close. Unpersisted data WILL BE LOST. + // unpersisted data (i.e. with WAL disabled) or all memtables support ConvertToSST. + // Upstream RocksDB skips flushing WAL-backed memtables on close: + // BuildTable is expensive and can cause a long wait. ConvertToSST is much + // cheaper, so when all memtables support it, we also flush (convert) them on close + // without that long wait and avoid WAL replay on the next open. + // Setting this option to true unconditionally skips all flushes (including + // conversion) during DB close. Data is lost only if there is unpersisted data + // written with WAL disabled; WAL-backed data is recovered by replaying WAL + // on the next open. Skipping conversion does not add a data-loss risk. // // DEFAULT: false // diff --git a/memtable/memtablerep_bench.cc b/memtable/memtablerep_bench.cc index 4f0c797b42..312ae1361c 100644 --- a/memtable/memtablerep_bench.cc +++ b/memtable/memtablerep_bench.cc @@ -48,7 +48,9 @@ DEFINE_string(benchmarks, "fillrandom", "\tfillrandom -- write N random values\n" "\tfillseq -- write N values in sequential order\n" "\treadrandom -- read N values in random order\n" + "\treaduniqrand -- read all keys in pre-shuffled order\n" "\treadseq -- scan the DB\n" + "\treadreverse -- scan the DB in reverse order\n" "\treadwrite -- 1 thread writes while N - 1 threads " "do random\n" "\t reads\n" @@ -68,6 +70,15 @@ DEFINE_string(memtablerep, "skiplist", "\t cspp:{\"mem_cap\":\"16G\"} or\n" "\t OffsetSkipList:{\"mem_cap\":\"16G\"}"); +// A file-backed memtable (cspp with convert_to_sst=kFileMmap in the json +// above) is created through the overload that takes the memtable file path, +// which is chroot_dir prefixed by the factory, so the json must set +// chroot_dir to an existing directory. LogRef modes are not supported: this +// bench has no WAL, so m_ref_to_wal stays kNoLogRef. +DEFINE_string(memtable_file_path, "", + "If non-empty, create the memtable rep through the file-backed " + "form, the path is relative to the json's chroot_dir"); + DEFINE_int64(bucket_count, 1000000, "bucket_count parameter to pass into NewHashSkiplistRepFactory or " "NewHashLinkListRepFactory"); @@ -94,6 +105,11 @@ DEFINE_bool(if_log_bucket_dist_when_flash, true, DEFINE_bool(enable_zero_copy, false, "enable zero copy"); +bool g_use_hint = false; +DEFINE_bool(use_concurrent_insert, false, + "Use Insert*Concurrently / Insert*WithHintConcurrently " + "(independent of whether hint is used)"); + DEFINE_bool(reverse, false, "readseq/scan in reverse order"); DEFINE_int32( @@ -134,6 +150,10 @@ bool g_is_topling_memtab = false; namespace ROCKSDB_NAMESPACE { namespace { +inline void EncodeBigEndian64(char* dst, uint64_t value) { + unaligned_save(dst, NativeOfBigEndian64(value)); +} + struct CallbackVerifyArgs { bool found; bool needs_user_key_cmp; @@ -199,6 +219,9 @@ class KeyGenerator { return std::numeric_limits::max(); } + uint64_t NextUniqRand(uint64_t& next) { return values_[next++]; } + bool IsUniqRand() const { return mode_ == UNIQUE_RANDOM; } + private: Random64* rand_; WriteMode mode_; @@ -248,11 +271,21 @@ class FillBenchmarkThread : public BenchmarkThread { auto internal_key_size = 16; uint64_t key = key_gen_->Next(); char key_buf[8]; // user key - EncodeFixed64(key_buf, key); + EncodeBigEndian64(key_buf, key); uint64_t tag = ++(*sequence_); Slice ukey(key_buf, sizeof(key_buf)); Slice value = generator_.Generate(FLAGS_item_size); - table_->InsertKeyValueConcurrently(tag, ukey, value); + if (g_use_hint) { + if (FLAGS_use_concurrent_insert) + table_->InsertKeyValueWithHintConcurrently(tag, ukey, value, &hint_); + else + table_->InsertKeyValueWithHint(tag, ukey, value, &hint_); + } else { + if (FLAGS_use_concurrent_insert) + table_->InsertKeyValueConcurrently(tag, ukey, value); + else + table_->InsertKeyValue(tag, ukey, value); + } *bytes_written_ += internal_key_size + FLAGS_item_size + 1; } else { @@ -268,7 +301,7 @@ class FillBenchmarkThread : public BenchmarkThread { assert(buf != nullptr); char* p = EncodeVarint32(buf, internal_key_size); auto key = key_gen_->Next(); - EncodeFixed64(p, key); + EncodeBigEndian64(p, key); p += 8; EncodeFixed64(p, ++(*sequence_)); p += 8; @@ -276,7 +309,17 @@ class FillBenchmarkThread : public BenchmarkThread { memcpy(p, bytes.data(), FLAGS_item_size); p += FLAGS_item_size; assert(p == buf + encoded_len); - table_->Insert(handle); + if (g_use_hint) { + if (FLAGS_use_concurrent_insert) + table_->InsertWithHintConcurrently(handle, &hint_); + else + table_->InsertWithHint(handle, &hint_); + } else { + if (FLAGS_use_concurrent_insert) + table_->InsertConcurrently(handle); + else + table_->Insert(handle); + } *bytes_written_ += encoded_len; } @@ -284,7 +327,18 @@ class FillBenchmarkThread : public BenchmarkThread { for (unsigned int i = 0; i < num_ops_; ++i) { FillOne(); } + // SkipList sequential InsertWithHint stores an arena splice; the + // default FinishHint is delete[] and is only valid for heap splices + // (concurrent InsertWithHint) or Topling token parking. + if (hint_ != nullptr && + (g_is_topling_memtab || FLAGS_use_concurrent_insert)) { + table_->FinishHint(hint_); + } + hint_ = nullptr; } + + protected: + void* hint_ = nullptr; }; class ConcurrentFillBenchmarkThread : public FillBenchmarkThread { @@ -313,13 +367,18 @@ class ConcurrentFillBenchmarkThread : public FillBenchmarkThread { class ReadBenchmarkThread : public BenchmarkThread { ReadOptions read_opt_; + uint64_t uniq_idx_; + bool is_uniqrand_; bool needs_user_key_cmp_; public: ReadBenchmarkThread(MemTableRep* table, KeyGenerator* key_gen, uint64_t* bytes_written, uint64_t* bytes_read, + uint64_t uniq_idx, uint64_t* sequence, uint64_t num_ops, uint64_t* read_hits) : BenchmarkThread(table, key_gen, bytes_written, bytes_read, sequence, num_ops, read_hits) { + uniq_idx_ = uniq_idx; + is_uniqrand_ = key_gen->IsUniqRand(); if (FLAGS_enable_zero_copy) { read_opt_.StartPin(); } @@ -347,10 +406,27 @@ class ReadBenchmarkThread : public BenchmarkThread { return false; } + static bool callbackPIK(void* arg, const MemTableRep::KeyValuePair&) { + auto self = static_cast(arg); + (*self->bytes_read_) += VarintLength(16) + 16 + FLAGS_item_size; + (*self->read_hits_)++; + return false; + } + void ReadOnePIK() { + char user_key[sizeof(uint64_t)]; + auto key = is_uniqrand_ ? key_gen_->NextUniqRand(uniq_idx_) : key_gen_->Next(); + EncodeBigEndian64(user_key, key); + ParsedInternalKey pik(Slice(user_key, sizeof(user_key)), *sequence_, kValueTypeForSeek); + table_->GetPIK(read_opt_, pik, this, callbackPIK); + } + void ReadOne() { + if (!needs_user_key_cmp_) { + return ReadOnePIK(); + } char user_key[sizeof(uint64_t)]; - auto key = key_gen_->Next(); - EncodeFixed64(user_key, key); + auto key = is_uniqrand_ ? key_gen_->NextUniqRand(uniq_idx_) : key_gen_->Next(); + EncodeBigEndian64(user_key, key); LookupKey lookup_key(Slice(user_key, sizeof(user_key)), *sequence_); InternalKeyComparator internal_key_comp(BytewiseComparator()); CallbackVerifyArgs verify_args; @@ -416,7 +492,7 @@ class ConcurrentReadBenchmarkThread : public ReadBenchmarkThread { uint64_t* sequence, uint64_t num_ops, uint64_t* read_hits, std::atomic_int* threads_done) - : ReadBenchmarkThread(table, key_gen, bytes_written, bytes_read, sequence, + : ReadBenchmarkThread(table, key_gen, bytes_written, bytes_read, 0, sequence, num_ops, read_hits) { threads_done_ = threads_done; } @@ -457,6 +533,8 @@ class SeqConcurrentReadBenchmarkThread : public SeqReadBenchmarkThread { class Benchmark { public: + double random_time = 0; + explicit Benchmark(MemTableRep* table, KeyGenerator* key_gen, uint64_t* sequence, uint32_t num_threads) : table_(table), @@ -474,6 +552,7 @@ class Benchmark { StopWatchNano timer(SystemClock::Default().get(), true); RunThreads(&threads, &bytes_written, &bytes_read, true, &read_hits); auto elapsed_time = static_cast(timer.ElapsedNanos() / 1000); + elapsed_time -= random_time; std::cout << "Elapsed time: " << static_cast(elapsed_time) << " us" << std::endl; @@ -539,9 +618,15 @@ class ReadBenchmark : public Benchmark { uint64_t* bytes_read, bool /*write*/, uint64_t* read_hits) override { for (int i = 0; i < FLAGS_num_threads; ++i) { + uint64_t uniq_idx = i * num_read_ops_per_thread_; + uint64_t num_ops = num_read_ops_per_thread_; + if (i + 1 == FLAGS_num_threads) { + num_ops = FLAGS_num_operations - uniq_idx; + } threads->emplace_back( ReadBenchmarkThread(table_, key_gen_, bytes_written, bytes_read, - sequence_, num_read_ops_per_thread_, read_hits)); + uniq_idx, + sequence_, num_ops, read_hits)); } for (auto& thread : *threads) { thread.join(); @@ -684,6 +769,13 @@ int main(int argc, char** argv) { uint64_t sequence; auto createMemtableRep = [&] { sequence = 0; + if (!FLAGS_memtable_file_path.empty()) { + ROCKSDB_NAMESPACE::MutableCFOptions mcfopt(options); + return factory->CreateMemTableRep(FLAGS_memtable_file_path, mcfopt, + key_comp, &arena, + options.prefix_extractor.get(), + options.info_log.get(), 0); + } return factory->CreateMemTableRep(key_comp, &arena, options.prefix_extractor.get(), options.info_log.get()); @@ -702,6 +794,7 @@ int main(int argc, char** argv) { name = ROCKSDB_NAMESPACE::Slice(benchmarks, sep - benchmarks); benchmarks = sep + 1; } + g_use_hint = false; std::unique_ptr benchmark; if (name == ROCKSDB_NAMESPACE::Slice("fillseq")) { memtablerep.reset(createMemtableRep()); @@ -709,6 +802,7 @@ int main(int argc, char** argv) { &rng, ROCKSDB_NAMESPACE::SEQUENTIAL, FLAGS_num_operations)); benchmark.reset(new ROCKSDB_NAMESPACE::FillBenchmark( memtablerep.get(), key_gen.get(), &sequence)); + g_use_hint = true; } else if (name == ROCKSDB_NAMESPACE::Slice("fillrandom")) { memtablerep.reset(createMemtableRep()); key_gen.reset(new ROCKSDB_NAMESPACE::KeyGenerator( @@ -720,7 +814,14 @@ int main(int argc, char** argv) { &rng, ROCKSDB_NAMESPACE::RANDOM, FLAGS_num_operations)); benchmark.reset(new ROCKSDB_NAMESPACE::ReadBenchmark( memtablerep.get(), key_gen.get(), &sequence)); - } else if (name == ROCKSDB_NAMESPACE::Slice("readseq")) { + } else if (name == ROCKSDB_NAMESPACE::Slice("readuniqrand")) { + FLAGS_seed++; + key_gen.reset(new ROCKSDB_NAMESPACE::KeyGenerator( + &rng, ROCKSDB_NAMESPACE::UNIQUE_RANDOM, FLAGS_num_operations)); + benchmark.reset(new ROCKSDB_NAMESPACE::ReadBenchmark( + memtablerep.get(), key_gen.get(), &sequence)); + } else if (name == ROCKSDB_NAMESPACE::Slice("readseq") || + name == ROCKSDB_NAMESPACE::Slice("readreverse")) { key_gen.reset(new ROCKSDB_NAMESPACE::KeyGenerator( &rng, ROCKSDB_NAMESPACE::SEQUENTIAL, FLAGS_num_operations)); benchmark.reset(new ROCKSDB_NAMESPACE::SeqReadBenchmark(memtablerep.get(), @@ -745,7 +846,29 @@ int main(int argc, char** argv) { continue; } std::cout << "Running " << name.ToString() << std::endl; + if (name == ROCKSDB_NAMESPACE::Slice("readrandom")) { + auto num_ops = FLAGS_num_operations / FLAGS_num_threads; + volatile uint64_t sink = 0; + ROCKSDB_NAMESPACE::StopWatchNano timer(ROCKSDB_NAMESPACE::SystemClock::Default().get(), true); + for (int i = 0; i < num_ops; i++) { + sink = key_gen->Next(); + } + benchmark->random_time = static_cast(timer.ElapsedNanos() / 1000); + std::cout << "Random generation time: " << benchmark->random_time << " us" << std::endl; + std::cout << "Random generation sink: " << sink << std::endl; + } + const bool saved_reverse = FLAGS_reverse; + FLAGS_reverse = saved_reverse || name == ROCKSDB_NAMESPACE::Slice("readreverse"); benchmark->Run(); + FLAGS_reverse = saved_reverse; + // approximate memory of the rep itself (arena/mmap), i.e. the working + // set the benchmark just walked + size_t usage = memtablerep->ApproximateMemoryUsage(); + std::cout << "ApproximateMemoryUsage: " << usage << " bytes (" + << usage / 1048576.0 << " MiB)" + << ", arena: " << arena.ApproximateMemoryUsage() << " bytes (" + << arena.ApproximateMemoryUsage() / 1048576.0 << " MiB)" + << std::endl; } return 0; diff --git a/options/db_options.cc b/options/db_options.cc index dc17320e4e..7c14532c89 100644 --- a/options/db_options.cc +++ b/options/db_options.cc @@ -328,6 +328,10 @@ static std::unordered_map {offsetof(struct ImmutableDBOptions, memtable_as_log_index), OptionType::kBoolean, OptionVerificationType::kNormal, OptionTypeFlags::kNone}}, + {"memtable_crash_safe_recover", + {offsetof(struct ImmutableDBOptions, memtable_crash_safe_recover), + OptionType::kBoolean, OptionVerificationType::kNormal, + OptionTypeFlags::kNone}}, {"check_wal_format", {offsetof(struct ImmutableDBOptions, check_wal_format), OptionType::kBoolean, OptionVerificationType::kNormal, @@ -768,6 +772,7 @@ ImmutableDBOptions::ImmutableDBOptions(const DBOptions& options) avoid_unnecessary_blocking_io(options.avoid_unnecessary_blocking_io), persist_stats_to_disk(options.persist_stats_to_disk), memtable_as_log_index(options.memtable_as_log_index), + memtable_crash_safe_recover(options.memtable_crash_safe_recover), check_wal_format(options.check_wal_format), write_dbid_to_manifest(options.write_dbid_to_manifest), log_readahead_size(options.log_readahead_size), @@ -938,6 +943,8 @@ void ImmutableDBOptions::Dump(Logger* log) const { persist_stats_to_disk); ROCKS_LOG_HEADER(log, " Options.memtable_as_log_index: %u", memtable_as_log_index); + ROCKS_LOG_HEADER(log, " Options.memtable_crash_safe_recover: %u", + memtable_crash_safe_recover); ROCKS_LOG_HEADER(log, " Options.check_wal_format: %u", check_wal_format); ROCKS_LOG_HEADER(log, " Options.write_dbid_to_manifest: %d", diff --git a/options/db_options.h b/options/db_options.h index e0c618d637..596d430084 100644 --- a/options/db_options.h +++ b/options/db_options.h @@ -90,6 +90,7 @@ struct ImmutableDBOptions { bool avoid_unnecessary_blocking_io; bool persist_stats_to_disk; bool memtable_as_log_index; + bool memtable_crash_safe_recover; bool check_wal_format; bool write_dbid_to_manifest; size_t log_readahead_size; diff --git a/options/options_helper.cc b/options/options_helper.cc index 61cce16d7e..623aca5fc0 100644 --- a/options/options_helper.cc +++ b/options/options_helper.cc @@ -118,6 +118,7 @@ DBOptions BuildDBOptions(const ImmutableDBOptions& immutable_db_options, mutable_db_options.stats_persist_period_sec; options.persist_stats_to_disk = immutable_db_options.persist_stats_to_disk; options.memtable_as_log_index = immutable_db_options.memtable_as_log_index; + options.memtable_crash_safe_recover = immutable_db_options.memtable_crash_safe_recover; options.stats_history_buffer_size = mutable_db_options.stats_history_buffer_size; options.advise_random_on_open = immutable_db_options.advise_random_on_open; diff --git a/options/options_settable_test.cc b/options/options_settable_test.cc index 2cd675a16a..6bf11d5f94 100644 --- a/options/options_settable_test.cc +++ b/options/options_settable_test.cc @@ -338,6 +338,7 @@ TEST_F(OptionsSettableTest, DBOptionsAllFieldsSettable) { "stats_persist_period_sec=54321;" "persist_stats_to_disk=true;" "memtable_as_log_index=true;" + "memtable_crash_safe_recover=true;" "check_wal_format=true;" "stats_history_buffer_size=14159;" "allow_fallocate=true;" diff --git a/options/options_test.cc b/options/options_test.cc index 450f38c400..35171f73b8 100644 --- a/options/options_test.cc +++ b/options/options_test.cc @@ -170,6 +170,7 @@ TEST_F(OptionsTest, GetOptionsFromMapTest) { {"stats_persist_period_sec", "57"}, {"persist_stats_to_disk", "false"}, {"memtable_as_log_index", "false"}, + {"memtable_crash_safe_recover", "false"}, {"check_wal_format", "false"}, {"stats_history_buffer_size", "69"}, {"advise_random_on_open", "true"}, @@ -355,6 +356,7 @@ TEST_F(OptionsTest, GetOptionsFromMapTest) { ASSERT_EQ(new_db_opt.stats_persist_period_sec, 57U); ASSERT_EQ(new_db_opt.persist_stats_to_disk, false); ASSERT_EQ(new_db_opt.memtable_as_log_index, false); + ASSERT_EQ(new_db_opt.memtable_crash_safe_recover, false); ASSERT_EQ(new_db_opt.check_wal_format, false); ASSERT_EQ(new_db_opt.stats_history_buffer_size, 69U); ASSERT_EQ(new_db_opt.advise_random_on_open, true); @@ -2394,6 +2396,7 @@ TEST_F(OptionsOldApiTest, GetOptionsFromMapTest) { {"stats_persist_period_sec", "57"}, {"persist_stats_to_disk", "false"}, {"memtable_as_log_index", "false"}, + {"memtable_crash_safe_recover", "false"}, {"check_wal_format", "false"}, {"stats_history_buffer_size", "69"}, {"advise_random_on_open", "true"}, @@ -2581,6 +2584,7 @@ TEST_F(OptionsOldApiTest, GetOptionsFromMapTest) { ASSERT_EQ(new_db_opt.stats_persist_period_sec, 57U); ASSERT_EQ(new_db_opt.persist_stats_to_disk, false); ASSERT_EQ(new_db_opt.memtable_as_log_index, false); + ASSERT_EQ(new_db_opt.memtable_crash_safe_recover, false); ASSERT_EQ(new_db_opt.check_wal_format, false); ASSERT_EQ(new_db_opt.stats_history_buffer_size, 69U); ASSERT_EQ(new_db_opt.advise_random_on_open, true); diff --git a/port/port_posix.h b/port/port_posix.h index ea0259a5ca..31e61d484f 100644 --- a/port/port_posix.h +++ b/port/port_posix.h @@ -24,6 +24,33 @@ #define __declspec(S) +// Force-inline a function whose body lives in another translation unit, and +// hide it, so that LTO may inline it into its callers. Both halves are needed +// and both have to be gated: +// +// * `visibility("hidden")` clears the semantic-interposition rule. Without +// it GCC refuses an inlining mandate outright -- "error: inlining failed +// in call to 'always_inline': function body can be overwritten at link +// time". Hidden on its own is only permission, not an order: measured to +// leave the call out of line, worth ~1 instruction. +// * `always_inline` is the order. It must be absent when the body is not +// reachable, so this macro is empty outside LTO -- without -flto GCC fails +// the build with "function body not available" rather than quietly not +// inlining. TOPLINGDB_HAVE_LTO is set by the Makefile under USE_LTO=1. +// * Unit-test builds are excluded. They construct such objects directly and +// link the shared library, so the symbol has to stay exported; +// MAKE_UNIT_TEST=1 defines ROCKSDB_UNIT_TEST for the whole test build, +// library objects included. +// +// `flatten` is not an alternative: it only flattens calls whose body is +// visible in the calling TU. +#if defined(TOPLINGDB_HAVE_LTO) && !defined(ROCKSDB_UNIT_TEST) +#define TOPLING_LTO_HIDDEN_INLINE \ + __attribute__((visibility("hidden"))) __attribute__((always_inline)) +#else +#define TOPLING_LTO_HIDDEN_INLINE +#endif + #undef PLATFORM_IS_LITTLE_ENDIAN #if defined(OS_MACOSX) #include diff --git a/port/win/port_win.h b/port/win/port_win.h index 387c08122f..7d1cb5425a 100644 --- a/port/win/port_win.h +++ b/port/win/port_win.h @@ -57,6 +57,11 @@ using ssize_t = SSIZE_T; #define ROCKSDB_PRIszt "Iu" #endif +// No LTO force-inlining here: the build never sets TOPLINGDB_HAVE_LTO on this +// platform, and `visibility("hidden")` has no meaning in PE/COFF. See +// port/port_posix.h for the POSIX definition and the reasoning. +#define TOPLING_LTO_HIDDEN_INLINE + #ifdef _MSC_VER #define __attribute__(A) diff --git a/sideplugin/rockside b/sideplugin/rockside index c308ced761..972f48eb7d 160000 --- a/sideplugin/rockside +++ b/sideplugin/rockside @@ -1 +1 @@ -Subproject commit c308ced7614708b9aa2635066d1527346e0e69c8 +Subproject commit 972f48eb7df345f9c5b46d593fa53197436eccee diff --git a/src.mk b/src.mk index bf8959d039..141e953c32 100644 --- a/src.mk +++ b/src.mk @@ -483,9 +483,11 @@ TEST_MAIN_SOURCES = \ db/db_compaction_filter_test.cc \ db/db_compaction_test.cc \ db/db_clip_test.cc \ + db/db_cspp_crash_safe_test.cc \ db/db_dynamic_level_test.cc \ db/db_encryption_test.cc \ db/db_flush_test.cc \ + db/db_memtable_convert_test.cc \ db/db_readonly_with_timestamp_test.cc \ db/db_with_timestamp_basic_test.cc \ db/import_column_family_test.cc \ diff --git a/table/get_context.h b/table/get_context.h index dbea262d6f..1c1de41e1a 100644 --- a/table/get_context.h +++ b/table/get_context.h @@ -101,6 +101,10 @@ class GetContext { // and false if all the merge operands associated with user_key has to be // returned. Id do_merge=false then all the merge operands are stored in // merge_context and they are never merged. The value pointer is untouched. + // Constructing this is a fixed per-Get cost. See TOPLING_LTO_HIDDEN_INLINE + // (port/port.h): the inlining mandate needs hiding to be legal, and both + // are absent outside LTO and in unit-test builds. + TOPLING_LTO_HIDDEN_INLINE GetContext(const Comparator* ucmp, const MergeOperator* merge_operator, Logger* logger, Statistics* statistics, GetState init_state, const Slice& user_key, PinnableSlice* value, @@ -111,6 +115,10 @@ class GetContext { PinnedIteratorsManager* _pinned_iters_mgr = nullptr, ReadCallback* callback = nullptr, bool* is_blob_index = nullptr, uint64_t tracing_get_id = 0, BlobFetcher* blob_fetcher = nullptr); + // Constructing this is a fixed per-Get cost. See TOPLING_LTO_HIDDEN_INLINE + // (port/port.h): the inlining mandate needs hiding to be legal, and both + // are absent outside LTO and in unit-test builds. + TOPLING_LTO_HIDDEN_INLINE GetContext(const Comparator* ucmp, const MergeOperator* merge_operator, Logger* logger, Statistics* statistics, GetState init_state, const Slice& user_key, PinnableSlice* value, diff --git a/table/table_reader.h b/table/table_reader.h index 8db7ade1d9..3884a2e130 100644 --- a/table/table_reader.h +++ b/table/table_reader.h @@ -111,6 +111,9 @@ class TableReader : public CacheAlignedNewDelete { virtual std::shared_ptr GetTableProperties() const = 0; + // Whether num_entries is exact for the table's visible contents. + virtual bool IsNumEntriesExact() const { return true; } + // Prepare work that can be done before the real Get() virtual void Prepare(const Slice& /*target*/) {} virtual void PreparePIK(const ParsedInternalKey& pik) { diff --git a/tools/crash_recover_bench.cc b/tools/crash_recover_bench.cc new file mode 100644 index 0000000000..7da18c40cf --- /dev/null +++ b/tools/crash_recover_bench.cc @@ -0,0 +1,281 @@ +// Copyright (c) 2026-present, Topling Inc. +// Abnormal-exit recovery benchmark. +// Same RocksDB API. Topling options come from TOPLINGDB_EASY_MIGRATE_CONF. +// +// Fill one memtable, _exit without Close, copy that directory aside, then +// time DB::Open on a fresh copy of the backup. Repeat and print the average. + +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "rocksdb/db.h" +#include "rocksdb/options.h" + +namespace fs = std::filesystem; +using ROCKSDB_NAMESPACE::DB; +using ROCKSDB_NAMESPACE::Options; +using ROCKSDB_NAMESPACE::ReadOptions; +using ROCKSDB_NAMESPACE::Slice; +using ROCKSDB_NAMESPACE::Status; +using ROCKSDB_NAMESPACE::WriteOptions; + +namespace { + +constexpr size_t kWriteBufferBytes = 2ull << 30; +constexpr int kFillExit = 42; + +struct Args { + std::string root = "/tmp/crash_recover_bench"; + int runs = 5; + int key_size = 16; + int value_size = 128; + // Stop filling once the active memtable reaches this fraction of 2GB, + // so the engine does not flush it before the abnormal exit. + double fill_frac = 0.75; + bool reuse_backup = false; +}; + +void Die(const std::string& msg) { + fprintf(stderr, "crash_recover_bench: %s\n", msg.c_str()); + std::_Exit(1); +} + +Args ParseArgs(int argc, char** argv) { + Args a; + for (int i = 1; i < argc; i++) { + std::string s = argv[i]; + auto need = [&](const char* name) -> std::string { + auto eq = s.find('='); + if (eq == std::string::npos || s.substr(0, eq) != name) { + return ""; + } + return s.substr(eq + 1); + }; + if (auto v = need("--root"); !v.empty()) { + a.root = v; + } else if (auto v = need("--runs"); !v.empty()) { + a.runs = std::atoi(v.c_str()); + } else if (auto v = need("--key_size"); !v.empty()) { + a.key_size = std::atoi(v.c_str()); + } else if (auto v = need("--value_size"); !v.empty()) { + a.value_size = std::atoi(v.c_str()); + } else if (auto v = need("--fill_frac"); !v.empty()) { + a.fill_frac = std::atof(v.c_str()); + } else if (s == "--reuse_backup") { + a.reuse_backup = true; + } else if (s == "--help") { + fprintf(stderr, + "Usage: crash_recover_bench [--root=DIR] [--runs=N]\n" + " [--key_size=N] [--value_size=N] [--fill_frac=0.75]\n" + " [--reuse_backup]\n" + "ToplingDB: TOPLINGDB_EASY_MIGRATE_CONF=tools/crash_recover_bench.yaml\n" + "RocksDB: leave that variable unset\n" + "write_buffer_size is 2GB either way.\n"); + std::exit(0); + } else { + Die("unknown arg " + s); + } + } + if (a.runs < 1 || a.key_size < 1 || a.value_size < 1 || a.fill_frac <= 0 || + a.fill_frac >= 1) { + Die("bad --runs/--key_size/--value_size/--fill_frac"); + } + return a; +} + +Options MakeOptions() { + Options opt; + opt.create_if_missing = true; + opt.write_buffer_size = kWriteBufferBytes; + opt.max_write_buffer_number = 4; + opt.disable_auto_compactions = true; + opt.level0_file_num_compaction_trigger = 1 << 20; + opt.avoid_flush_during_recovery = true; + return opt; +} + +std::string KeyAt(int key_size, uint64_t i) { + std::string k(static_cast(key_size), '0'); + for (int p = key_size - 1; p >= 0 && i > 0; --p) { + k[static_cast(p)] = static_cast('0' + (i % 10)); + i /= 10; + } + return k; +} + +uint64_t ActiveMemBytes(DB* db) { + std::string v; + if (!db->GetProperty("rocksdb.cur-size-active-mem-table", &v)) { + return 0; + } + return std::strtoull(v.c_str(), nullptr, 10); +} + +// Child only. Writes until the live memtable reaches the target, then _exit +// without Close so the directory is a process-crash image. +void FillAndAbort(const Args& args, const fs::path& dbpath, + const fs::path& nkeys_path) { + fs::remove_all(dbpath); + fs::create_directories(dbpath); + DB* db = nullptr; + Status s = DB::Open(MakeOptions(), dbpath.string(), &db); + if (!s.ok()) { + Die("fill open: " + s.ToString()); + } + const uint64_t target = + static_cast(kWriteBufferBytes * args.fill_frac); + WriteOptions wo; + wo.disableWAL = false; + std::string value(static_cast(args.value_size), 'v'); + uint64_t n = 0; + uint64_t mem = 0; + while (mem < target) { + std::string key = KeyAt(args.key_size, n); + s = db->Put(wo, key, value); + if (!s.ok()) { + Die("put: " + s.ToString()); + } + n++; + if ((n & 1023) == 0) { + mem = ActiveMemBytes(db); + } + } + fprintf(stderr, "filled keys=%llu active_mem=%llu target=%llu\n", + static_cast(n), + static_cast(mem), + static_cast(target)); + FILE* nf = fopen(nkeys_path.c_str(), "w"); + if (nf == nullptr || fprintf(nf, "%llu\n", static_cast(n)) < 0) { + Die("write nkeys"); + } + fclose(nf); + // Leave the DB open. _exit skips destructors, same as kill -9. + std::_Exit(kFillExit); +} + +void CopyTree(const fs::path& src, const fs::path& dst) { + std::error_code ec; + fs::remove_all(dst, ec); + fs::create_directories(dst.parent_path(), ec); + if (ec) { + Die("mkdir " + dst.parent_path().string() + ": " + ec.message()); + } + // Keep holes. A plain copy fills the memtab tail and makes ftruncate drop + // real pages. + const pid_t child = ::fork(); + if (child < 0) { + Die("fork cp: " + std::string(std::strerror(errno))); + } + if (child == 0) { + ::execlp("cp", "cp", "-a", "--sparse=always", "--", src.c_str(), + dst.c_str(), static_cast(nullptr)); + std::_Exit(127); + } + int status = 0; + pid_t waited; + do { + waited = ::waitpid(child, &status, 0); + } while (waited < 0 && errno == EINTR); + if (waited < 0 || !WIFEXITED(status) || WEXITSTATUS(status) != 0) { + Die("cp --sparse=always " + src.string() + " -> " + dst.string()); + } +} + +double OpenMillis(const fs::path& dbpath, const Args& args, uint64_t nkeys) { + DB* db = nullptr; + auto t0 = std::chrono::steady_clock::now(); + Status s = DB::Open(MakeOptions(), dbpath.string(), &db); + auto t1 = std::chrono::steady_clock::now(); + if (!s.ok()) { + Die("recover open: " + s.ToString()); + } + std::string got; + s = db->Get(ReadOptions(), KeyAt(args.key_size, 0), &got); + if (!s.ok() || got.size() != static_cast(args.value_size)) { + Die("spot check key 0: " + s.ToString()); + } + if (nkeys > 1) { + s = db->Get(ReadOptions(), KeyAt(args.key_size, nkeys - 1), &got); + if (!s.ok()) { + Die("spot check last key: " + s.ToString()); + } + } + s = db->Close(); + delete db; + if (!s.ok()) { + Die("close: " + s.ToString()); + } + return std::chrono::duration(t1 - t0).count(); +} + +uint64_t ReadNKeys(const fs::path& path) { + FILE* nf = fopen(path.c_str(), "r"); + if (nf == nullptr) { + Die("open " + path.string()); + } + unsigned long long n = 0; + if (fscanf(nf, "%llu", &n) != 1 || n == 0) { + fclose(nf); + Die("bad nkeys file"); + } + fclose(nf); + return n; +} + +} // namespace + +int main(int argc, char** argv) { + setenv("ROCKSDB_KICK_OUT_OPTIONS_FILE", "1", 1); + Args args = ParseArgs(argc, argv); + const fs::path root = args.root; + const fs::path crashed = root / "crashed"; + const fs::path backup = root / "backup"; + const fs::path nkeys_path = root / "nkeys.txt"; + fs::create_directories(root); + + const bool have_backup = args.reuse_backup && fs::exists(backup / "CURRENT"); + if (!have_backup) { + const pid_t pid = ::fork(); + if (pid < 0) { + Die("fork"); + } + if (pid == 0) { + FillAndAbort(args, crashed, nkeys_path); + } + int st = 0; + if (::waitpid(pid, &st, 0) != pid || !WIFEXITED(st) || + WEXITSTATUS(st) != kFillExit) { + Die("fill child failed"); + } + CopyTree(crashed, backup); + fprintf(stderr, "backup %s\n", backup.c_str()); + } + + const uint64_t nkeys = ReadNKeys(nkeys_path); + fprintf(stderr, "keys=%llu runs=%d write_buffer=%zuMB\n", + static_cast(nkeys), args.runs, + kWriteBufferBytes >> 20); + + double sum = 0; + for (int i = 0; i < args.runs; i++) { + const fs::path run = root / ("run-" + std::to_string(i)); + CopyTree(backup, run); + const double ms = OpenMillis(run, args, nkeys); + sum += ms; + printf("run %d open_ms %.3f\n", i, ms); + fflush(stdout); + fs::remove_all(run); + } + printf("avg_open_ms %.3f runs %d\n", sum / args.runs, args.runs); + return 0; +} diff --git a/tools/crash_recover_bench.md b/tools/crash_recover_bench.md new file mode 100644 index 0000000000..3c89ec190c --- /dev/null +++ b/tools/crash_recover_bench.md @@ -0,0 +1,30 @@ +# Abnormal-exit recovery: DB::Open + +For the recovery mechanism, configuration, and limitations, see the [Chinese crash-safe recovery guide](https://github.com/topling/rockside/wiki/Crash-Safe-Recovery). + +`crash_recover_bench` fills one memtable to 75% of a 2GiB (2,147,483,648-byte) +`write_buffer_size`, targeting 1.5GiB (1,610,612,736 bytes), then `_exit`s +without `Close`. Keys are 16 bytes and values are 128 bytes. Each run copies that +directory with `cp -a --sparse=always` and times only `DB::Open`. A Get of +key 0 and the last key runs after the timer. Default is 5 runs. + +Measured on 2026-10-02, directories under `/dev/shm`. + +CSPP crash-safe sets `TOPLINGDB_EASY_MIGRATE_CONF=tools/crash_recover_bench.yaml`. +Default SkipList / WAL replay leaves that variable unset, so it does not read +the yaml. Both configurations use the same benchmark binary and the same +ToplingDB build (`librocksdb.so.8.10.2`); the control is not an independently +built upstream RocksDB binary. +`write_buffer_size` is 2GiB either way. CSPP reached 9,664,512 keys and +1,610,644,224 active-memtable bytes; default SkipList reached 9,090,048 keys +and 1,610,614,784 bytes. The active memtable stops at the same 75% +byte target; the key counts differ. + +| recovery path | five runs (ms) | avg (ms) | ratio | +| --- | --- | ---: | ---: | +| CSPP crash-safe | 6.545, 4.692, 3.820, 4.768, 5.120 | 4.989 | 1.0 | +| Default SkipList / WAL replay | 6124.357, 5849.884, 5984.324, 5943.391, 6033.802 | 5987.152 | 1200.1 | + +Both groups exited with status 0. The copied images contained no SST and had +not flushed early. Separate LOG checks confirmed CSPP leftover conversion +and WAL-tail recovery without fallback, and full WAL recovery for default SkipList. diff --git a/tools/crash_recover_bench.yaml b/tools/crash_recover_bench.yaml new file mode 100644 index 0000000000..9a542ac2d5 --- /dev/null +++ b/tools/crash_recover_bench.yaml @@ -0,0 +1,45 @@ +# Easy-migrate config for crash_recover_bench. +# Point TOPLINGDB_EASY_MIGRATE_CONF at this file. Unset it for plain RocksDB. +http: + auto_start_http: false +setenv: + ROCKSDB_KICK_OUT_OPTIONS_FILE: + overwrite: true + value: "1" + TOPLINGDB_WARMUP_PROVIDER: + overwrite: true + value: "willneed" +MemTableRepFactory: + cspp: + class: cspp + params: + mem_cap: 1G + convert_to_sst: kFileMmap + sync_sst_file: false +TableFactory: + cspp_memtab_sst: + class: CSPPMemTabTable + params: + populate_read: false + dispatch: + class: DispatcherTable + params: + default: cspp_memtab_sst + readers: + CSPPMemTabTable: cspp_memtab_sst +CFOptions: + default: + write_buffer_size: 2G + max_write_buffer_number: 4 + memtable_factory: "${cspp}" + table_factory: dispatch + disable_auto_compactions: true + level0_file_num_compaction_trigger: 1048576 +DBOptions: + default: + create_if_missing: true + memtable_crash_safe_recover: true + memtable_as_log_index: true + avoid_flush_during_recovery: true + allow_fdatasync: false + max_background_jobs: 2 diff --git a/tools/db_bench_tool.cc b/tools/db_bench_tool.cc index 9970dc08bc..ccb50267b1 100644 --- a/tools/db_bench_tool.cc +++ b/tools/db_bench_tool.cc @@ -5191,9 +5191,30 @@ class Benchmark { DB* SelectDB(ThreadState* thread) { return SelectDBWithCfh(thread)->db; } DBWithColumnFamilies* SelectDBWithCfh(ThreadState* thread) { + // A single-DB run always resolves to db_, so the draw below would only + // advance the per-thread RNG stream (and add an mt19937_64 step plus a + // modulo to every read) for nothing. The key generator draws from the same + // stream, so this does change which keys a run reads -- that is the point: + // the benchmark exists to measure the DB, not its own bookkeeping. + if (LIKELY(db_.db != nullptr)) { + return &db_; + } return SelectDBWithCfh(thread->rand.Next()); } + // The default column family handle never changes for a given DB, but asking + // for it is a virtual call (and, in the shared library, goes through the + // PLT). Per-operation loops should not pay for that on every key. + ColumnFamilyHandle* DefaultCfh(DBWithColumnFamilies* d) { + if (UNLIKELY(d != cfh_cache_db_)) { + cfh_cache_db_ = d; + cfh_cache_ = d->db->DefaultColumnFamily(); + } + return cfh_cache_; + } + DBWithColumnFamilies* cfh_cache_db_ = nullptr; + ColumnFamilyHandle* cfh_cache_ = nullptr; + DBWithColumnFamilies* SelectDBWithCfh(uint64_t rand_int) { if (db_.db != nullptr) { return &db_; @@ -6308,7 +6329,7 @@ class Benchmark { if (FLAGS_num_column_families > 1) { cfh = db_with_cfh->GetCfh(key_rand); } else { - cfh = db_with_cfh->db->DefaultColumnFamily(); + cfh = DefaultCfh(db_with_cfh); } if (read_operands_) { for (size_t i = 0; i < pinnable_vals.size(); ++i) { diff --git a/utilities/transactions/lock/point/point_lock_manager.cc b/utilities/transactions/lock/point/point_lock_manager.cc index 03845958c3..43ac473897 100644 --- a/utilities/transactions/lock/point/point_lock_manager.cc +++ b/utilities/transactions/lock/point/point_lock_manager.cc @@ -309,7 +309,7 @@ PointLockManager::PointLockManager(PessimisticTransactionDB* txn_db, terark_forceinline size_t LockMap::GetStripe(const LockString& key, size_t hash) const { assert(num_stripes_ > 0); - auto col = hash % num_stripes_; + auto col = FastRange64(hash, num_stripes_); if (1 == super_stripes_) { return col; } else { @@ -317,7 +317,7 @@ size_t LockMap::GetStripe(const LockString& key, size_t hash) const { size_t plen = std::min(size_t(key_prefix_len_), key.size()); ROCKSDB_ASSUME(plen <= sizeof(pref)); memcpy(&pref, key.data(), plen); - size_t row = pref % super_stripes_; + size_t row = FastRange64(pref, super_stripes_); return row * num_stripes_ + col; } } diff --git a/utilities/transactions/pessimistic_transaction_db.cc b/utilities/transactions/pessimistic_transaction_db.cc index 682ccd84f8..ec6896fb89 100644 --- a/utilities/transactions/pessimistic_transaction_db.cc +++ b/utilities/transactions/pessimistic_transaction_db.cc @@ -306,8 +306,16 @@ Status TransactionDB::Open( const bool use_batch_per_txn = txn_db_options.write_policy == WRITE_COMMITTED || txn_db_options.write_policy == WRITE_PREPARED; + MaybeOptionsUpdateFrom(&db_options_2pc, &column_families_copy, dbname); + s = DBImpl::PrepareCrashSafeKindForOpen(db_options_2pc, dbname, + column_families_copy, + use_seq_per_batch, use_batch_per_txn); + if (!s.ok()) { + return s; + } s = DBImpl::Open(db_options_2pc, dbname, column_families_copy, handles, &db, - use_seq_per_batch, use_batch_per_txn); + use_seq_per_batch, use_batch_per_txn, + true /* options_already_updated */); if (s.ok()) { ROCKS_LOG_WARN(db->GetDBOptions().info_log, "Transaction write_policy is %s", diff --git a/utilities/write_batch_with_index/write_batch_with_index_internal.cc b/utilities/write_batch_with_index/write_batch_with_index_internal.cc index dedc4b186a..72411b8f46 100644 --- a/utilities/write_batch_with_index/write_batch_with_index_internal.cc +++ b/utilities/write_batch_with_index/write_batch_with_index_internal.cc @@ -293,7 +293,6 @@ void BaseDeltaIterator::AssertInvariants() { #endif } -ROCKSDB_FLATTEN void BaseDeltaIterator::Advance(bool const_forward) { if (UNLIKELY(equal_keys_)) { assert(BaseValid() && DeltaValid()); @@ -498,7 +497,6 @@ struct BDI_VirtualCmpNoTS { const Comparator* cmp; }; -ROCKSDB_FLATTEN void BaseDeltaIterator::UpdateCurrent(bool const_forward) { if (0 == opt_cmp_type_) UpdateCurrentTpl(const_forward, BDI_BytewiseCmpNoTS());