Repository navigation
Expand file tree
/
Copy path.env.example
More file actions
44 lines (38 loc) · 2.29 KB
/
Copy path.env.example
File metadata and controls
44 lines (38 loc) · 2.29 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
# AutoData environment variables — copy to .env and source before running.
# --- Local nanochat checkout (must be a clone of karpathy/nanochat) ---
export AUTODATA_NANOCHAT_ROOT="${HOME}/nanochat"
# --- Local corpus + features (ClimbMix pre-processing) ---
# Raw ClimbMix parquet shards (shard_00000.parquet ... shard_06542.parquet)
export AUTODATA_POOL_DIR="${HOME}/.cache/autodata/climbmix_pool"
# Held-out validation shard (copy from the pool; we use shard_06542)
export AUTODATA_VAL_SHARD="${AUTODATA_POOL_DIR}/shard_06542.parquet"
# Pre-computed per-doc features (.npy files; see DATASET_CARD.md)
# doc_tokens.npy, doc_chars.npy, distinct_{1..5}gram_bpe.npy,
# avg_distinct_ngram_bpe.npy, logppl_qwen.npy, topic_id.npy, format_id.npy
# Download the published bank (~22.7 GiB of annotations):
# hf download WecoAI/autodata-climbmix-features --repo-type dataset \
# --local-dir "$AUTODATA_FEATURES_DIR"
export AUTODATA_FEATURES_DIR="${HOME}/.cache/autodata/meta_climbmix_full"
# --- Local target training (pipeline/build_and_launch.py) ---
# Comma-separated CUDA device IDs. Reduce the per-device batch size with the
# launcher's --device-batch-size flag if your GPUs have less memory than H100s.
export AUTODATA_TRAIN_GPUS="0,1,2,3,4,5,6,7"
export AUTODATA_LOCAL_RUNS_DIR="results/local_runs"
# Materialized selections can require ~25 GiB. Point this at a fast local disk
# if /dev/shm is too small.
export AUTODATA_STAGE_DIR="/dev/shm/autodata_stage"
# --- Optional Modal target training backend ---
# Volume names — created via `modal volume create <name>` once
export AUTODATA_SCRATCH_VOL="autodata-scratch"
export AUTODATA_ARCHIVE_VOL="autodata-archive"
# Modal app name (matches pipeline/modal_train.py)
export AUTODATA_MODAL_APP="autodata-train"
# --- WECO config ---
# Set BEFORE invoking launchers; the WECO CLI reads its own API key from
# ~/.weco_credentials (run `weco login` once)
export AUTODATA_WECO_RUN_PREFIX="autodata_run"
export AUTODATA_WECO_GPUS="0,1" # comma-separated GPU ids for d8 proxy training
# --- LLM provider keys (one of these per WECO run, matching `--model`) ---
# export OPENAI_API_KEY=sk-... # for --model gpt-5.5
# export ANTHROPIC_API_KEY=sk-ant-... # for --model claude-opus-4-7
# export GEMINI_API_KEY=... # for --model gemini-3-pro-preview