# Python-generated files
__pycache__/
*.py[oc]
build/
dist/
wheels/
*.egg-info

# Virtual environments
.venv

*.7z
*.txt
plans/overfit/dataset/
dataset/*
!dataset/.gitkeep
dataset_finetune/
checkpoint/
checkpoints/
outputs/
*.tsv
!src/aligner/testdata/*.tsv
*.onnx
*.onnx.data
dataset_*/
*.env
target/
wandb/
model-he/
*.npy
*.wav
*.tar.gz
backup_checkpoints/
models/
plans/*.pdf
data/

# Training outputs and model weights (never commit)
*.safetensors
outputs_phased/
outputs_phased80.1/
outputs_double/
NeoBERT/
hebrew-homographs-lexicons/
outputs_phased80.1
outputs_phased79.9
outputs_phased_single
outputs_phased
outputs_phased_modern_single
outputs_double
outputs_new
outputs
outputs_gate_dual
outputs_phased80.1
outputs_phased79.9
outputs_phased_single
outputs_phased_modern_single
outputs_phased
outputs_new
outputs_dual
outputs_double
outputs_homographs

# Large local assets, NOT scratch -- never delete these (§110 tree tidy).
# `pretrain/ckpts/*` is named by `encoder_name` in 22 checkpoints' arch.json;
# `logs/` holds every arm's training/predict log.
pretrain/
logs/
# The two frozen baseline prediction dumps `scripts/score_predictions.py` reads by
# name from the repo root. §71: both are dumps over `dataset/test.tsv`, so they are
# EVAL assets under the same eval-only rule -- never train on them, never mine them.
gemini.csv
renikudplus.csv