sae: type: matryoshka_batch_topk activation_dim: 4096 expansion_factor: 4 layer_id: 15 hookpoint: residual k: 256 group_fractions: - 0.5 - 0.25 - 0.125 - 0.0625 - 0.0625 group_weights: null trainer: epochs: 1 lr: null l1_penalty: 0.1 warmup_steps: 10 sparsity_warmup_steps: 0 decay_start: null resample_steps: null seed: 42 device: cuda:0 log_every_n_steps: 1000 logger_type: mlflow validate: false auxk_alpha: 0.03125 threshold_beta: 0.999 threshold_start_step: 1000 threshold_dead_features: 100000 data: dataset_names: - mimic_findings_temporal activations_type: per_token num_workers: 18 batch_size: 8192 val_samples: 512000 train_samples: null norm_act: true norm_to_sqrt_act_dim: false input_unit_norm: false filter_dict: null