| { | |
| "architecture": "topk", | |
| "d_in": 768, | |
| "d_sae": 3072, | |
| "dtype": "float32", | |
| "device": "cpu", | |
| "model_name": "EleutherAI/pythia-160m", | |
| "hook_name": "model.layers.6", | |
| "hook_layer": 6, | |
| "hook_head_index": null, | |
| "activation_fn_str": "topk", | |
| "activation_fn_kwargs": {}, | |
| "apply_b_dec_to_input": false, | |
| "finetuning_scaling_factor": false, | |
| "sae_lens_training_version": "deception-v4-sandbagging-v1", | |
| "prepend_bos": false, | |
| "dataset_path": "Solshine/deception-behavioral-multimodel", | |
| "dataset_trust_remote_code": false, | |
| "context_size": null, | |
| "normalize_activations": "none", | |
| "training_condition": "mixed", | |
| "training_notes": "V4 sandbagging SAE. Same-prompt MMLU behavioral divergence. Conditions: genuine_only / sandbagging_only / mixed. Model: EleutherAI/pythia-160m, Layer 6. See https://github.com/SolshineCode/deception-nanochat-sae-research" | |
| } |