encoder: name: jhu-clsp/ettin-encoder-17m max_length: 512 dtype: bfloat16 trust_remote_code: false decoder: d_model: 64 n_layer: 2 n_head: 4 d_ff: 256 max_seq_len: 128 prefix_len: 1 dropout: 0.0 tie_embeddings: true data: hf_name: OLMo-Coding/starcoder-python-instruct split: train val_fraction: 0.1 max_identifiers: 26 max_tokens: 128 instruction_field: instruction code_field: text stdlib_whitelist_path: null num_proc: 16 cache_dir: /workspace-SR008.nfs2/nsushko/nsushko/polygon/tiny_decoder/.cache/tiny_decoder/shared vocab: path: configs/vocab_98.yaml builtin_aliases_path: configs/builtin_aliases.yaml attr_aliases_path: configs/attr_aliases.yaml train: batch_size: 128 epochs: 50 lr: 0.0003 warmup_frac: 0.05 weight_decay: 0.1 grad_clip: 1.0 early_stop_patience: 5 seed: 0 log_every: 50 eval_every_epochs: 1 wandb_project: null tensorboard_dir: null eval: split: val max_samples: null compute_exact_match: true compute_syntactic_validity: true run: dir: /workspace-SR008.nfs2/nsushko/nsushko/polygon/tiny_decoder/runs/prefix_k1 save_every_epochs: 1