{ "architectures": [ "TAMELM" ], "attention_backend": "flex", "attention_top_k": 2, "auto_map": { "AutoConfig": "modeling_tamelm.TAMEConfig", "AutoModelForCausalLM": "modeling_tamelm.TAMELM" }, "balance_coef": 0.01, "bos_token_id": 1, "depth_dim": 768, "depth_gate_init": 0.01, "depth_heads": 8, "depth_mode": "attention", "depth_slots": 16, "dtype": "float32", "embed_dim": 1024, "eos_token_id": 2, "expert_backend": "grouped", "expert_hidden_dim": 4096, "expert_top_k": 2, "graph_recent": 256, "graph_top_r": 2, "init_scheme": "v1", "local_window": 256, "loss_chunk_size": 256, "max_seq_len": 1024, "model_type": "tamelm_two_axis", "n_kv_heads": 4, "n_q_heads": 16, "num_experts": 4, "num_layers": 12, "pad_token_id": 0, "rms_norm_eps": 1e-06, "rope_type": "warped", "router_init_std": 0.01, "strided_every": 4, "strided_recent": 128, "tie_word_embeddings": true, "tile": 128, "tokenizer_sha256": "0d93d8f943b1a6627d62395c2fd583894cfe577b8aca1ca946495f7abba50958", "transformers_version": "5.5.0", "use_cache": false, "vocab_size": 49152 }