{
  "_name_or_path": "/shared/models/llama3-checkpoint-sft-padding-opencoder-concat-dlerp16-cosine-loss/checkpoint-11000",
  "architectures": [
    "CausalLMExpansion"
  ],
  "attention_bias": false,
  "attention_dropout": 0.0,
  "bos_token_id": 128000,
  "eos_token_id": [
    128001,
    128008,
    128009
  ],
  "expansion": {
    "expand_type": "concat",
    "expanded_from": "/shared/models/Meta-Llama-3-8B-Instruct",
    "freeze_interpolation_factor": true,
    "freeze_ori_layers": true,
    "freeze_ori_non_layers": true,
    "freezed_layers": [
      0,
      2,
      4,
      6,
      8,
      10,
      12,
      14,
      16,
      18,
      20,
      22,
      24,
      26,
      28,
      30
    ],
    "interpolation_factor": 0,
    "interpolation_loss_alpha": 1,
    "interpolation_loss_type": "cosine",
    "interpolation_norm_alpha": 0,
    "merge_method": "dlerp",
    "num_exp_layers": 16,
    "num_ori_layers": 32
  },
  "head_dim": 128,
  "hidden_act": "silu",
  "hidden_size": 4096,
  "initializer_range": 0.02,
  "intermediate_size": 14336,
  "max_position_embeddings": 131072,
  "mlp_bias": false,
  "model_type": "causallmexpansion",
  "num_attention_heads": 32,
  "num_hidden_layers": 32,
  "num_key_value_heads": 8,
  "pretraining_tp": 1,
  "rms_norm_eps": 1e-05,
  "rope_scaling": {
    "factor": 8.0,
    "high_freq_factor": 4.0,
    "low_freq_factor": 1.0,
    "original_max_position_embeddings": 8192,
    "rope_type": "llama3"
  },
  "rope_theta": 500000.0,
  "tie_word_embeddings": false,
  "torch_dtype": "bfloat16",
  "trained_by": [
    "OpenCoderSFTStage2"
  ],
  "trained_from": "/shared/models/Meta-Llama-3-8B-Instruct",
  "transformers_version": "4.46.2",
  "use_cache": false,
  "vocab_size": 128258
}
