File size: 396 Bytes
0839743
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
quant_stage:
  quant_modifiers:
    SmoothQuantModifier:
      smoothing_strength: 0.885
      mappings:
      - - ['re:.*qkv_proj']
        - re:.*input_layernorm
      - - ['re:.*gate_up_proj']
        - re:.*post_attention_layernorm
    GPTQModifier:
      sequential_update: true
      dampening_frac: 0.01
      ignore: [lm_head]
      scheme: W8A8
      targets: Linear
      observer: mse