numind
/

NuExtract3-FP8

default_stage:
  default_modifiers:
    QuantizationModifier:
      targets: [Linear]
      ignore: [lm_head, 're:.*visual.*', 're:.*linear_attn.*']
      scheme: FP8
      kv_cache_scheme:
        num_bits: 8
        type: float
        symmetric: true
        group_size: null
        strategy: tensor
        block_structure: null
        dynamic: false
        actorder: null
        scale_dtype: null
        zp_dtype: null
        observer: memoryless_minmax
        observer_kwargs: {}
      bypass_divisibility_checks: false