1"""Auto-generated Pydantic models from config_schema.yaml."""
3from __future__
import annotations
4from typing
import Any, Dict, List, Optional, Union
5from pydantic
import BaseModel, Field, ConfigDict, AliasChoices, model_validator
8from pathlib
import Path
10SCHEMA_VERSION =
"1.0.0"
15 @brief Configuration for uint8 input handling
17 @details Defines whether inputs should be treated as uint8 and which specific inputs to apply this to.
19 @param apply bool. If true, treat specified inputs as uint8
20 @param inputs List[str]. List of input names to treat as uint8. If empty and apply is true, applies to all inputs
21 @param division_factor float. Division factor for uint8 to float conversion (e.g., 255.0 for [0,1], 127.5 for [0,2])
24 model_config = ConfigDict(
25 populate_by_name=
True,
30 default=
False, description=
"If true, treat specified inputs as uint8"
32 inputs: List[str] = Field(
34 description=
"List of input names to treat as uint8. If empty and apply is true, applies to all inputs",
36 division_factor: float = Field(
38 alias=
"divisionFactor",
39 description=
"Division factor for uint8 to float conversion (e.g., 255.0 for [0,1], 127.5 for [0,2])",
43 """Return a copy with updated fields."""
44 return self.model_copy(update=kwargs)
49 @brief Configuration for input preprocessing pipeline
51 @details Defines preprocessing operations to be applied to model inputs,
52 including operations like resize, normalize, color conversion, etc.
53 Each pipeline entry is {"op": <name>, ...op keys}; unknown keys are rejected.
54 resize/letterbox take "backend": "torch" (default, float interpolate), "pil"
55 (Pillow, bit-exact, bilinear/bicubic) or "opencv" (cv2 INTER_LINEAR, bit-exact),
56 so calibration sees the same pixels as an evaluator built on that library.
57 letterbox "alignType" is 0 (default: center, padding split around the image) or 1
58 (top-left: padding only at the bottom and right, as YOLOX and DAMO-YOLO use).
59 letterbox "roundType" is 0 (default: each backend's rounding of the resized size) or 1
60 (floor: upstream YOLOX and DAMO-YOLO's int(w * r)); a fuseIntoFirstLayer letterbox
61 ignores it, since its exact odd decimation has no fraction to round.
63 @param apply bool. If true, apply preprocessing pipeline
64 @param auto_convert_format bool. If true, automatically convert input format
65 @param pipeline List[Dict[str, Any]]. List of preprocessing operations to apply globally
66 @param input_configs Dict[str, Any]. Per-input preprocessing configurations. Keys are input names
69 model_config = ConfigDict(
70 populate_by_name=
True,
75 default=
False, description=
"If true, apply preprocessing pipeline"
77 auto_convert_format: bool = Field(
79 alias=
"autoConvertFormat",
80 description=
"If true, automatically convert input format",
82 pipeline: Any = Field(
83 default=[], description=
"List of preprocessing operations to apply globally"
85 input_configs: Any = Field(
88 description=
"Per-input preprocessing configurations. Keys are input names",
92 """Return a copy with updated fields."""
93 return self.model_copy(update=kwargs)
98 @brief Configuration for resource management during model compilation
100 @details Controls GPU and memory management settings during the compilation process.
102 @param weight_dtype str. Weight data type for calibration (e.g., 'float32', 'float16')
103 @param gpu_memory_budget_mb int. Device memory budget for weight quantization in MiB; 0 = unlimited, -1 = auto
104 @param weight_memory WeightMemory. Weight memory management configuration
107 model_config = ConfigDict(
108 populate_by_name=
True,
112 weight_dtype: str = Field(
115 description=
"Weight data type for calibration (e.g., 'float32', 'float16')",
117 gpu_memory_budget_mb: int = Field(
119 alias=
"gpuMemoryBudgetMB",
120 description=
"Device memory budget for weight quantization in MiB; 0 = unlimited, -1 = auto",
125 @brief Weight memory management configuration
127 @param method int. Weight memory management method index:<br>
128 0: DeleteFloat - Delete float weights after quantization.<br>
129 1: SaveFloat - Save float weights to disk.<br>
130 2: MoveFloat - Move float weights to CPU.<br>
131 3: KeepFloat - Keep float weights in memory.<br>
132 4: KeepAll - Keep all weights in memory.<br>
135 method_list: List[str] = Field(
136 default=[
"DeleteFloat",
"SaveFloat",
"MoveFloat",
"KeepFloat",
"KeepAll"],
139 method: int = Field(default=0, alias=
"method")
141 weight_memory: WeightMemory = Field(
142 default_factory=WeightMemory, alias=
"weightMemory"
146 """Return a copy with updated fields."""
147 return self.model_copy(update=kwargs)
152 @brief Configuration for calibration during quantization
154 @details Defines calibration and quantization parameterization used to derive activation/weight scales
155 and related statistics during quantized compilation.
157 @param method int. Calibration method index:<br>
158 0: WChALayer - Weight per-channel, Activation per-layer.<br>
159 1: WChAMulti - Weight per-channel, Activation multi-layer.<br>
160 2: WChALayerZeropoint - Weight per-channel, Activation per-layer with zeropoint.<br>
161 3: WChAMultiZeropoint - Weight per-channel, Activation multi-layer with zeropoint.<br>
162 @param output int. Output quantization type index:<br>
163 0: Layer - Per-layer quantization.<br>
164 1: Ch - Per-channel quantization.<br>
165 2: Sigmoid - Sigmoid-based quantization.<br>
166 @param mode int. Quantization mode index:<br>
167 0: Max - Maximum value calibration.<br>
168 1: MaxPercentile - Maximum percentile calibration.<br>
169 2: Histogram - Histogram-based calibration.<br>
170 @param act_scale_min float. Minimum allowed activation scale (lower bound clamp)
171 @param act16_scale_min float. Minimum 16-bit activation scale (actScaleMin / 256)
172 @param weight_scale_min float. Minimum allowed weight scale (lower bound clamp)
173 @param weight16_scale_min float. Minimum 16-bit weight scale (weightScaleMin / 256)
174 @param min_clip_ratio float. Minimum clip ratio constraint applied during calibration
175 @param max_calib_data_size int. Maximum number of calibration samples kept after loading or generation
176 @param max_sample_size_for_quant_scheme int. Maximum number of calibration samples used per quant scheme stage
177 @param max_percentile MaxPercentile. MaxPercentile mode configuration
178 @param fast_dist FastDist. Fast distribution calibration configuration
179 @param histogram Histogram. Histogram-based calibration configuration
180 @param layer_overrides LayerOverrides. Layer-specific override settings for calibration
181 @param statistics Statistics. Statistics save/load configuration with percentile selection
182 @param group_lut GroupLut. LUT grouping algorithm configuration
185 model_config = ConfigDict(
186 populate_by_name=
True,
190 method_list: List[str] = Field(
191 default=[
"WChALayer",
"WChAMulti",
"WChALayerZeropoint",
"WChAMultiZeropoint"],
194 method: int = Field(default=1, alias=
"method")
195 output_list: List[str] = Field(
196 default=[
"Layer",
"Ch",
"Sigmoid"], alias=
"outputList"
198 output: int = Field(default=0, alias=
"output")
199 mode_list: List[str] = Field(
200 default=[
"Max",
"MaxPercentile",
"Histogram"], alias=
"modeList"
202 mode: int = Field(default=1, alias=
"mode")
204 act_scale_min: float = Field(
207 description=
"Minimum allowed activation scale (lower bound clamp)",
211 act16_scale_min: float = Field(
212 default=1.953125e-06,
213 alias=
"act16ScaleMin",
214 description=
"Minimum 16-bit activation scale (actScaleMin / 256)",
216 weight_scale_min: float = Field(
218 alias=
"weightScaleMin",
219 description=
"Minimum allowed weight scale (lower bound clamp)",
223 weight16_scale_min: float = Field(
225 alias=
"weight16ScaleMin",
226 description=
"Minimum 16-bit weight scale (weightScaleMin / 256)",
228 min_clip_ratio: float = Field(
230 alias=
"minClipRatio",
231 description=
"Minimum clip ratio constraint applied during calibration",
235 max_calib_data_size: int = Field(
237 alias=
"maxCalibDataSize",
238 description=
"Maximum number of calibration samples kept after loading or generation",
241 max_sample_size_for_quant_scheme: int = Field(
243 alias=
"maxSampleSizeForQuantScheme",
244 description=
"Maximum number of calibration samples used per quant scheme stage",
249 model_config = ConfigDict(populate_by_name=
True)
252 @brief MaxPercentile mode configuration
254 @param percentile float. Percentile value for maxPercentile mode
255 @param topk_ratio float. Top-k ratio used in maxPercentile mode
256 @param max_each int. Maximum number of samples processed per iteration
257 @param max_total int. Total maximum number of samples
258 @param per_ch_divisor int. Divisor for per-channel buffer capacity (bufferCap = max(maxTotal / perChDivisor, maxEach))
260 percentile: float = Field(
261 default=0.9999, description=
"Percentile value for maxPercentile mode"
263 topk_ratio: float = Field(
266 description=
"Top-k ratio used in maxPercentile mode",
268 max_each: int = Field(
271 description=
"Maximum number of samples processed per iteration",
273 max_total: int = Field(
276 description=
"Total maximum number of samples",
278 per_ch_divisor: int = Field(
280 alias=
"perChDivisor",
281 description=
"Divisor for per-channel buffer capacity (bufferCap = max(maxTotal / perChDivisor, maxEach))",
285 max_percentile: MaxPercentile = Field(
286 default_factory=MaxPercentile, alias=
"maxPercentile"
290 model_config = ConfigDict(populate_by_name=
True)
293 @brief Fast distribution calibration configuration
295 @param size_cali int.
296 @param kernel_size int.
297 @param stack_size int.
299 size_cali: int = Field(default=100, alias=
"sizeCali")
300 kernel_size: int = Field(default=9, alias=
"kernelSize")
301 stack_size: int = Field(default=32768, alias=
"stackSize")
303 fast_dist: FastDist = Field(default_factory=FastDist, alias=
"fastDist")
306 model_config = ConfigDict(populate_by_name=
True)
309 @brief Histogram-based calibration configuration
311 @param search_type int. Search type for histogram calibration:<br>
315 @param percentile float. Percentile value for histogram calibration
316 @param use_gpu bool. Use GPU for histogram computation
317 @param num_bins int. Number of bins for histogram
318 @param num_samples int. Number of samples for histogram calibration
319 @param buffer_size int. Buffer size for histogram computation (-1 for auto)
320 @param min_bin_width float. Minimum bin width for histogram
321 @param search_percentile_min float. Minimum search percentile
322 @param search_percentile_max float. Maximum search percentile
323 @param num_search int. Number of search iterations
325 search_type_list: List[str] = Field(
326 default=[
"Percentile",
"MSE",
"KL"], alias=
"searchTypeList"
328 search_type: int = Field(default=0, alias=
"searchType")
329 percentile: float = Field(
330 default=0.9999, description=
"Percentile value for histogram calibration"
332 use_gpu: bool = Field(
335 description=
"Use GPU for histogram computation",
337 num_bins: int = Field(
338 default=256, alias=
"numBins", description=
"Number of bins for histogram"
340 num_samples: int = Field(
343 description=
"Number of samples for histogram calibration",
345 buffer_size: int = Field(
348 description=
"Buffer size for histogram computation (-1 for auto)",
350 min_bin_width: float = Field(
353 description=
"Minimum bin width for histogram",
355 search_percentile_min: float = Field(
357 alias=
"searchPercentileMin",
358 description=
"Minimum search percentile",
360 search_percentile_max: float = Field(
362 alias=
"searchPercentileMax",
363 description=
"Maximum search percentile",
365 num_search: int = Field(
366 default=128, alias=
"numSearch", description=
"Number of search iterations"
369 histogram: Histogram = Field(default_factory=Histogram, alias=
"histogram")
372 model_config = ConfigDict(populate_by_name=
True)
375 @brief Layer-specific override settings for calibration
377 @param act_scale_min dict. Per-layer activation scale minimum overrides. Keys are actScaleMin values (e.g. '0.0005'), values are lists of layer names to apply that override. (e.g. {'0.0005' : ['layer1', 'layer2']})
378 @param percentile dict. Per-layer maxPercentile percentile overrides. Keys are layer names, values are percentile floats. (e.g. {'/model/layer0/conv': 0.999, '/model/layer5/attn': 0.99})
379 @param method dict. Per-layer calibration method overrides. Keys are layer names, values are method ints (0=WChALayer, 1=WChAMulti, 2=WChALayerZeropoint, 3=WChAMultiZeropoint). (e.g. {'/model/layer0/conv': 1, '/model/layer5/attn': 3})
381 act_scale_min: Any = Field(
384 description=
"Per-layer activation scale minimum overrides. Keys are actScaleMin values (e.g. '0.0005'), values are lists of layer names to apply that override. (e.g. {'0.0005' : ['layer1', 'layer2']})",
386 percentile: Any = Field(
388 description=
"Per-layer maxPercentile percentile overrides. Keys are layer names, values are percentile floats. (e.g. {'/model/layer0/conv': 0.999, '/model/layer5/attn': 0.99})",
392 description=
"Per-layer calibration method overrides. Keys are layer names, values are method ints (0=WChALayer, 1=WChAMulti, 2=WChALayerZeropoint, 3=WChAMultiZeropoint). (e.g. {'/model/layer0/conv': 1, '/model/layer5/attn': 3})",
395 layer_overrides: LayerOverrides = Field(
396 default_factory=LayerOverrides, alias=
"layerOverrides"
400 model_config = ConfigDict(populate_by_name=
True)
403 @brief Statistics save/load configuration with percentile selection
405 @param apply bool. Enable statistics save/load
406 @param save_path str. Path to save statistics. If empty, not saved
407 @param load_path str. Path to load statistics. If empty, not loaded
408 @param percentiles List[float]. List of percentile candidates
409 @param percentile_index int. Index into percentiles list to select active percentile
411 apply: bool = Field(default=
False, description=
"Enable statistics save/load")
412 save_path: str = Field(
415 description=
"Path to save statistics. If empty, not saved",
417 load_path: str = Field(
420 description=
"Path to load statistics. If empty, not loaded",
422 percentiles: List[float] = Field(
423 default=[0.9999, 0.999, 0.99, 0.9],
424 description=
"List of percentile candidates",
426 percentile_index: int = Field(
428 alias=
"percentileIndex",
429 description=
"Index into percentiles list to select active percentile",
432 statistics: Statistics = Field(default_factory=Statistics, alias=
"statistics")
435 model_config = ConfigDict(populate_by_name=
True)
438 @brief LUT grouping algorithm configuration
440 @param irls_iter int. Number of IRLS iterations for optimal scale computation (1 = single WLS)
441 @param dro_eps float. DRO worst-point coefficient; larger = more conservative clipping (0 = disable)
442 @param cover_floor float. LUTAM cluster scale lower bound as fraction of coverage
444 irls_iter: int = Field(
447 description=
"Number of IRLS iterations for optimal scale computation (1 = single WLS)",
450 dro_eps: float = Field(
453 description=
"DRO worst-point coefficient; larger = more conservative clipping (0 = disable)",
456 cover_floor: float = Field(
459 description=
"LUTAM cluster scale lower bound as fraction of coverage",
464 group_lut: GroupLut = Field(default_factory=GroupLut, alias=
"groupLut")
467 """Return a copy with updated fields."""
468 return self.model_copy(update=kwargs)
473 @brief Configuration for bit precision
475 @details Defines bit-width parameterization for activations and weights used in
476 mixed-precision quantization (e.g., attention and FFN components).
478 @param transformer Transformer. Transformer-specific bit-width configuration
479 @param save_info SaveInfo. Bit allocation save/load configuration
480 @param layer_overrides LayerOverrides. Layer-specific bit-width override settings
483 model_config = ConfigDict(
484 populate_by_name=
True,
489 model_config = ConfigDict(populate_by_name=
True)
492 @brief Transformer-specific bit-width configuration
494 @param activation Activation. Activation bit-widths for transformer components
495 @param weight Weight. Weight bit-widths for transformer components
496 @param mixed_precision MixedPrecision. Mixed precision configuration
500 model_config = ConfigDict(populate_by_name=
True)
503 @brief Activation bit-widths for transformer components
505 @param query int. Query activation bit-width
506 @param key int. Key activation bit-width
507 @param value int. Value activation bit-width
508 @param output int. Output activation bit-width
509 @param head int. Head activation bit-width
510 @param router int. MoE router gate activation bit-width
511 @param ffn Ffn. FFN activation bit-widths (int shorthand sets all sublayers)
513 query: int = Field(default=8, description=
"Query activation bit-width")
514 key: int = Field(default=8, description=
"Key activation bit-width")
515 value: int = Field(default=8, description=
"Value activation bit-width")
516 output: int = Field(default=16, description=
"Output activation bit-width")
517 head: int = Field(default=8, description=
"Head activation bit-width")
519 default=8, description=
"MoE router gate activation bit-width"
524 @brief FFN activation bit-widths (int shorthand sets all sublayers)
526 @param up int. FFN up-projection activation bit-width
527 @param gate int. FFN gate (SwiGLU) activation bit-width
528 @param down int. FFN down-projection activation bit-width
531 @model_validator(mode="before")
533 def expand_int_shorthand(cls, v):
534 if isinstance(v, bool):
535 raise ValueError(
"bool is not a valid bit-width")
536 if isinstance(v, int):
537 return {
"up": int(v),
"gate": int(v),
"down": int(v)}
541 default=16, description=
"FFN up-projection activation bit-width"
544 default=16, description=
"FFN gate (SwiGLU) activation bit-width"
547 default=16, description=
"FFN down-projection activation bit-width"
550 ffn: Ffn = Field(default_factory=Ffn, alias=
"ffn")
552 activation: Activation = Field(default_factory=Activation, alias=
"activation")
555 model_config = ConfigDict(populate_by_name=
True)
558 @brief Weight bit-widths for transformer components
560 @param query int. Query weight bit-width
561 @param key int. Key weight bit-width
562 @param value int. Value weight bit-width
563 @param output int. Output weight bit-width
564 @param head int. Head weight bit-width
565 @param router int. MoE router gate weight bit-width
566 @param ffn Ffn. FFN weight bit-widths (int shorthand sets all sublayers)
568 query: int = Field(default=8, description=
"Query weight bit-width")
569 key: int = Field(default=8, description=
"Key weight bit-width")
570 value: int = Field(default=8, description=
"Value weight bit-width")
571 output: int = Field(default=8, description=
"Output weight bit-width")
572 head: int = Field(default=8, description=
"Head weight bit-width")
574 default=8, description=
"MoE router gate weight bit-width"
579 @brief FFN weight bit-widths (int shorthand sets all sublayers)
581 @param up int. FFN up-projection weight bit-width
582 @param gate int. FFN gate (SwiGLU) weight bit-width
583 @param down int. FFN down-projection weight bit-width
586 @model_validator(mode="before")
588 def expand_int_shorthand(cls, v):
589 if isinstance(v, bool):
590 raise ValueError(
"bool is not a valid bit-width")
591 if isinstance(v, int):
592 return {
"up": int(v),
"gate": int(v),
"down": int(v)}
596 default=8, description=
"FFN up-projection weight bit-width"
599 default=8, description=
"FFN gate (SwiGLU) weight bit-width"
602 default=8, description=
"FFN down-projection weight bit-width"
605 ffn: Ffn = Field(default_factory=Ffn, alias=
"ffn")
607 weight: Weight = Field(default_factory=Weight, alias=
"weight")
610 model_config = ConfigDict(populate_by_name=
True)
613 @brief Mixed precision configuration
615 @param weight Weight. Mixed precision configuration for weights
616 @param activation Activation. Per-layer activation mixed precision configuration
620 model_config = ConfigDict(populate_by_name=
True)
623 @brief Mixed precision configuration for weights
625 @param apply bool. If true, apply mixed-precision according to the specified bit-widths
626 @param type_wise bool. Apply type-wise mixed precision
627 @param prune float. Pruning ratio
628 @param bit_2 float. Ratio of 2-bit quantization
629 @param bit_4 float. Ratio of 4-bit quantization
630 @param bit_8 float. Ratio of 8-bit quantization
631 @param importance_threshold_low float. Low importance threshold
632 @param importance_threshold_high float. High importance threshold
636 description=
"If true, apply mixed-precision according to the specified bit-widths",
638 type_wise: bool = Field(
641 description=
"Apply type-wise mixed precision",
643 prune: float = Field(default=0, description=
"Pruning ratio")
644 bit_2: float = Field(
646 validation_alias=AliasChoices(
"bit_2",
"bit2"),
647 description=
"Ratio of 2-bit quantization",
649 bit_4: float = Field(
651 validation_alias=AliasChoices(
"bit_4",
"bit4"),
652 description=
"Ratio of 4-bit quantization",
654 bit_8: float = Field(
656 validation_alias=AliasChoices(
"bit_8",
"bit8"),
657 description=
"Ratio of 8-bit quantization",
659 importance_threshold_low: float = Field(
661 alias=
"importanceThreshold_low",
662 description=
"Low importance threshold",
664 importance_threshold_high: float = Field(
666 alias=
"importanceThreshold_high",
667 description=
"High importance threshold",
670 weight: Weight = Field(default_factory=Weight, alias=
"weight")
673 model_config = ConfigDict(populate_by_name=
True)
676 @brief Per-layer activation mixed precision configuration
678 @param apply bool. If true, apply per-layer activation mixed precision
679 @param ratio_16bit float. Ratio of layers assigned 16-bit (used when importanceThreshold < 0)
680 @param importance_threshold float. Normalized importance threshold for 16-bit assignment (negative = use ratio_16bit)
681 @param search_range int. Target layer range: -1=all layers, N=first N layers
685 description=
"If true, apply per-layer activation mixed precision",
687 ratio_16bit: float = Field(
690 description=
"Ratio of layers assigned 16-bit (used when importanceThreshold < 0)",
692 importance_threshold: float = Field(
694 alias=
"importanceThreshold",
695 description=
"Normalized importance threshold for 16-bit assignment (negative = use ratio_16bit)",
697 search_range: int = Field(
700 description=
"Target layer range: -1=all layers, N=first N layers",
703 activation: Activation = Field(
704 default_factory=Activation, alias=
"activation"
707 mixed_precision: MixedPrecision = Field(
708 default_factory=MixedPrecision, alias=
"mixedPrecision"
711 transformer: Transformer = Field(default_factory=Transformer, alias=
"transformer")
714 model_config = ConfigDict(populate_by_name=
True)
717 @brief Bit allocation save/load configuration
719 @param save_path str. Path to save the bit allocation. If empty, not saved
720 @param load_path str. Path to load the bit allocation. If empty, not loaded
722 save_path: str = Field(
725 description=
"Path to save the bit allocation. If empty, not saved",
727 load_path: str = Field(
730 description=
"Path to load the bit allocation. If empty, not loaded",
733 save_info: SaveInfo = Field(default_factory=SaveInfo, alias=
"saveInfo")
736 model_config = ConfigDict(populate_by_name=
True)
739 @brief Layer-specific bit-width override settings
741 @param activation_16bits list[string]. Layer names to force 16-bit activations
742 @param weight_16bits list[string]. Layer names to force 16-bit weights
743 @param weight_8bits list[string]. Layer names to force 8-bit weights
745 activation_16bits: List[str] = Field(
747 alias=
"activation16Bits",
748 description=
"Layer names to force 16-bit activations",
750 weight_16bits: List[str] = Field(
752 alias=
"weight16Bits",
753 description=
"Layer names to force 16-bit weights",
755 weight_8bits: List[str] = Field(
758 description=
"Layer names to force 8-bit weights",
761 layer_overrides: LayerOverrides = Field(
762 default_factory=LayerOverrides, alias=
"layerOverrides"
766 """Return a copy with updated fields."""
767 return self.model_copy(update=kwargs)
772 @brief Configuration for HessianQuant algorithm
774 @details Defines parameters controlling whether and how HessianQuant is applied during quantization,
775 including layer-level inclusion/exclusion lists.
777 @param apply bool. If true, apply HessianQuant
778 @param hessian_dtype str. Storage dtype for the accumulated HessianQuant Hessian. bf16 halves its memory footprint (host RAM when the Hessian lives on CPU, VRAM when on GPU) — critical for large models such as MoE with thousands of expert FFN Hessians. Compute stays float32 regardless: the per-batch matmul and accumulation run in float32 and the solve upcasts back to float32; only the persistent accumulator is bfloat16. Accepted values:<br>
779 "fp32" - Store the Hessian in float32. The default.<br>
780 "bf16" - Store the Hessian in bfloat16 (half the memory).<br>
781 @param solver str. HessianQuant solve algorithm used to compute per-layer integer weights. symmetric targets the output on the quantized input; the asymmetric solvers target the original-weight float output and differ in how they absorb the input mismatch. Accepted values:<br>
782 "symmetric" - The default. min ||(W - Q) X_q||^2: sequential greedy rounding via a Cholesky-based Hessian solve.<br>
783 "asymmetric_causal" - min ||W X_fp - Q X_q||^2: each column's input-mismatch residual is fed only to the columns after it, scaled by alpha.<br>
784 "asymmetric_refit" - min ||W X_fp - Q X_q||^2: least-squares refit W*^T = H^-1 G W^T of every column, then a symmetric solve around W*.<br>
785 @param rescomp bool. Enable residual-compensation correction. When true, each solver additionally applies an R correction term that compensates for accumulated weight drift from inter-block error propagation. Requires crossGram collection even for solver=symmetric.
786 @param attributes Attributes. HessianQuant algorithm attributes
789 model_config = ConfigDict(
790 populate_by_name=
True,
794 apply: bool = Field(default=
False, description=
"If true, apply HessianQuant")
795 hessian_dtype: str = Field(
797 alias=
"hessianDtype",
798 description=
"Storage dtype for the accumulated HessianQuant Hessian. bf16 halves its memory footprint (host RAM when the Hessian lives on CPU, VRAM when on GPU) — critical for large models such as MoE with thousands of expert FFN Hessians. Compute stays float32 regardless: the per-batch matmul and accumulation run in float32 and the solve upcasts back to float32; only the persistent accumulator is bfloat16.",
802 description=
"HessianQuant solve algorithm used to compute per-layer integer weights. symmetric targets the output on the quantized input; the asymmetric solvers target the original-weight float output and differ in how they absorb the input mismatch.",
804 rescomp: bool = Field(
806 description=
"Enable residual-compensation correction. When true, each solver additionally applies an R correction term that compensates for accumulated weight drift from inter-block error propagation. Requires crossGram collection even for solver=symmetric.",
810 model_config = ConfigDict(populate_by_name=
True)
813 @brief HessianQuant algorithm attributes
815 @param act_order bool. If true, use activation order
816 @param block_size int. Block size used for HessianQuant
817 @param perc_damp float. Percentage dampening factor
818 @param apply_layers List[str]. Layer names to apply HessianQuant. If empty, applies to all eligible layers
819 @param exclude_layers List[str]. Layer names to exclude from HessianQuant
820 @param alpha float. Scale of the secondary correction terms: the asymmetric_causal P term and the rescomp R term. symmetric and asymmetric_refit use it only through rescomp.
822 act_order: bool = Field(
823 default=
True, alias=
"actOrder", description=
"If true, use activation order"
825 block_size: int = Field(
828 description=
"Block size used for HessianQuant",
830 perc_damp: float = Field(
831 default=0.01, alias=
"percDamp", description=
"Percentage dampening factor"
833 apply_layers: List[str] = Field(
836 description=
"Layer names to apply HessianQuant. If empty, applies to all eligible layers",
838 exclude_layers: List[str] = Field(
840 alias=
"excludeLayers",
841 description=
"Layer names to exclude from HessianQuant",
843 alpha: float = Field(
845 description=
"Scale of the secondary correction terms: the asymmetric_causal P term and the rescomp R term. symmetric and asymmetric_refit use it only through rescomp.",
848 attributes: Attributes = Field(default_factory=Attributes, alias=
"attributes")
851 """Return a copy with updated fields."""
852 return self.model_copy(update=kwargs)
857 @brief Configuration for in-schedule integer bias correction
859 @details During weight quantization, measures the per-channel activation error of each
860 quantized convolution (plain, depthwise or transposed; not grouped) against the
861 original-weight float path over every calibration sample, folds the correction
862 into the integer bias within its range, skipping a channel whose slope is too
863 flat to invert, and patches the already-produced integer
864 activations so downstream layers quantize against corrected outputs.
865 Corrections propagate layer by layer in one pass; there is no damping rate
866 and no iteration count. Under group-wise quantization both streams start each
867 group from the float boundary batch, so the correction covers the error the
868 group itself adds. A run the engine cannot drive leaf by leaf (expert
869 aggregation, an external input narrower than its per-channel qparam) keeps the
870 calibrated bias. A layer whose LUT window is re-chosen is measured on its
871 accumulator, ahead of the window; any other layer on its requantized codes.
873 @param apply bool. If true, correct systematic per-layer quantization bias
874 @param attributes Attributes. Bias correction attributes
877 model_config = ConfigDict(
878 populate_by_name=
True,
884 description=
"If true, correct systematic per-layer quantization bias",
888 model_config = ConfigDict(populate_by_name=
True)
891 @brief Bias correction attributes
893 @param apply_layers List[str]. Layer names to apply bias correction. If empty, applies to all eligible layers
894 @param exclude_layers List[str]. Layer names to exclude from bias correction
896 apply_layers: List[str] = Field(
899 description=
"Layer names to apply bias correction. If empty, applies to all eligible layers",
901 exclude_layers: List[str] = Field(
903 alias=
"excludeLayers",
904 description=
"Layer names to exclude from bias correction",
907 attributes: Attributes = Field(default_factory=Attributes, alias=
"attributes")
910 """Return a copy with updated fields."""
911 return self.model_copy(update=kwargs)
916 @brief Configuration for Minimum Output Difference algorithm
918 @details Defines parameters controlling whether and how MOD is applied during quantization,
919 including layer-level inclusion/exclusion lists.
921 @param apply bool. If true, apply MOD
922 @param attributes Attributes. MOD algorithm attributes
925 model_config = ConfigDict(
926 populate_by_name=
True,
930 apply: bool = Field(default=
False, description=
"If true, apply MOD")
933 model_config = ConfigDict(populate_by_name=
True)
936 @brief MOD algorithm attributes
938 @param epochs int. Number of training epochs
939 @param warmup_epochs int. Number of warmup epochs
940 @param lr_min_ratio float. Minimum learning rate ratio
941 @param save_dir str. Directory to save MOD results
942 @param seed int. Random seed for MOD
943 @param apply_layers List[str]. Layer names to apply MOD. If empty, applies to all eligible layers
944 @param exclude_layers List[str]. Layer names to exclude from MOD
945 @param mod_after_layer_name str. Apply MOD after this layer
946 @param anchors List. Anchor configurations for detection models. Nested list of anchor box sizes for detection models. Structure: List[List[List[int]]] where outer list is per detection head (e.g. small/medium/large), middle list is anchors per head, inner list is [width, height]. Example: [[[12,16],[19,36],[40,28]], [[36,75],[76,55],[72,146]], [[142,110],[192,243],[459,401]]]
947 @param use_xyxy bool. Use XYXY format for bounding boxes
948 @param learning_rates LearningRates. Learning rate configuration for MOD
949 @param training Training. MOD training configuration
950 @param loss Loss. MOD loss configuration
951 @param post_processing PostProcessing. Post-processing configuration for detection models
953 epochs: int = Field(default=4, description=
"Number of training epochs")
954 warmup_epochs: int = Field(
955 default=1, alias=
"warmupEpochs", description=
"Number of warmup epochs"
957 lr_min_ratio: float = Field(
960 description=
"Minimum learning rate ratio",
962 save_dir: str = Field(
963 default=
"", alias=
"saveDir", description=
"Directory to save MOD results"
965 seed: int = Field(default=0, description=
"Random seed for MOD")
966 apply_layers: List[str] = Field(
969 description=
"Layer names to apply MOD. If empty, applies to all eligible layers",
971 exclude_layers: List[str] = Field(
973 alias=
"excludeLayers",
974 description=
"Layer names to exclude from MOD",
976 mod_after_layer_name: str = Field(
978 alias=
"modAfterLayerName",
979 description=
"Apply MOD after this layer",
981 anchors: Any = Field(
983 description=
"Anchor configurations for detection models. Nested list of anchor box sizes for detection models. Structure: List[List[List[int]]] where outer list is per detection head (e.g. small/medium/large), middle list is anchors per head, inner list is [width, height]. Example: [[[12,16],[19,36],[40,28]], [[36,75],[76,55],[72,146]], [[142,110],[192,243],[459,401]]]",
985 use_xyxy: bool = Field(
988 description=
"Use XYXY format for bounding boxes",
992 model_config = ConfigDict(populate_by_name=
True)
995 @brief Learning rate configuration for MOD
997 @param act_scale float. Learning rate for activation scale
998 @param zeropoint float. Learning rate for zeropoint
999 @param weight_scale float. Learning rate for weight scale
1000 @param weight float. Learning rate for weight
1001 @param bias float. Learning rate for bias
1003 act_scale: float = Field(
1006 description=
"Learning rate for activation scale",
1008 zeropoint: float = Field(
1009 default=0.0, description=
"Learning rate for zeropoint"
1011 weight_scale: float = Field(
1013 alias=
"weightScale",
1014 description=
"Learning rate for weight scale",
1016 weight: float = Field(default=4e-06, description=
"Learning rate for weight")
1017 bias: float = Field(default=4e-06, description=
"Learning rate for bias")
1019 learning_rates: LearningRates = Field(
1020 default_factory=LearningRates, alias=
"learningRates"
1024 model_config = ConfigDict(populate_by_name=
True)
1027 @brief MOD training configuration
1029 @param batch_size int. Batch size for MOD training
1030 @param q_drop float. Quantization drop probability
1031 @param quantize_weight bool. Whether to quantize weights
1032 @param weight_scale_init str. Weight scale initialization method
1033 @param downresol_mode str. Downresolution mode
1034 @param scheduler_type str. LR scheduler type
1036 batch_size: int = Field(
1037 default=1, alias=
"batchSize", description=
"Batch size for MOD training"
1039 q_drop: float = Field(
1040 default=0.0, alias=
"qDrop", description=
"Quantization drop probability"
1042 quantize_weight: bool = Field(
1044 alias=
"quantizeWeight",
1045 description=
"Whether to quantize weights",
1047 weight_scale_init: str = Field(
1049 alias=
"weightScaleInit",
1050 description=
"Weight scale initialization method",
1052 downresol_mode: str = Field(
1053 default=
"STE", alias=
"downresolMode", description=
"Downresolution mode"
1055 scheduler_type: str = Field(
1056 default=
"Cosine", alias=
"schedulerType", description=
"LR scheduler type"
1059 training: Training = Field(default_factory=Training, alias=
"training")
1062 model_config = ConfigDict(populate_by_name=
True)
1065 @brief MOD loss configuration
1067 @param type str. Loss type (MSE, KL, etc.)
1068 @param use_outputs bool. Use model outputs for loss computation
1069 @param kl_temperature float. KL divergence temperature
1070 @param recon_prob float. Reconstruction probability
1071 @param recon_coeff float. Reconstruction coefficient
1072 @param lambda_0 float. Loss weight lambda_0
1073 @param lambda_1 float. Loss weight lambda_1
1074 @param lambda_2 float. Loss weight lambda_2
1075 @param lambda_3 float. Loss weight lambda_3
1076 @param custom_loss_jit_path str. Path to custom JIT-compiled loss function. Refer to /workspace/quantizer/pyutils/mel.pt
1078 type: str = Field(default=
"MSE", description=
"Loss type (MSE, KL, etc.)")
1079 use_outputs: bool = Field(
1082 description=
"Use model outputs for loss computation",
1084 kl_temperature: float = Field(
1086 alias=
"KLTemperature",
1087 description=
"KL divergence temperature",
1089 recon_prob: float = Field(
1090 default=1.0, alias=
"reconProb", description=
"Reconstruction probability"
1092 recon_coeff: float = Field(
1095 description=
"Reconstruction coefficient",
1097 lambda_0: float = Field(
1099 validation_alias=AliasChoices(
"lambda_0",
"lambda0"),
1100 description=
"Loss weight lambda_0",
1102 lambda_1: float = Field(
1104 validation_alias=AliasChoices(
"lambda_1",
"lambda1"),
1105 description=
"Loss weight lambda_1",
1107 lambda_2: float = Field(
1109 validation_alias=AliasChoices(
"lambda_2",
"lambda2"),
1110 description=
"Loss weight lambda_2",
1112 lambda_3: float = Field(
1114 validation_alias=AliasChoices(
"lambda_3",
"lambda3"),
1115 description=
"Loss weight lambda_3",
1117 custom_loss_jit_path: str = Field(
1119 alias=
"customLossJITPath",
1120 description=
"Path to custom JIT-compiled loss function. Refer to /workspace/quantizer/pyutils/mel.pt",
1123 loss: Loss = Field(default_factory=Loss, alias=
"loss")
1126 model_config = ConfigDict(populate_by_name=
True)
1129 @brief Post-processing configuration for detection models
1131 @param post str. Post-processing type
1132 @param box_conf_thres float. Box confidence threshold
1133 @param box_iou_thres float. Box IoU threshold
1135 post: str = Field(default=
"", description=
"Post-processing type")
1136 box_conf_thres: float = Field(
1137 default=0, alias=
"boxConfThres", description=
"Box confidence threshold"
1139 box_iou_thres: float = Field(
1140 default=0, alias=
"boxIoUThres", description=
"Box IoU threshold"
1143 post_processing: PostProcessing = Field(
1144 default_factory=PostProcessing, alias=
"postProcessing"
1147 attributes: Attributes = Field(default_factory=Attributes, alias=
"attributes")
1150 """Return a copy with updated fields."""
1151 return self.model_copy(update=kwargs)
1156 @brief Configuration for Large Language Model (LLM) compilation
1158 @details Defines LLM-specific settings including sequence lengths, cache configurations,
1159 and runtime parameters for efficient LLM inference.
1161 @param apply bool. If True, apply LLM-specific configurations
1162 @param npu_parallel_degree int. Number of NPU partitions for FFN tensor parallelism (1 = disabled). Applied before OptimizeFFN if both active.
1163 @param attributes Attributes. LLM attributes configuration
1166 model_config = ConfigDict(
1167 populate_by_name=
True,
1171 apply: bool = Field(
1172 default=
False, description=
"If True, apply LLM-specific configurations"
1174 npu_parallel_degree: int = Field(
1176 alias=
"npuParallelDegree",
1177 description=
"Number of NPU partitions for FFN tensor parallelism (1 = disabled). Applied before OptimizeFFN if both active.",
1181 model_config = ConfigDict(populate_by_name=
True)
1184 @brief LLM attributes configuration
1186 @param max_data_length int. Maximum data length
1187 @param max_sequence_length int. Maximum sequence length
1188 @param max_cache_length int. Maximum cache length
1189 @param max_core_data_length int. Maximum core data length
1190 @param calibration Calibration. LLM calibration settings
1191 @param runtime Runtime. LLM runtime settings
1192 @param debug Debug. LLM debug settings
1194 max_data_length: int = Field(
1195 default=4096, alias=
"maxDataLength", description=
"Maximum data length"
1197 max_sequence_length: int = Field(
1199 alias=
"maxSequenceLength",
1200 description=
"Maximum sequence length",
1202 max_cache_length: int = Field(
1203 default=4096, alias=
"maxCacheLength", description=
"Maximum cache length"
1205 max_core_data_length: int = Field(
1207 alias=
"maxCoreDataLength",
1208 description=
"Maximum core data length",
1212 model_config = ConfigDict(populate_by_name=
True)
1215 @brief LLM calibration settings
1217 @param random_seq_length int. Random sequence length used for calibration
1218 @param use_full_seq_length bool. If True, use the full sequence length for calibration
1220 random_seq_length: int = Field(
1222 alias=
"randomSeqLength",
1223 description=
"Random sequence length used for calibration",
1225 use_full_seq_length: bool = Field(
1227 alias=
"useFullSeqLength",
1228 description=
"If True, use the full sequence length for calibration",
1231 calibration: Calibration = Field(
1232 default_factory=Calibration, alias=
"calibration"
1236 model_config = ConfigDict(populate_by_name=
True)
1239 @brief LLM runtime settings
1241 @param use_global_core bool. If True, use a global core
1242 @param batch_size int. Batch size
1243 @param npu_core_ids List[int]. List of NPU core IDs
1244 @param dynamic_rope bool. If True, enable dynamic RoPE (rotary position embedding)
1245 @param dynamic_mask bool. If True, enable dynamic mask (attention mask as runtime input)
1247 use_global_core: bool = Field(
1249 alias=
"useGlobalCore",
1250 description=
"If True, use a global core",
1252 batch_size: int = Field(
1253 default=1, alias=
"batchSize", description=
"Batch size"
1255 npu_core_ids: List[int] = Field(
1256 default=[0], alias=
"npuCoreIds", description=
"List of NPU core IDs"
1258 dynamic_rope: bool = Field(
1260 alias=
"dynamicRope",
1261 description=
"If True, enable dynamic RoPE (rotary position embedding)",
1263 dynamic_mask: bool = Field(
1265 alias=
"dynamicMask",
1266 description=
"If True, enable dynamic mask (attention mask as runtime input)",
1269 runtime: Runtime = Field(default_factory=Runtime, alias=
"runtime")
1272 model_config = ConfigDict(populate_by_name=
True)
1275 @brief LLM debug settings
1277 @param apply bool. Enable LLM debug mode
1278 @param batch_debug_bundle_size int. Batch debug bundle size
1280 apply: bool = Field(default=
False, description=
"Enable LLM debug mode")
1281 batch_debug_bundle_size: int = Field(
1283 alias=
"batchDebugBundleSize",
1284 description=
"Batch debug bundle size",
1287 debug: Debug = Field(default_factory=Debug, alias=
"debug")
1289 attributes: Attributes = Field(default_factory=Attributes, alias=
"attributes")
1292 """Return a copy with updated fields."""
1293 return self.model_copy(update=kwargs)
1298 @brief Sparse MoE expert-selection configuration (calibration only)
1300 @details Controls which experts are calibrated inside SparseMoe modules. selectionMode
1301 is a calibration-only knob: it picks which experts collect statistics (and, for
1302 TopK, on which tokens). It does NOT change inference routing — the forward path
1303 always routes the router's top-K experts regardless of this setting.
1304 scoreThreshold is only used when selectionMode is Threshold.
1306 @param selection_mode int. Expert selection mode index (calibration only; inference is always top-K):<br>
1307 0: TopK - Calibrate only the router's top-K experts per token (matches inference routing).<br>
1308 1: All - Calibrate every expert on the full sequence.<br>
1309 2: Threshold - Calibrate all experts whose routing score exceeds scoreThreshold.<br>
1310 @param score_threshold float. Routing score threshold used when selectionMode is Threshold (calibration only)
1313 model_config = ConfigDict(
1314 populate_by_name=
True,
1318 selection_mode_list: List[str] = Field(
1319 default=[
"TopK",
"All",
"Threshold"], alias=
"selectionModeList"
1321 selection_mode: int = Field(default=0, alias=
"selectionMode")
1323 score_threshold: float = Field(
1325 alias=
"scoreThreshold",
1326 description=
"Routing score threshold used when selectionMode is Threshold (calibration only)",
1331 """Return a copy with updated fields."""
1332 return self.model_copy(update=kwargs)
1337 @brief Configuration for equivalent transformation techniques
1339 @details Defines parameters for various equivalent transformation methods including
1340 NormConv, QK smoothing, and rotation matrices for improved quantization.
1342 @param seed int. Random seed for transformation
1343 @param apply_hadamard_rotation_matrix bool. Apply Hadamard rotation matrix
1344 @param norm_conv NormConv. NormConv equivalent transformation
1345 @param qk Qk. QK smoothing transformation
1346 @param ud Ud. UD transformation
1347 @param vo Vo. VO transformation
1348 @param feed_forward_multi_lut FeedForwardMultiLut. Feed-forward multi-LUT transformation
1349 @param spin_r1 SpinR1. SpinR1 rotation transformation
1350 @param head_out_ch_rotation HeadOutChRotation. Head output channel rotation transformation
1351 @param in_rotation InRotation. Input rotation transformation
1352 @param spin_r2 SpinR2. SpinR2 rotation transformation
1353 @param qk_rotation QkRotation. QK rotation transformation
1354 @param flatten_quant FlattenQuant. Flatten quantization transformation
1355 @param optimize_ffn OptimizeFfn. FFN optimization
1358 model_config = ConfigDict(
1359 populate_by_name=
True,
1363 seed: int = Field(default=0, description=
"Random seed for transformation")
1364 apply_hadamard_rotation_matrix: bool = Field(
1366 alias=
"applyHadamardRotationMatrix",
1367 description=
"Apply Hadamard rotation matrix",
1371 model_config = ConfigDict(populate_by_name=
True)
1374 @brief NormConv equivalent transformation
1376 @param apply bool. Apply NormConv transformation
1377 @param learn bool. Learn transformation parameters
1378 @param smoothing_factor float. Smoothing factor
1379 @param min_gamma float. Minimum gamma value
1380 @param max_gamma float. Maximum gamma value
1382 apply: bool = Field(default=
False, description=
"Apply NormConv transformation")
1383 learn: bool = Field(
1384 default=
False, description=
"Learn transformation parameters"
1386 smoothing_factor: float = Field(
1387 default=0.5, alias=
"smoothingFactor", description=
"Smoothing factor"
1389 min_gamma: float = Field(
1390 default=0.0001, alias=
"minGamma", description=
"Minimum gamma value"
1392 max_gamma: float = Field(
1393 default=10000.0, alias=
"maxGamma", description=
"Maximum gamma value"
1396 norm_conv: NormConv = Field(default_factory=NormConv, alias=
"NormConv")
1399 model_config = ConfigDict(populate_by_name=
True)
1402 @brief QK smoothing transformation
1404 @param apply bool. Apply QK transformation
1405 @param smoothing_factor float. Smoothing factor
1406 @param min_gamma float. Minimum gamma value
1407 @param max_gamma float. Maximum gamma value
1409 apply: bool = Field(default=
False, description=
"Apply QK transformation")
1410 smoothing_factor: float = Field(
1411 default=0.5, alias=
"smoothingFactor", description=
"Smoothing factor"
1413 min_gamma: float = Field(
1414 default=0.0001, alias=
"minGamma", description=
"Minimum gamma value"
1416 max_gamma: float = Field(
1417 default=10000.0, alias=
"maxGamma", description=
"Maximum gamma value"
1420 qk: Qk = Field(default_factory=Qk, alias=
"QK")
1423 model_config = ConfigDict(populate_by_name=
True)
1426 @brief UD transformation
1428 @param apply bool. Apply UD transformation
1429 @param learn bool. Learn transformation parameters
1430 @param smoothing_factor float. Smoothing factor
1431 @param min_gamma float. Minimum gamma value
1432 @param max_gamma float. Maximum gamma value
1434 apply: bool = Field(default=
False, description=
"Apply UD transformation")
1435 learn: bool = Field(
1436 default=
False, description=
"Learn transformation parameters"
1438 smoothing_factor: float = Field(
1439 default=0.5, alias=
"smoothingFactor", description=
"Smoothing factor"
1441 min_gamma: float = Field(
1442 default=0.0001, alias=
"minGamma", description=
"Minimum gamma value"
1444 max_gamma: float = Field(
1445 default=10000.0, alias=
"maxGamma", description=
"Maximum gamma value"
1448 ud: Ud = Field(default_factory=Ud, alias=
"UD")
1451 model_config = ConfigDict(populate_by_name=
True)
1454 @brief VO transformation
1456 @param apply bool. Apply VO transformation
1457 @param smoothing_factor float. Smoothing factor
1458 @param min_gamma float. Minimum gamma value
1459 @param max_gamma float. Maximum gamma value
1461 apply: bool = Field(default=
False, description=
"Apply VO transformation")
1462 smoothing_factor: float = Field(
1463 default=0.5, alias=
"smoothingFactor", description=
"Smoothing factor"
1465 min_gamma: float = Field(
1466 default=0.0001, alias=
"minGamma", description=
"Minimum gamma value"
1468 max_gamma: float = Field(
1469 default=10000.0, alias=
"maxGamma", description=
"Maximum gamma value"
1472 vo: Vo = Field(default_factory=Vo, alias=
"VO")
1476 @brief Feed-forward multi-LUT transformation
1478 @param apply bool. Apply feed-forward multi-LUT transformation
1479 @param breakpoints List[float]. Breakpoints for multi-LUT
1482 apply: bool = Field(
1483 default=
False, description=
"Apply feed-forward multi-LUT transformation"
1485 breakpoints: List[float] = Field(
1486 default=[-8.0, -4.0, 0], description=
"Breakpoints for multi-LUT"
1489 feed_forward_multi_lut: FeedForwardMultiLut = Field(
1490 default_factory=FeedForwardMultiLut, alias=
"FeedForwardMultiLUT"
1494 model_config = ConfigDict(populate_by_name=
True)
1497 @brief SpinR1 rotation transformation
1499 @param apply bool. Apply SpinR1 transformation
1500 @param matrix_path str. Path to rotation matrix file
1502 apply: bool = Field(default=
False, description=
"Apply SpinR1 transformation")
1503 matrix_path: str = Field(
1504 default=
"", alias=
"matrixPath", description=
"Path to rotation matrix file"
1507 spin_r1: SpinR1 = Field(default_factory=SpinR1, alias=
"SpinR1")
1510 model_config = ConfigDict(populate_by_name=
True)
1513 @brief Head output channel rotation transformation
1515 @param apply bool. Apply head output channel rotation
1516 @param matrix_path str. Path to rotation matrix file
1518 apply: bool = Field(
1519 default=
False, description=
"Apply head output channel rotation"
1521 matrix_path: str = Field(
1522 default=
"", alias=
"matrixPath", description=
"Path to rotation matrix file"
1525 head_out_ch_rotation: HeadOutChRotation = Field(
1526 default_factory=HeadOutChRotation, alias=
"HeadOutChRotation"
1530 model_config = ConfigDict(populate_by_name=
True)
1533 @brief Input rotation transformation
1535 @param apply bool. Apply input rotation
1536 @param matrix_path str. Path to rotation matrix file
1537 @param input_names List[str]. Names of the input layers to rotate
1539 apply: bool = Field(default=
False, description=
"Apply input rotation")
1540 matrix_path: str = Field(
1541 default=
"", alias=
"matrixPath", description=
"Path to rotation matrix file"
1543 input_names: List[str] = Field(
1546 description=
"Names of the input layers to rotate",
1549 in_rotation: InRotation = Field(default_factory=InRotation, alias=
"InRotation")
1552 model_config = ConfigDict(populate_by_name=
True)
1555 @brief SpinR2 rotation transformation
1557 @param apply bool. Apply SpinR2 transformation
1558 @param learn bool. Learn rotation matrix
1559 @param matrix_path str. Path to rotation matrix file
1561 apply: bool = Field(default=
False, description=
"Apply SpinR2 transformation")
1562 learn: bool = Field(default=
False, description=
"Learn rotation matrix")
1563 matrix_path: str = Field(
1564 default=
"", alias=
"matrixPath", description=
"Path to rotation matrix file"
1567 spin_r2: SpinR2 = Field(default_factory=SpinR2, alias=
"SpinR2")
1570 model_config = ConfigDict(populate_by_name=
True)
1573 @brief QK rotation transformation
1575 @param apply bool. Apply QK rotation transformation
1576 @param matrix_path str. Path to rotation matrix file
1578 apply: bool = Field(
1579 default=
False, description=
"Apply QK rotation transformation"
1581 matrix_path: str = Field(
1582 default=
"", alias=
"matrixPath", description=
"Path to rotation matrix file"
1585 qk_rotation: QkRotation = Field(default_factory=QkRotation, alias=
"QKRotation")
1588 model_config = ConfigDict(populate_by_name=
True)
1591 @brief Flatten quantization transformation
1593 @param apply bool. Apply flatten quantization
1594 @param learn bool. Learn flattening parameters
1595 @param apply_threshold float. Threshold for applying flatten quantization
1596 @param max_overhead float. Maximum overhead allowed for flattening
1598 apply: bool = Field(default=
False, description=
"Apply flatten quantization")
1599 learn: bool = Field(default=
False, description=
"Learn flattening parameters")
1600 apply_threshold: float = Field(
1602 alias=
"applyThreshold",
1603 description=
"Threshold for applying flatten quantization",
1605 max_overhead: float = Field(
1607 alias=
"maxOverhead",
1608 description=
"Maximum overhead allowed for flattening",
1611 flatten_quant: FlattenQuant = Field(
1612 default_factory=FlattenQuant, alias=
"FlattenQuant"
1616 model_config = ConfigDict(populate_by_name=
True)
1619 @brief FFN optimization
1621 @param apply bool. Apply FFN optimization
1622 @param ch_per_ffn int. Optimize FFN split (-1 for auto)
1624 apply: bool = Field(default=
False, description=
"Apply FFN optimization")
1625 ch_per_ffn: int = Field(
1626 default=-1, alias=
"chPerFFN", description=
"Optimize FFN split (-1 for auto)"
1629 optimize_ffn: OptimizeFfn = Field(default_factory=OptimizeFfn, alias=
"OptimizeFFN")
1632 """Return a copy with updated fields."""
1633 return self.model_copy(update=kwargs)
1638 @brief Configuration for weight scale search
1640 @details Defines which transformer components should have their weight scales
1641 searched for optimal quantization.
1643 @param apply bool. If true, apply weight scale search
1644 @param transformer Transformer. Transformer components for weight scale search
1647 model_config = ConfigDict(
1648 populate_by_name=
True,
1652 apply: bool = Field(default=
False, description=
"If true, apply weight scale search")
1656 @brief Transformer components for weight scale search
1658 @param query bool. Search weight scale for query
1659 @param key bool. Search weight scale for key
1660 @param value bool. Search weight scale for value
1661 @param out bool. Search weight scale for output
1662 @param ffn bool. Search weight scale for FFN
1665 query: bool = Field(default=
False, description=
"Search weight scale for query")
1666 key: bool = Field(default=
False, description=
"Search weight scale for key")
1667 value: bool = Field(default=
False, description=
"Search weight scale for value")
1668 out: bool = Field(default=
False, description=
"Search weight scale for output")
1669 ffn: bool = Field(default=
False, description=
"Search weight scale for FFN")
1671 transformer: Transformer = Field(default_factory=Transformer, alias=
"transformer")
1674 """Return a copy with updated fields."""
1675 return self.model_copy(update=kwargs)
1680 @brief QAT activation scale loading configuration
1682 @details Loads pre-trained activation scales from safetensors files and applies them
1683 to specified layers before scale/zeropoint computation.
1684 NOTE: entries is stored as raw JSON (type: list) because the generator does
1685 not support list[CustomStruct]. The schema records the field shape; parsing
1686 is done manually in applyQATLoadScaleQuantType / applyQATLoadScales.
1688 @param apply bool. If true, load and apply QAT scales from safetensors files
1689 @param entries List. List of file entries. Each entry is a dict:
1690 { path: str, scales: [ { key: str, layers: [str] } ] }
1691 path: safetensors file path; key: tensor name in the file;
1692 layers: layer names whose activation scale will be overridden.
1696 model_config = ConfigDict(
1697 populate_by_name=
True,
1701 apply: bool = Field(
1703 description=
"If true, load and apply QAT scales from safetensors files",
1705 entries: Any = Field(
1707 description=
"List of file entries. Each entry is a dict: { path: str, scales: [ { key: str, layers: [str] } ] } path: safetensors file path; key: tensor name in the file; layers: layer names whose activation scale will be overridden.",
1711 """Return a copy with updated fields."""
1712 return self.model_copy(update=kwargs)
1717 @brief Runtime options for compilation
1719 @details Contains runtime-specific settings like version info and cache options.
1721 @param version str. Compiler version string (e.g., 0.0.0)
1722 @param deterministic_algorithms bool. Run LUT grouping, the only compile stage found to vary between runs on one GPU, with PyTorch deterministic algorithms; false gives a faster grouping whose result can differ between runs
1725 model_config = ConfigDict(
1726 populate_by_name=
True,
1730 version: str = Field(
1731 default=
"0.0.0", description=
"Compiler version string (e.g., 0.0.0)"
1733 deterministic_algorithms: bool = Field(
1735 alias=
"deterministicAlgorithms",
1736 description=
"Run LUT grouping, the only compile stage found to vary between runs on one GPU, with PyTorch deterministic algorithms; false gives a faster grouping whose result can differ between runs",
1740 """Return a copy with updated fields."""
1741 return self.model_copy(update=kwargs)
1746 @brief Sample data generation and saving configuration
1748 @param apply bool. Enable sample data saving
1749 @param mode str. Inference mode: infer (standard) or inferWithCache (LLM cache models)
1750 @param batch_size int. Number of inference batches to generate
1751 @param batch_seq_lens List. Per-batch step-wise sequence lengths for inferWithCache mode. e.g. [[80, 1], [240, 10]]
1752 @param save_folder str. Output folder for sample data
1753 @param dtype str. Data type for saved samples: float or int8
1756 model_config = ConfigDict(
1757 populate_by_name=
True,
1761 apply: bool = Field(default=
False, description=
"Enable sample data saving")
1764 description=
"Inference mode: infer (standard) or inferWithCache (LLM cache models)",
1766 batch_size: int = Field(
1769 description=
"Number of inference batches to generate",
1771 batch_seq_lens: Any = Field(
1773 alias=
"batchSeqLens",
1774 description=
"Per-batch step-wise sequence lengths for inferWithCache mode. e.g. [[80, 1], [240, 10]]",
1776 save_folder: str = Field(
1777 default=
"sampleInout",
1779 description=
"Output folder for sample data",
1782 default=
"float", description=
"Data type for saved samples: float or int8"
1786 """Return a copy with updated fields."""
1787 return self.model_copy(update=kwargs)
1792 @brief Promote named intermediate layers to additional model outputs
1794 @details Applied after quantization, so the rest of the bundle is unchanged. Names
1795 are matched against the post-fusion graph, and an unknown one is an error.
1797 @param apply bool. Promote the named layers. Off by default: promotion is an explicit opt-in.
1798 @param layers List[str]. Layer names to promote; one that is already an output is left alone
1801 model_config = ConfigDict(
1802 populate_by_name=
True,
1806 apply: bool = Field(
1808 description=
"Promote the named layers. Off by default: promotion is an explicit opt-in.",
1810 layers: List[str] = Field(
1812 description=
"Layer names to promote; one that is already an output is left alone",
1816 """Return a copy with updated fields."""
1817 return self.model_copy(update=kwargs)
1820class GroupWiseConfig(BaseModel):
1822 @brief Group-wise streaming quantization configuration for large LLMs
1824 @details Configuration for the group-wise (streaming) quantization pipeline used for
1825 LLMs that do not fit fully on GPU. Partitions the model into groups of
1826 transformer blocks and quantizes each group independently.
1828 @param apply bool. Enable group-wise streaming quantization pipeline
1829 @param group_size int. Group size in number of transformer blocks (0 = auto)
1830 @param gpu_budget_gb float. GPU memory budget in GiB for group-wise execution (0 = auto-detect)
1831 @param gpu_safety_margin_gb float. Safety margin in GiB subtracted from detected GPU budget
1832 @param cache_dir str. Directory for group-wise activation/state cache (empty = system temp)
1833 @param keep_cache bool. Retain group-wise cache after run completes (for debugging)
1834 @param partition_policy str. Partitioning policy (e.g. transformer_block, moe_expert_subgroup)
1835 @param expert_groups List[List[int]]. MoE fallback partitioner: list of expert-index lists (List[List[int]]).
1836 @param retain_topology_weights bool. Keep inflated FP weights across groups (false = release after each group for tight memory budgets).
1837 @param checkpoint bool. Save per-group checkpoint so quantization can resume from the last completed group after a crash.
1840 model_config = ConfigDict(
1841 populate_by_name=
True,
1845 apply: bool = Field(
1846 default=
False, description=
"Enable group-wise streaming quantization pipeline"
1848 group_size: int = Field(
1851 description=
"Group size in number of transformer blocks (0 = auto)",
1853 gpu_budget_gb: float = Field(
1855 alias=
"gpuBudgetGb",
1856 description=
"GPU memory budget in GiB for group-wise execution (0 = auto-detect)",
1858 gpu_safety_margin_gb: float = Field(
1860 alias=
"gpuSafetyMarginGb",
1861 description=
"Safety margin in GiB subtracted from detected GPU budget",
1863 cache_dir: str = Field(
1866 description=
"Directory for group-wise activation/state cache (empty = system temp)",
1868 keep_cache: bool = Field(
1871 description=
"Retain group-wise cache after run completes (for debugging)",
1873 partition_policy: str = Field(
1874 default=
"transformer_block",
1875 alias=
"partitionPolicy",
1876 description=
"Partitioning policy (e.g. transformer_block, moe_expert_subgroup)",
1878 expert_groups: Any = Field(
1880 alias=
"expertGroups",
1881 description=
"MoE fallback partitioner: list of expert-index lists (List[List[int]]).",
1883 retain_topology_weights: bool = Field(
1885 alias=
"retainTopologyWeights",
1886 description=
"Keep inflated FP weights across groups (false = release after each group for tight memory budgets).",
1888 checkpoint: bool = Field(
1890 description=
"Save per-group checkpoint so quantization can resume from the last completed group after a crash.",
1894 """Return a copy with updated fields."""
1895 return self.model_copy(update=kwargs)
1899 """Unified compilation configuration for Mobilint MXQ compilation."""
1901 model_config = ConfigDict(
1902 populate_by_name=
True,
1906 model_paths: List[str] = Field(
1907 default=[], alias=
"modelPaths", description=
"Paths to model files"
1909 calib_data_path: List[str] = Field(
1910 default=[], alias=
"calibDataPaths", description=
"Paths to calibration datasets"
1912 save_paths: List[str] = Field(
1913 default=[
"./tmp.mxq"],
1915 description=
"Output MXQ filename/paths",
1917 use_random_calib: bool = Field(
1918 default=
False, alias=
"useRandomCalib", description=
"Use random calibration"
1920 inference_scheme: str = Field(
1921 default=
"single", alias=
"inferenceScheme", description=
"NPU inference scheme"
1923 cpu_offload: bool = Field(
1926 description=
"Enable CPU offload for unsupported operators",
1928 force_npu_input_reposition: bool = Field(
1930 alias=
"forceNpuInputReposition",
1931 description=
"Force input reposition operations to run on NPU instead of CPU",
1933 force_npu_output_reposition: bool = Field(
1935 alias=
"forceNpuOutputReposition",
1936 description=
"Force output reposition operations to run on NPU instead of CPU",
1938 optimize_option: int = Field(
1940 alias=
"optimizeOption",
1941 description=
"Compiler optimization selector",
1944 buffer_mode: int = Field(
1945 default=1, alias=
"bufferMode", description=
"Buffer serialization mode"
1947 device: str = Field(default=
"gpu", description=
"Device for computation")
1948 dtype: str = Field(default=
"float", description=
"Data type for computation")
1949 debug: bool = Field(default=
False, description=
"Enable debug mode")
1950 trace: bool = Field(default=
False, description=
"Enable trace mode")
1951 image_channels: int = Field(
1953 alias=
"imageChannels",
1954 description=
"Number of image channels (0 for auto-detect)",
1956 config_version: str = Field(
1957 default=
"1.0.0", alias=
"configVersion", description=
"Config schema version"
1959 split_blocks: List[int] = Field(
1961 alias=
"splitBlocks",
1962 description=
"Multi-MXQ split points by transformer block index",
1964 split_parts: int = Field(
1967 description=
"Evenly split transformer blocks into N MXQ parts",
1970 uint8_input: Uint8InputConfig = Field(
1971 default_factory=Uint8InputConfig, alias=
"uint8Input"
1973 preprocessing: PreprocessingConfig = Field(default_factory=PreprocessingConfig)
1974 resource_management: ResourceManagementConfig = Field(
1975 default_factory=ResourceManagementConfig, alias=
"resourceManagement"
1977 calibration: CalibrationConfig = Field(default_factory=CalibrationConfig)
1978 bit: BitConfig = Field(default_factory=BitConfig)
1979 hessian_quant: HessianQuantConfig = Field(
1980 default_factory=HessianQuantConfig, alias=
"hessianQuant"
1982 bias_correction: BiasCorrectionConfig = Field(
1983 default_factory=BiasCorrectionConfig, alias=
"biasCorrection"
1985 mod: ModConfig = Field(default_factory=ModConfig)
1986 llm: LlmConfig = Field(default_factory=LlmConfig)
1987 moe: MoeConfig = Field(default_factory=MoeConfig)
1988 equivalent_transformation: EquivalentTransformationConfig = Field(
1989 default_factory=EquivalentTransformationConfig, alias=
"equivalentTransformation"
1991 search_weight_scale: SearchWeightScaleConfig = Field(
1992 default_factory=SearchWeightScaleConfig, alias=
"searchWeightScale"
1994 load_scale: LoadScaleConfig = Field(
1995 default_factory=LoadScaleConfig, alias=
"loadScale"
1997 runtime_options: RuntimeOptions = Field(
1998 default_factory=RuntimeOptions, alias=
"runtimeOptions"
2000 save_sample: SaveSampleConfig = Field(
2001 default_factory=SaveSampleConfig, alias=
"saveSample"
2003 extra_output: ExtraOutputConfig = Field(
2004 default_factory=ExtraOutputConfig, alias=
"extraOutput"
2006 group_wise: GroupWiseConfig = Field(
2007 default_factory=GroupWiseConfig, alias=
"groupWise"
2011 """Return a copy with uint8_input settings enabled."""
2012 data = {
"apply":
True, **kwargs}
2013 new_cfg = self.
uint8_input.model_copy(update=data)
2014 return self.model_copy(update={
"uint8_input": new_cfg})
2017 """Return a copy with preprocessing settings enabled."""
2018 data = {
"apply":
True, **kwargs}
2020 return self.model_copy(update={
"preprocessing": new_cfg})
2023 """Return a copy with hessian_quant settings enabled."""
2024 data = {
"apply":
True, **kwargs}
2026 return self.model_copy(update={
"hessian_quant": new_cfg})
2029 """Return a copy with bias_correction settings enabled."""
2030 data = {
"apply":
True, **kwargs}
2032 return self.model_copy(update={
"bias_correction": new_cfg})
2035 """Return a copy with mod settings enabled."""
2036 data = {
"apply":
True, **kwargs}
2037 new_cfg = self.
mod.model_copy(update=data)
2038 return self.model_copy(update={
"mod": new_cfg})
2041 """Return a copy with llm settings enabled."""
2042 data = {
"apply":
True, **kwargs}
2043 new_cfg = self.
llm.model_copy(update=data)
2044 return self.model_copy(update={
"llm": new_cfg})
2047 """Return a copy with search_weight_scale settings enabled."""
2048 data = {
"apply":
True, **kwargs}
2050 return self.model_copy(update={
"search_weight_scale": new_cfg})
2053 """Return a copy with load_scale settings enabled."""
2054 data = {
"apply":
True, **kwargs}
2055 new_cfg = self.
load_scale.model_copy(update=data)
2056 return self.model_copy(update={
"load_scale": new_cfg})
2059 """Return a copy with save_sample settings enabled."""
2060 data = {
"apply":
True, **kwargs}
2061 new_cfg = self.
save_sample.model_copy(update=data)
2062 return self.model_copy(update={
"save_sample": new_cfg})
2065 """Return a copy with extra_output settings enabled."""
2066 data = {
"apply":
True, **kwargs}
2068 return self.model_copy(update={
"extra_output": new_cfg})
2070 def with_group_wise(self, **kwargs) -> "CompileConfig":
2071 """Return a copy with group_wise settings enabled."""
2072 data = {
"apply":
True, **kwargs}
2073 new_cfg = self.
group_wise.model_copy(update=data)
2074 return self.model_copy(update={
"group_wise": new_cfg})
2077 def from_file(cls, path: Union[str, Path]) ->
"CompileConfig":
2078 """Load config from YAML or JSON file."""
2080 with open(path)
as f:
2081 if path.suffix
in (
".yaml",
".yml"):
2082 data = yaml.safe_load(f)
2086 return cls.model_validate(data)
2090 """Flatten grouped JSON keys (e.g. quantization.calibration) to flat structure."""
2092 if "quantization" in data:
2093 group = data.pop(
"quantization")
2094 if "calibration" in group:
2095 data[
"calibration"] = group[
"calibration"]
2097 data[
"bit"] = group[
"bit"]
2098 if "advancedQuantization" in data:
2099 group = data.pop(
"advancedQuantization")
2100 if "hessianQuant" in group:
2101 data[
"hessianQuant"] = group[
"hessianQuant"]
2102 if "biasCorrection" in group:
2103 data[
"biasCorrection"] = group[
"biasCorrection"]
2105 data[
"mod"] = group[
"mod"]
2106 if "EquivalentTransformation" in group:
2107 data[
"equivalentTransformation"] = group[
"EquivalentTransformation"]
2108 if "searchWeightScale" in group:
2109 data[
"searchWeightScale"] = group[
"searchWeightScale"]
2110 if "loadScale" in group:
2111 data[
"loadScale"] = group[
"loadScale"]
2116 """Load config from a preset."""
2117 from .presets
import get_preset
2119 return get_preset(name)
2123 """Group flat keys back into nested JSON structure (inverse of _flatten_grouped_json)."""
2125 quantization_group = {}
2126 if "calibration" in data:
2127 quantization_group[
"calibration"] = data.pop(
"calibration")
2129 quantization_group[
"bit"] = data.pop(
"bit")
2130 if quantization_group:
2131 data[
"quantization"] = quantization_group
2132 advancedQuantization_group = {}
2133 if "hessianQuant" in data:
2134 advancedQuantization_group[
"hessianQuant"] = data.pop(
"hessianQuant")
2135 if "biasCorrection" in data:
2136 advancedQuantization_group[
"biasCorrection"] = data.pop(
"biasCorrection")
2138 advancedQuantization_group[
"mod"] = data.pop(
"mod")
2139 if "equivalentTransformation" in data:
2140 advancedQuantization_group[
"EquivalentTransformation"] = data.pop(
2141 "equivalentTransformation"
2143 if "searchWeightScale" in data:
2144 advancedQuantization_group[
"searchWeightScale"] = data.pop(
2147 if "loadScale" in data:
2148 advancedQuantization_group[
"loadScale"] = data.pop(
"loadScale")
2149 if advancedQuantization_group:
2150 data[
"advancedQuantization"] = advancedQuantization_group
2153 def to_file(self, path: Union[str, Path]) ->
None:
2154 """Save config to YAML or JSON file."""
2156 data = self.model_dump(by_alias=
True, exclude_none=
True)
2158 with open(path,
"w")
as f:
2159 if path.suffix
in (
".yaml",
".yml"):
2160 yaml.dump(data, f, default_flow_style=
False)
2162 json.dump(data, f, indent=2)
Configuration for in-schedule integer bias correction.
"BiasCorrectionConfig" with_updates(self, **kwargs)
Return a copy with updated fields.
Configuration for bit precision.
"BitConfig" with_updates(self, **kwargs)
Return a copy with updated fields.
Configuration for calibration during quantization.
"CalibrationConfig" with_updates(self, **kwargs)
Return a copy with updated fields.
Unified compilation configuration for Mobilint MXQ compilation.
"CompileConfig" with_extra_output(self, **kwargs)
Return a copy with extra_output settings enabled.
"CompileConfig" with_bias_correction(self, **kwargs)
Return a copy with bias_correction settings enabled.
"CompileConfig" with_llm(self, **kwargs)
Return a copy with llm settings enabled.
PreprocessingConfig preprocessing
"CompileConfig" from_preset(cls, str name)
Load config from a preset.
"CompileConfig" with_search_weight_scale(self, **kwargs)
Return a copy with search_weight_scale settings enabled.
LoadScaleConfig load_scale
Uint8InputConfig uint8_input
SearchWeightScaleConfig search_weight_scale
HessianQuantConfig hessian_quant
"CompileConfig" with_load_scale(self, **kwargs)
Return a copy with load_scale settings enabled.
BiasCorrectionConfig bias_correction
SaveSampleConfig save_sample
"CompileConfig" with_preprocessing(self, **kwargs)
Return a copy with preprocessing settings enabled.
dict _flatten_grouped_json(dict data)
Flatten grouped JSON keys (e.g.
"CompileConfig" with_mod(self, **kwargs)
Return a copy with mod settings enabled.
"CompileConfig" with_hessian_quant(self, **kwargs)
Return a copy with hessian_quant settings enabled.
dict _group_to_json(dict data)
Group flat keys back into nested JSON structure (inverse of _flatten_grouped_json).
"CompileConfig" with_save_sample(self, **kwargs)
Return a copy with save_sample settings enabled.
"CompileConfig" from_file(cls, Union[str, Path] path)
Load config from YAML or JSON file.
None to_file(self, Union[str, Path] path)
Save config to YAML or JSON file.
"CompileConfig" with_uint8_input(self, **kwargs)
Return a copy with uint8_input settings enabled.
ExtraOutputConfig extra_output
Configuration for HessianQuant algorithm.
"HessianQuantConfig" with_updates(self, **kwargs)
Return a copy with updated fields.
Configuration for Large Language Model (LLM) compilation.
"LlmConfig" with_updates(self, **kwargs)
Return a copy with updated fields.
"LoadScaleConfig" with_updates(self, **kwargs)
Return a copy with updated fields.
Configuration for Minimum Output Difference algorithm.
"ModConfig" with_updates(self, **kwargs)
Return a copy with updated fields.
Sparse MoE expert-selection configuration (calibration only)
"MoeConfig" with_updates(self, **kwargs)
Return a copy with updated fields.
Configuration for input preprocessing pipeline.
"PreprocessingConfig" with_updates(self, **kwargs)
Return a copy with updated fields.
Weight memory management configuration.
Configuration for resource management during model compilation.
"ResourceManagementConfig" with_updates(self, **kwargs)
Return a copy with updated fields.
Runtime options for compilation.
"RuntimeOptions" with_updates(self, **kwargs)
Return a copy with updated fields.
Sample data generation and saving configuration.
"SaveSampleConfig" with_updates(self, **kwargs)
Return a copy with updated fields.
Configuration for weight scale search.
"SearchWeightScaleConfig" with_updates(self, **kwargs)
Return a copy with updated fields.